diff --git a/content/.metadata.json b/content/.metadata.json index f836cd24e..0b541f71e 100644 --- a/content/.metadata.json +++ b/content/.metadata.json @@ -1,7 +1,7 @@ { "metadata": { "version": "2.0", - "fetch_date": "2026-07-20T17:16:17.039299Z", + "fetch_date": "2026-07-21T04:08:55.835714Z", "section": "all" }, "items": [ @@ -534,8 +534,8 @@ "url": "https://platform.claude.com/docs/en/build-with-claude/claude-platform-on-aws", "status": "success", "path": "en/build-with-claude/claude-platform-on-aws.md", - "sha256": "313f11ace3f0e635498589cf5e4144a1e879d022f03134652ec9250906b5b30c", - "size": 73698 + "sha256": "9dba7a30358c3913f5c8b5541ecb2557f33d88c388e40b276f423e8125926690", + "size": 73179 }, { "url": "https://platform.claude.com/docs/en/build-with-claude/claude-on-vertex-ai", @@ -877,8 +877,8 @@ "url": "https://platform.claude.com/docs/en/manage-claude/data-residency", "status": "success", "path": "en/manage-claude/data-residency.md", - "sha256": "75e5b931541fd0ba151e4688c3308f4ba5ad0aa77ea097afb60dd3ec8dac8aac", - "size": 10402 + "sha256": "8d40d43e76ee848525d8ff4fb90837a2c257fdec8da67e89bc0b100c34a5f9d9", + "size": 13023 }, { "url": "https://platform.claude.com/docs/en/manage-claude/api-and-data-retention", @@ -898,8 +898,8 @@ "url": "https://platform.claude.com/docs/en/manage-claude/cmek", "status": "success", "path": "en/manage-claude/cmek.md", - "sha256": "017a53ddc38aa9e5db1a14bee2484388b4fa6dfd8e4e8ad97ad428b8838ef57c", - "size": 11963 + "sha256": "7e88a049d315e0ba503de11047376c5262f55594f581b831318cca2d19f646fb", + "size": 11901 }, { "url": "https://platform.claude.com/docs/en/manage-claude/cmek-aws-kms", @@ -978,6 +978,405 @@ "sha256": "4322dfefec2d97b7ea49d5727cd9d0cc498698c19a139999a743bf412c7f07b3", "size": 9114 }, + { + "url": "https://platform.claude.com/docs/en/about-claude/use-case-guides/overview", + "status": "success", + "path": "en/about-claude/use-case-guides/overview.md", + "sha256": "e1065cacc6ed14df80a82f338a211a63c30b0300c07854467aa9ec408d16797c", + "size": 1263 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/use-case-guides/ticket-routing", + "status": "success", + "path": "en/about-claude/use-case-guides/ticket-routing.md", + "sha256": "61f9e0b4caf7f6ab7e20c89da30aaf06305e2582812bf0eadb155264ccc79884", + "size": 29561 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/use-case-guides/customer-support-chat", + "status": "success", + "path": "en/about-claude/use-case-guides/customer-support-chat.md", + "sha256": "7fb23880f46d37c18474d0fd6e263783fd30a098c277b41576197c75bf10da6f", + "size": 34324 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/use-case-guides/content-moderation", + "status": "success", + "path": "en/about-claude/use-case-guides/content-moderation.md", + "sha256": "f15685d0cdd049c6c8ec56de2b88f3746152d22d56b6af5a61ca06ffc13c642f", + "size": 25329 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/use-case-guides/legal-summarization", + "status": "success", + "path": "en/about-claude/use-case-guides/legal-summarization.md", + "sha256": "ae3c6e5caaf590854875e6737a8a3f3c71a5c8fc3836fee3cdc7cdcd0b82a501", + "size": 21606 + }, + { + "url": "https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/overview", + "status": "success", + "path": "en/build-with-claude/prompt-engineering/overview.md", + "sha256": "aa097f81d8de347dc3287f591464d36511e582d3b1a48872419c0cfa88f0ebdd", + "size": 2705 + }, + { + "url": "https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices", + "status": "success", + "path": "en/build-with-claude/prompt-engineering/claude-prompting-best-practices.md", + "sha256": "30eec771316c9a604fdedbb1b660ed7a7a7cb84dfa11174a106c7f2d86aa0697", + "size": 56683 + }, + { + "url": "https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/prompting-claude-fable-5", + "status": "success", + "path": "en/build-with-claude/prompt-engineering/prompting-claude-fable-5.md", + "sha256": "61f3b61423a1c953edbd242e2e9e639a213397a6eb3a3fdc53da7232d4aac9f3", + "size": 18421 + }, + { + "url": "https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8", + "status": "success", + "path": "en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8.md", + "sha256": "12c9d1b216b09eb9441cefffe1898297305bac697a657212b0cae490c53890b3", + "size": 15679 + }, + { + "url": "https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/prompting-claude-sonnet-5", + "status": "success", + "path": "en/build-with-claude/prompt-engineering/prompting-claude-sonnet-5.md", + "sha256": "21910f8721d95d94110b30805573ececeeb36e1f3a0bb15aae1a83cda44d1b58", + "size": 15873 + }, + { + "url": "https://platform.claude.com/docs/en/build-with-claude/prompt-engineering/prompting-tools", + "status": "success", + "path": "en/build-with-claude/prompt-engineering/prompting-tools.md", + "sha256": "3d2e08806e114a5858a64027749815fb25e9473e96f70169d59423bd77959a91", + "size": 10636 + }, + { + "url": "https://platform.claude.com/docs/en/test-and-evaluate/develop-tests", + "status": "success", + "path": "en/test-and-evaluate/develop-tests.md", + "sha256": "5ff3afee747f992b1ff2475ed6c7340e42d459adddfb8ab371e30b3f2edb9198", + "size": 137073 + }, + { + "url": "https://platform.claude.com/docs/en/test-and-evaluate/eval-tool", + "status": "success", + "path": "en/test-and-evaluate/eval-tool.md", + "sha256": "7c5ebadc8977cdd759b8075f9cd280b9875e9cace3258876e478b9c5ccdc0be2", + "size": 5470 + }, + { + "url": "https://platform.claude.com/docs/en/test-and-evaluate/strengthen-guardrails/reduce-latency", + "status": "success", + "path": "en/test-and-evaluate/strengthen-guardrails/reduce-latency.md", + "sha256": "edef1e95694fb0da454b007ae621d66389c10517ec8dc517cd3f77a3429faa89", + "size": 9358 + }, + { + "url": "https://platform.claude.com/docs/en/test-and-evaluate/strengthen-guardrails/reduce-hallucinations", + "status": "success", + "path": "en/test-and-evaluate/strengthen-guardrails/reduce-hallucinations.md", + "sha256": "d14cde11ee5bb46df0e8ebab1acf754338db693b27dc26a028f7aad6fceb87fa", + "size": 6309 + }, + { + "url": "https://platform.claude.com/docs/en/test-and-evaluate/strengthen-guardrails/increase-consistency", + "status": "success", + "path": "en/test-and-evaluate/strengthen-guardrails/increase-consistency.md", + "sha256": "77943d1e35eee93dbd41773811fd5814a0552e8f0e747d7483c5d6698b0e0187", + "size": 29800 + }, + { + "url": "https://platform.claude.com/docs/en/test-and-evaluate/strengthen-guardrails/mitigate-jailbreaks", + "status": "success", + "path": "en/test-and-evaluate/strengthen-guardrails/mitigate-jailbreaks.md", + "sha256": "dc2b6d2684c3633e0a9805cddda5522f36cf6f2b08f99fc4068f97913ba91199", + "size": 16432 + }, + { + "url": "https://platform.claude.com/docs/en/test-and-evaluate/strengthen-guardrails/reduce-prompt-leak", + "status": "success", + "path": "en/test-and-evaluate/strengthen-guardrails/reduce-prompt-leak.md", + "sha256": "02800705b1e5a6dc99be395c194a3045209a00a8fbd6de095341438484fdaddd", + "size": 4210 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/glossary", + "status": "success", + "path": "en/about-claude/glossary.md", + "sha256": "b14f34de299b7f7d529437478c35077e05708aa25df88cd502b48559fd683871", + "size": 9006 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/models/overview", + "status": "success", + "path": "en/about-claude/models/overview.md", + "sha256": "82da5d9645a34bd159a9432f6161af7d8fa3efae8893adc8a94fdcae8baa6800", + "size": 27280 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/models/model-ids-and-versions", + "status": "success", + "path": "en/about-claude/models/model-ids-and-versions.md", + "sha256": "f0ea5b0e1c73692b2647e52e77a7c3bd28edad461149e16c63f1b2c11bbaaa50", + "size": 3774 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/models/choosing-a-model", + "status": "success", + "path": "en/about-claude/models/choosing-a-model.md", + "sha256": "3f19cdc89ee4e49a2ad3e93951909ed69a32ad7dc57aeccc35c4692c50aab8f7", + "size": 6711 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5", + "status": "success", + "path": "en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5.md", + "sha256": "b42118e8c6d787db55460dde549d739732400c5bc714051412389246a16e1839", + "size": 9301 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/models/whats-new-claude-4-8", + "status": "success", + "path": "en/about-claude/models/whats-new-claude-4-8.md", + "sha256": "74dfc471a1a01877327451e471f2c64e91b16a107b3e053ddee529d19e35ad10", + "size": 14595 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/models/whats-new-sonnet-5", + "status": "success", + "path": "en/about-claude/models/whats-new-sonnet-5.md", + "sha256": "e9838313d3e2564bd487883b4bd18624327dc009fe5b0a59487cbef444700cae", + "size": 9322 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/models/migration-guide", + "status": "success", + "path": "en/about-claude/models/migration-guide.md", + "sha256": "655ea2cc4c08b42af6f6c65a803f56082e35a71e2312946ea786569de8309945", + "size": 114599 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/model-deprecations", + "status": "success", + "path": "en/about-claude/model-deprecations.md", + "sha256": "b6b9b9565500e881ae8ae933a9d43488d3b9eb44f02b346905d7d4c3341000ff", + "size": 12643 + }, + { + "url": "https://platform.claude.com/docs/en/resources/overview", + "status": "success", + "path": "en/resources/overview.md", + "sha256": "68cdf54414b07f31800901f267c942a8560dcadaae75f8aeda015faa14a89884", + "size": 4503 + }, + { + "url": "https://platform.claude.com/docs/en/release-notes/system-prompts", + "status": "success", + "path": "en/release-notes/system-prompts.md", + "sha256": "940d92070074f90a03472f95d56fca9cbd4020deaf64428b9b3a4ce9a44aca2a", + "size": 448446 + }, + { + "url": "https://platform.claude.com/docs/en/about-claude/pricing", + "status": "success", + "path": "en/about-claude/pricing.md", + "sha256": "dd512bbde856d6b185e0c861c433be17bada2c4f3c31677c4540457c4fe81c9f", + "size": 36869 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/overview", + "status": "success", + "path": "en/cli-sdks-libraries/overview.md", + "sha256": "e792a88b576d37ff1020a809f029f7e2bfe787ca692de94aa78fa833145270ec", + "size": 3230 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/cli/quickstart", + "status": "success", + "path": "en/cli-sdks-libraries/cli/quickstart.md", + "sha256": "1858a694752b9323a2c22dcfaee18bae6193ded55b22be7bf81443d7f6ffee7c", + "size": 4786 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/cli/authentication", + "status": "success", + "path": "en/cli-sdks-libraries/cli/authentication.md", + "sha256": "bcc80c33a35d00ebc0413439a860941340cd2b37a858b8a9389fd05c16c313d8", + "size": 7295 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/cli/using", + "status": "success", + "path": "en/cli-sdks-libraries/cli/using.md", + "sha256": "71a95ffdd2a6359caadb6b5371a994cea70505f85dabc6c61a404f6d3bf9c9d0", + "size": 9980 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/cli/scripting", + "status": "success", + "path": "en/cli-sdks-libraries/cli/scripting.md", + "sha256": "e1eb6003c70b56dfe5925816d5218c988c55fea24e25cc505a28ae29cea97210", + "size": 6192 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/middleware", + "status": "success", + "path": "en/cli-sdks-libraries/middleware.md", + "sha256": "a2824c8576947c98174cbc5f96d8dc2a979ae8a75dbcd90297728f1c108671cf", + "size": 5848 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/sdks/python", + "status": "success", + "path": "en/cli-sdks-libraries/sdks/python.md", + "sha256": "4aef6a9653e2a65162792bdd476f5190d74dfe831918caeff8b28ab6fe51e7ab", + "size": 24957 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/sdks/typescript", + "status": "success", + "path": "en/cli-sdks-libraries/sdks/typescript.md", + "sha256": "f149156785f457f35ea21142298d12089634e03a6dd87d5b52e231e0667c86f9", + "size": 28352 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/sdks/csharp", + "status": "success", + "path": "en/cli-sdks-libraries/sdks/csharp.md", + "sha256": "e565b176a0086431d55230af17316f9cc8a1c435208b8e4c2dadd80e313b50de", + "size": 15696 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/sdks/go", + "status": "success", + "path": "en/cli-sdks-libraries/sdks/go.md", + "sha256": "2f3c3f6ec0bbe2d6493eca69911db34c614b0264ee04f453fb48343d76e6cdab", + "size": 25599 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/sdks/java", + "status": "success", + "path": "en/cli-sdks-libraries/sdks/java.md", + "sha256": "510ce0e9868bb03b76cf3bcdfd36b65382e53bcc73949d8650786eebb957df89", + "size": 46811 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/sdks/php", + "status": "success", + "path": "en/cli-sdks-libraries/sdks/php.md", + "sha256": "d16f8c5722be5af1f87e632021469e8ea85feba2e367c072ad5a94e72546dc62", + "size": 8785 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/sdks/ruby", + "status": "success", + "path": "en/cli-sdks-libraries/sdks/ruby.md", + "sha256": "06bf7c527282df070f28ad29f1022e6ec2b9fc5da570dfe4f319f24f9757bcdc", + "size": 14104 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/libraries/apple-foundation-models", + "status": "success", + "path": "en/cli-sdks-libraries/libraries/apple-foundation-models.md", + "sha256": "d571d2143882693e5e84167def79bae78aed799c4b8e3cd5473c7b3a62b14e2c", + "size": 11879 + }, + { + "url": "https://platform.claude.com/docs/en/cli-sdks-libraries/libraries/openai-sdk", + "status": "success", + "path": "en/cli-sdks-libraries/libraries/openai-sdk.md", + "sha256": "9aa394fd753ae37b66aa048b6562ceb588bbe0df8717bb5262ce80858f9f6188", + "size": 18619 + }, + { + "url": "https://platform.claude.com/docs/en/api/overview", + "status": "success", + "path": "en/api/overview.md", + "sha256": "c3b4ad9c93bf27c3cfb2a7f2cbddc5f10d0dee07131396ef0a855bb8ecf226bc", + "size": 14717 + }, + { + "url": "https://platform.claude.com/docs/en/api/beta-headers", + "status": "success", + "path": "en/api/beta-headers.md", + "sha256": "3f4f857e58927f12d0f0bafe50dda2a392875b63ea48b031d0a8526fe90068e6", + "size": 7307 + }, + { + "url": "https://platform.claude.com/docs/en/api/errors", + "status": "success", + "path": "en/api/errors.md", + "sha256": "180b373a012a3df0cf77896efda5fd98907834ac70784e62de7572bd37ba0e15", + "size": 19892 + }, + { + "url": "https://platform.claude.com/docs/en/api/claude-code/routines-fire", + "status": "success", + "path": "en/api/claude-code/routines-fire.md", + "sha256": "e4cc31df472f37ebc90e29851418fc1036a91f89f18d7426f52fc7574b24e2f6", + "size": 13441 + }, + { + "url": "https://platform.claude.com/docs/en/api/rate-limits", + "status": "success", + "path": "en/api/rate-limits.md", + "sha256": "bb126c24d0faebfd476ef1a1e77783e33f71069deffebeaeff6c9c1195e3e597", + "size": 25831 + }, + { + "url": "https://platform.claude.com/docs/en/api/service-tiers", + "status": "success", + "path": "en/api/service-tiers.md", + "sha256": "53b2a7a36efbef398133c50c4b2eabb87c93bcc73d2e7a4a5e32744f766df9dd", + "size": 8648 + }, + { + "url": "https://platform.claude.com/docs/en/api/claude-platform-on-aws-iam-actions", + "status": "success", + "path": "en/api/claude-platform-on-aws-iam-actions.md", + "sha256": "e41f594ec26cc38317dd3a9dacea04fe371517bd68f0caf40b870eefb967d290", + "size": 42989 + }, + { + "url": "https://platform.claude.com/docs/en/api/versioning", + "status": "success", + "path": "en/api/versioning.md", + "sha256": "e692856b0e16e611948e411790e0dc615a66fde02af448d4962a1e8723d29dc6", + "size": 1674 + }, + { + "url": "https://platform.claude.com/docs/en/api/ip-addresses", + "status": "success", + "path": "en/api/ip-addresses.md", + "sha256": "925b05d453f1f390ae620ff75ecd6ca8f7d5d029a61f12988b925d49a7f24d22", + "size": 1357 + }, + { + "url": "https://platform.claude.com/docs/en/api/supported-regions", + "status": "success", + "path": "en/api/supported-regions.md", + "sha256": "53522a8dd4579642d4c2f1604c0b30bd1c5734fe65477df0a191994bb7a0ac23", + "size": 2242 + }, + { + "url": "https://platform.claude.com/docs/en/agents-and-tools/agent-skills/claude-api-skill", + "status": "success", + "path": "en/agents-and-tools/agent-skills/claude-api-skill.md", + "sha256": "864d0996404e2a74db6f65e131e06869e14ff2d712ca2b7201e9995f0c30d2fe", + "size": 11667 + }, + { + "url": "https://platform.claude.com/docs/en/release-notes/overview", + "status": "success", + "path": "en/release-notes/overview.md", + "sha256": "bbd33d60d94b563b90b8d8b0c006f7b3a4c8761499875419ea803c57078785ec", + "size": 69863 + }, { "url": "https://platform.claude.com/docs/en/api/completions", "status": "success", @@ -10873,8 +11272,8 @@ "url": "https://platform.claude.com/docs/en/api/admin", "status": "success", "path": "en/api/admin.md", - "sha256": "520f3edeaa3c38aa68f23f2e2fc5f54d43c0d66446e207105e73ae3f83221de9", - "size": 529144 + "sha256": "e73c4037238326798e02cd012c08481503f8a4a81fafcf6c0506e86964a8ca27", + "size": 529494 }, { "url": "https://platform.claude.com/docs/en/api/admin/organizations", @@ -11209,15 +11608,15 @@ "url": "https://platform.claude.com/docs/en/api/admin/api_keys", "status": "success", "path": "en/api/admin/api_keys.md", - "sha256": "9ba2a911545196251c1362fbc8028318b5209dff175113d26fb6d1b762a72765", - "size": 8735 + "sha256": "7156ce6eb5bd6e21db2bac2d031ccd45d688689f3f59d36acb47eff6bdf54523", + "size": 9085 }, { "url": "https://platform.claude.com/docs/en/api/admin/api_keys/retrieve", "status": "success", "path": "en/api/admin/api_keys/retrieve.md", - "sha256": "b318baf05f9dd7fa45768935c08ad5688eb2082e6eedffa24100c01de4d6f14c", - "size": 2438 + "sha256": "8d10ffee44a046215a1bd867fad9da2c55770c1751fc0cb77a4c65b1c347e8b8", + "size": 2788 }, { "url": "https://platform.claude.com/docs/en/api/admin/api_keys/list", @@ -12164,1195 +12563,1209 @@ "sha256": "4203dd2d89a5521dc985deab1f64c1bfa6d1dd29c1300f7d1569dfb020e10a09", "size": 1072 }, + { + "url": "https://platform.claude.com/docs/en/about-claude/use-case-guides/classification", + "status": "success", + "path": "en/about-claude/use-case-guides/classification.md", + "sha256": "c4acf12d1d9a842d0edd216bcc28a53d6378da4c083a709d853488a963b433bf", + "size": 8293 + }, + { + "url": "https://platform.claude.com/docs/en/claude_api_primer", + "status": "success", + "path": "en/claude_api_primer.md", + "sha256": "2b1ee28f697f849eb1ec14d162cdeec9a35ae772469f8ba8e11278d16f9602c3", + "size": 25669 + }, { "url": "https://code.claude.com/docs/en/accessibility", "status": "success", "path": "en/docs/claude-code/accessibility.md", - "sha256": "574aafb62e5ef9c5f17cd4e51c5bdafe4e86fa849f64b18a054f8ecad8dd699c", - "size": 10578 + "sha256": "c5d715d12d3b7943daa3b6000633587ae20defc6a10418c48b918f12cecefee1", + "size": 10688 }, { "url": "https://code.claude.com/docs/en/admin-setup", "status": "success", "path": "en/docs/claude-code/admin-setup.md", - "sha256": "a84979910df19e73e7f17d21464e8409e4e451225246cc2d96ad44bb9f3ffbcf", - "size": 32303 + "sha256": "8a258d966111c547a29265bc71fcc518add3eea18fdfb9facbf6aafbaa569811", + "size": 34531 }, { "url": "https://code.claude.com/docs/en/advisor", "status": "success", "path": "en/docs/claude-code/advisor.md", - "sha256": "a031c810457a8ac02747fdbaedc7ad743ad1e047acc679bd2845e7ebf16ce65f", - "size": 15819 + "sha256": "3f6239560bfa1e62b0944cf2f6f16e933d43f6c10030ad48614263291d693f80", + "size": 15904 }, { "url": "https://code.claude.com/docs/en/agent-sdk/agent-loop", "status": "success", "path": "en/docs/claude-code/agent-sdk/agent-loop.md", - "sha256": "d500e9d41b47ce4e7b4b678cb580077f93aae037a5986d7384ee1ee559394955", - "size": 43326 + "sha256": "bcbddde0d3d1376359df97d619509175d2ddfeb1518e479364c60148e1e23cb1", + "size": 43611 }, { "url": "https://code.claude.com/docs/en/agent-sdk/claude-code-features", "status": "success", "path": "en/docs/claude-code/agent-sdk/claude-code-features.md", - "sha256": "ce4a7ad96c530e2d627607344318abdd3c2bc59aed270916dcc94eec04d8933b", - "size": 26047 + "sha256": "70e0525878ffb62188f4b47aba4bdfa086be5add0079d2e70b3d0049de58ac1d", + "size": 26237 }, { "url": "https://code.claude.com/docs/en/agent-sdk/cost-tracking", "status": "success", "path": "en/docs/claude-code/agent-sdk/cost-tracking.md", - "sha256": "c8a0f8ab6a5d63eaca1c5ab73a1bb03f08792ac6434340230c745bc3d329caad", - "size": 17889 + "sha256": "39742a18899f21eaaeb0080861f6a607a575aac76bc1e1dbdd157a909d7b8f4b", + "size": 17979 }, { "url": "https://code.claude.com/docs/en/agent-sdk/custom-tools", "status": "success", "path": "en/docs/claude-code/agent-sdk/custom-tools.md", - "sha256": "9ba425210b84b82ca3e4b66fbbbc284bba4aaa181514f3d187fbb71696fdea8b", - "size": 40360 + "sha256": "0ce3287c56927e8c76b6cff678fcc65e9f4c6f686c609a5c64d629806152023c", + "size": 41400 }, { "url": "https://code.claude.com/docs/en/agent-sdk/file-checkpointing", "status": "success", "path": "en/docs/claude-code/agent-sdk/file-checkpointing.md", - "sha256": "a4f936654dc78a04ba3273df44fdb3e07268b446771595d821a7ddec179f75b1", - "size": 32950 + "sha256": "edc4eb39cf9a62ed1b70aba2cbbfef7d8eedf33130a7ce21ce80792ff579f696", + "size": 32980 }, { "url": "https://code.claude.com/docs/en/agent-sdk/hooks", "status": "success", "path": "en/docs/claude-code/agent-sdk/hooks.md", - "sha256": "5c928bf514bcbd65a861966821af23ab521132ed9ba093f2969688dbf94ab4e7", - "size": 49687 + "sha256": "df572790d39a7200dc3cf91f7f2329aaf293731b922bcf1e1bcb4013e4055627", + "size": 49832 }, { "url": "https://code.claude.com/docs/en/agent-sdk/hosting", "status": "success", "path": "en/docs/claude-code/agent-sdk/hosting.md", - "sha256": "eb14647932dc27dd2a57c6140e3c8c897ff771883322509cca02e1f2a3e9a923", - "size": 22974 + "sha256": "46492f937f14492e2eb8c164dd430c6c6bb5affaaea18f19988a1657578224cc", + "size": 23109 }, { "url": "https://code.claude.com/docs/en/agent-sdk/mcp", "status": "success", "path": "en/docs/claude-code/agent-sdk/mcp.md", - "sha256": "a8865b51b7521c270615881809898c6a2484b206b43280be8add630007a02f05", - "size": 32824 + "sha256": "7a47900cc0972123acb2f14d002008f92760b5df80c3336933ac5715c3dcd0c8", + "size": 32929 }, { "url": "https://code.claude.com/docs/en/agent-sdk/migration-guide", "status": "success", "path": "en/docs/claude-code/agent-sdk/migration-guide.md", - "sha256": "ce2431a74b12a75eb79472144829b1741ef2134e7ce45afe24864af23f38f34c", - "size": 9718 + "sha256": "9810a05e9c2a606b3d0814eab2031f97d96aaecce424c592a720578d5d0d5d92", + "size": 9753 }, { "url": "https://code.claude.com/docs/en/agent-sdk/modifying-system-prompts", "status": "success", "path": "en/docs/claude-code/agent-sdk/modifying-system-prompts.md", - "sha256": "6a6232cfa38771700c82718cd03d573288abe70d1b12b706b7b66d3c9848a5c0", - "size": 24019 + "sha256": "99eed2b8e57e645c5ce12fd537d6e637209827cb452678cc1c472c6bb3aed0c6", + "size": 24079 }, { "url": "https://code.claude.com/docs/en/agent-sdk/observability", "status": "success", "path": "en/docs/claude-code/agent-sdk/observability.md", - "sha256": "71b92a662cf3200410828a5f7f029757d16a1dea6cbd64723b2bed401ef7deb7", - "size": 18179 + "sha256": "6cf771d618224364c832cda432ddf9d3c7e163bef8b96e1476cb563d25a61da1", + "size": 18249 }, { "url": "https://code.claude.com/docs/en/agent-sdk/overview", "status": "success", "path": "en/docs/claude-code/agent-sdk/overview.md", - "sha256": "c66d4f86d302ee2e8a0640759b48bdabfd45a340cd44dfc78e25c47c2a1344ad", - "size": 28121 + "sha256": "72f434e264cd03bbad3d11ede2056ef909980c790568380027fd9b65014cc4b5", + "size": 29648 }, { "url": "https://code.claude.com/docs/en/agent-sdk/permissions", "status": "success", "path": "en/docs/claude-code/agent-sdk/permissions.md", - "sha256": "cfa2aee712391ca1ed858c0f9a007988321b76978f676331cf9d033f61443c2e", - "size": 20578 + "sha256": "b91593e834c4e4cabd80a352494f9947a4cac191553db558cc471398a7ec7826", + "size": 20718 }, { "url": "https://code.claude.com/docs/en/agent-sdk/plugins", "status": "success", "path": "en/docs/claude-code/agent-sdk/plugins.md", - "sha256": "8975ffbfd10b03e66e6677c900bf1b0858f6b15158eaec9f031bc21fe400eb75", - "size": 12373 + "sha256": "e141b7d8f77621f0e145f65b49104f6109173ffc2b5ce9f2e3e8057daf2c511a", + "size": 12418 }, { "url": "https://code.claude.com/docs/en/agent-sdk/python", "status": "success", "path": "en/docs/claude-code/agent-sdk/python.md", - "sha256": "6504777a50ab5419dd8078ee4f3e346715a8e70d7623c828908eb907da38fef5", - "size": 188878 + "sha256": "e8141eaf2bc92d721e817c0739fe3fd172a936c62646a07cc198809991993d6e", + "size": 190872 }, { "url": "https://code.claude.com/docs/en/agent-sdk/quickstart", "status": "success", "path": "en/docs/claude-code/agent-sdk/quickstart.md", - "sha256": "93d2fda7d4e4a7b30767c5899294024359f546caf0949b25c4d464718a525812", - "size": 17865 + "sha256": "548b0ae0312ae5776965cbbcae240eeb0215c33133ece3f9c65f77b3588dc81b", + "size": 17955 }, { "url": "https://code.claude.com/docs/en/agent-sdk/secure-deployment", "status": "success", "path": "en/docs/claude-code/agent-sdk/secure-deployment.md", - "sha256": "2a461c7caf4cc2d5352db78389442e8f99fff389c4d4d28a2074eec0095457cf", - "size": 24596 + "sha256": "c818e40053277fa22277c57062871296e633e046cc72cb7e7681968e17841f0e", + "size": 24631 }, { "url": "https://code.claude.com/docs/en/agent-sdk/session-storage", "status": "success", "path": "en/docs/claude-code/agent-sdk/session-storage.md", - "sha256": "7fe6cc103c5ed040387c4c7afb39de0578898c989d70a271a5d14296f52c5cbd", - "size": 18995 + "sha256": "f9841faface9ddcfd8f04125454fed5beefff06cabcc16cb5259576c2111f132", + "size": 19075 }, { "url": "https://code.claude.com/docs/en/agent-sdk/sessions", "status": "success", "path": "en/docs/claude-code/agent-sdk/sessions.md", - "sha256": "471156adbc8f014302a1fa569ec8ded93c59bd690cb142c5a597bd8ff085cb1e", - "size": 21491 + "sha256": "a624c41567c45d3cd61f1e39498e66e63fceb6ece65213491596104f8480b51a", + "size": 21631 }, { "url": "https://code.claude.com/docs/en/agent-sdk/skills", "status": "success", "path": "en/docs/claude-code/agent-sdk/skills.md", - "sha256": "a4e882ab3e5ac9252a29da0ddd4418db933d8e039a846f1ee8daaa375cc2d67c", - "size": 12608 + "sha256": "4cebd3d4713c08ed1853246151a06e4234271266634043a44c62a942b03db6ad", + "size": 12663 }, { "url": "https://code.claude.com/docs/en/agent-sdk/slash-commands", "status": "success", "path": "en/docs/claude-code/agent-sdk/slash-commands.md", - "sha256": "6fe87483e6f619e87b32a153ab3af6dd77e63a29e72b3d65986f04bab2b631e0", - "size": 20439 + "sha256": "53703e1c38fce6081d84e2d0a6e3cc7683d3c1f66c4b9657b7e65eeca8531e73", + "size": 20489 }, { "url": "https://code.claude.com/docs/en/agent-sdk/streaming-output", "status": "success", "path": "en/docs/claude-code/agent-sdk/streaming-output.md", - "sha256": "881fecf52ff379c6fa3b10404e9ffe5a3eabe147c6ac42523b611a8eb3196a20", - "size": 15702 + "sha256": "bf3957a10096d4ba157d2da1fd8bcbc28123385d6ae334fc3546aeaeeeaa6652", + "size": 15737 }, { "url": "https://code.claude.com/docs/en/agent-sdk/streaming-vs-single-mode", "status": "success", "path": "en/docs/claude-code/agent-sdk/streaming-vs-single-mode.md", - "sha256": "1e1433fe43358dc8e5efa9723333e04df680bbf79fe9281972f81510c5c407e3", - "size": 10684 + "sha256": "2870c21f7ed23cb85cf8de342b9ec3fa3f2c43701a0d44c3ad41a1077271c8f7", + "size": 10694 }, { "url": "https://code.claude.com/docs/en/agent-sdk/structured-outputs", "status": "success", "path": "en/docs/claude-code/agent-sdk/structured-outputs.md", - "sha256": "5c2feff4d988b9d341290f13469d301970be2be9457fe29f08715701e76ca589", - "size": 21439 + "sha256": "74d557b7cc219b110ca577dd2e26c368d08cd577c0af8ea47ef3964370ee5c1b", + "size": 21449 }, { "url": "https://code.claude.com/docs/en/agent-sdk/subagents", "status": "success", "path": "en/docs/claude-code/agent-sdk/subagents.md", - "sha256": "6cd6baec3853c5bb25ca78a4f9d78ba4d086962ef53837f2f0206e6af9fcefe6", - "size": 38287 + "sha256": "39e20485c190a0133a29820c27f3aa9280539699652d21da6142502b70bb092b", + "size": 38382 }, { "url": "https://code.claude.com/docs/en/agent-sdk/todo-tracking", "status": "success", "path": "en/docs/claude-code/agent-sdk/todo-tracking.md", - "sha256": "3c46eae308ade3016c8baa5102c08bfe7b77c7bd2d4938b3c0338f0a460e50e9", - "size": 15912 + "sha256": "cfb9228c67c0c23b9b3ee2c605b58f0c896fd1056a10c8c5f4a099155c88c99e", + "size": 15942 }, { "url": "https://code.claude.com/docs/en/agent-sdk/tool-search", "status": "success", "path": "en/docs/claude-code/agent-sdk/tool-search.md", - "sha256": "cea80398215b9facc208b7e2be8ee924f4e07695a687bb06a679c718b0ccfdab", - "size": 10948 + "sha256": "92e7582fd652e96cee0c04ef47872309c777a72129280dbdf29374668541c35f", + "size": 10988 }, { "url": "https://code.claude.com/docs/en/agent-sdk/typescript", "status": "success", "path": "en/docs/claude-code/agent-sdk/typescript.md", - "sha256": "5b407ed31ea891b90e4a38d8a5d5321967932d208a46661583fadcd726ce761e", - "size": 215095 + "sha256": "de34611900433ea7e970a902b535d9600c24a9048bbd4605dcb8e4e55a0e6929", + "size": 218616 }, { "url": "https://code.claude.com/docs/en/agent-sdk/typescript-v2-preview", "status": "success", "path": "en/docs/claude-code/agent-sdk/typescript-v2-preview.md", - "sha256": "f6a71545c9c84bcfe62470b91591118b76ef4bd315438503e7a18603209c7ad8", - "size": 12025 + "sha256": "9299fb45b5fc6be56a7efbc6167f73dbea4d9c73db575ddf624297566b17b328", + "size": 12050 }, { "url": "https://code.claude.com/docs/en/agent-sdk/user-input", "status": "success", "path": "en/docs/claude-code/agent-sdk/user-input.md", - "sha256": "b77cea41effd6acf5affaed28be86c5669d12b1e95012b3aabe2e2f06d2b9890", - "size": 39647 + "sha256": "91cbe97f95e3d88100bf4f038c27926e8f917384bac46c0951b179a4abb350f5", + "size": 39747 }, { "url": "https://code.claude.com/docs/en/agent-teams", "status": "success", "path": "en/docs/claude-code/agent-teams.md", - "sha256": "8a09e2b9aaf1241ef40b8f3876d152d7bc228af64c8c43f1925200558f33ac6d", - "size": 33051 + "sha256": "fc6e6b7a60c3b770b560ba0df53f5742ff1e08a7f3cfa45938c3d135807dcd9c", + "size": 33191 }, { "url": "https://code.claude.com/docs/en/agent-view", "status": "success", "path": "en/docs/claude-code/agent-view.md", - "sha256": "f9cb7f0ce54fe5fb423ec9068b8902a0eeabdc1bf169ee92a68ccc9bcd8c3dbd", - "size": 142433 + "sha256": "afb7b4a9f231dcfc9a9e09c6d4b1e807c98b757a148426f744eaa3adfd074537", + "size": 144395 }, { "url": "https://code.claude.com/docs/en/agents", "status": "success", "path": "en/docs/claude-code/agents.md", - "sha256": "5565a2466f2a1576900c21638cf20500b1811499e7045d971cd61c279cc00ff1", - "size": 8011 + "sha256": "f1e18a3c16e99e639466660864faaa0c2e9166f80cd01406661bbd23110f8b39", + "size": 8176 }, { "url": "https://code.claude.com/docs/en/amazon-bedrock", "status": "success", "path": "en/docs/claude-code/amazon-bedrock.md", - "sha256": "4043c2835f81bdcf9d976c191e07a49032c5f81bc0e7b5ab9aaa79b447e7849c", - "size": 35798 + "sha256": "0e73a6e9a80bdbc4278b267f8af7b00129f1083167f16f72799279d61f80ed78", + "size": 35918 }, { "url": "https://code.claude.com/docs/en/analytics", "status": "success", "path": "en/docs/claude-code/analytics.md", - "sha256": "dc84be935b175bd46522b061b5508986736849f66bcec7701276a38a56fa88fe", - "size": 12520 + "sha256": "6e1a46fbf128d24c0a252cd28d1c7377b3c050b24cc9827caf6e7e537e31fa97", + "size": 12545 }, { "url": "https://code.claude.com/docs/en/artifacts", "status": "success", "path": "en/docs/claude-code/artifacts.md", - "sha256": "86a72b661993a4c413290f762f222c908dbde0fd6e643243798eea6392d271d5", - "size": 25940 + "sha256": "7102c251acfc6a3d50b300e2f4394c2a18b6d64e2b2f260717117bb75314e92a", + "size": 26030 }, { "url": "https://code.claude.com/docs/en/authentication", "status": "success", "path": "en/docs/claude-code/authentication.md", - "sha256": "0fb2aa7a865d1dd0eff488fe36c543cb7f23066639a7e1300dd7d0718c78e47b", - "size": 16932 + "sha256": "75d1338f741e8e064df1a5587b09421d2f01c42b28d30702c11f66811a1eeb70", + "size": 17513 }, { "url": "https://code.claude.com/docs/en/auto-mode-config", "status": "success", "path": "en/docs/claude-code/auto-mode-config.md", - "sha256": "a47d532e6e308a14d951132ee33578dc4e640cc0df6b8be79895cdf81e46d459", - "size": 26748 + "sha256": "c1f5b0bae8e485922979d4953b4170213df022da6409a098a97884e6d9a2aed7", + "size": 26838 }, { "url": "https://code.claude.com/docs/en/best-practices", "status": "success", "path": "en/docs/claude-code/best-practices.md", - "sha256": "4feccb5c94ada554ac8f9c559611b84aa221bcf62121b843876b92aed2612630", - "size": 39377 + "sha256": "a39640f06323961248f5cd17b89b46d6e4aafac74b64fccf4dbbcb216516183c", + "size": 39582 }, { "url": "https://code.claude.com/docs/en/champion-kit", "status": "success", "path": "en/docs/claude-code/champion-kit.md", - "sha256": "c3574420815a3daae4062535ff3db41b5be6dd0c269cb5e4b6d500dc615ca176", - "size": 22309 + "sha256": "9cf42a49f32f911a39094ec1087356ffcc890f0a043933c0487ed81f6dff257f", + "size": 22389 }, { "url": "https://code.claude.com/docs/en/changelog", "status": "success", "path": "en/docs/claude-code/changelog.md", - "sha256": "db9342e4baa525eea914128706442a9d8fc0fe230af73ffa246238139c8a1d68", - "size": 489044 + "sha256": "0e5079420711a8106453f00dcb10c7d67a6eb5b17e56b06b29185dbae9179aee", + "size": 494233 }, { "url": "https://code.claude.com/docs/en/channels", "status": "success", "path": "en/docs/claude-code/channels.md", - "sha256": "4c1990a45da18850c2a7ce638d827bde8650e93962ef60c357664b8bcfcc01b4", - "size": 22356 + "sha256": "385ab2b9d5d8eb64920822549dfbd816458f3742fec1a4973526561b1f1a5f40", + "size": 22446 }, { "url": "https://code.claude.com/docs/en/channels-reference", "status": "success", "path": "en/docs/claude-code/channels-reference.md", - "sha256": "905c590b5cb3c3f3efeac10f05ce7b9366fe78ab956a2e22c0666e9b3a8b7e29", - "size": 46667 + "sha256": "8ee3179450eb3f034a959ed44d2de2d56dea83b4d4f1575bbe08421a7925a932", + "size": 46737 }, { "url": "https://code.claude.com/docs/en/checkpointing", "status": "success", "path": "en/docs/claude-code/checkpointing.md", - "sha256": "2f0ab38cfbb0fcc855af5b30d625781b597ffebf8ec2110aa4614b4870fe5782", - "size": 7012 + "sha256": "515f68b9d53c737dad4b10e9363f3c1ca4ce28fc3ea9c10443a1158ae0522b66", + "size": 7037 }, { "url": "https://code.claude.com/docs/en/chrome", "status": "success", "path": "en/docs/claude-code/chrome.md", - "sha256": "85e5cc0785f6817d9a88240c6145857f7ed09903f310ccfa45c4bdee22fa8cfe", - "size": 15327 + "sha256": "8dc6461269855d7c01b35a9424a2fd24e0ce473a380cb101b026557bd8c4f639", + "size": 15382 }, { "url": "https://code.claude.com/docs/en/claude-apps-gateway", "status": "success", "path": "en/docs/claude-code/claude-apps-gateway.md", - "sha256": "077bc655b3b52c6385883f8ead0864368e781b1022dbf1e3763e1980cf039923", - "size": 53152 + "sha256": "455d0099f2e34ea291a7a7fccb1fa2808a7411cd57adc7f28b8dd3a0099cc0a3", + "size": 53422 }, { "url": "https://code.claude.com/docs/en/claude-apps-gateway-config", "status": "success", "path": "en/docs/claude-code/claude-apps-gateway-config.md", - "sha256": "7b0125f0d118595e9fe2107a1fc63ac47aeffc5ce78a7e698fe9c78f75f72a3b", - "size": 79175 + "sha256": "9f8f9e41a2066e046e73cbbe1df5bce9897136fe2880f8c1d79caee4c81e8a58", + "size": 79365 }, { "url": "https://code.claude.com/docs/en/claude-apps-gateway-deploy", "status": "success", "path": "en/docs/claude-code/claude-apps-gateway-deploy.md", - "sha256": "4d29d6ca6268813f27d324033e3342d79a650eb179b3351a43acec43d0e18195", - "size": 52777 + "sha256": "3c2d60758fc16477e4985a5f36e75c7c553ca767742c51dd69e1eef3e2022ea1", + "size": 52922 }, { "url": "https://code.claude.com/docs/en/claude-apps-gateway-on-gcp", "status": "success", "path": "en/docs/claude-code/claude-apps-gateway-on-gcp.md", - "sha256": "bc51771a94f39fdaa67b2cef912c4383ef744362caa19a1c7f2d3b4685d38c05", - "size": 24197 + "sha256": "84e103b298008d88ce7238a81d55b916c9213471f6091839d4cd3a112036ae86", + "size": 24282 }, { "url": "https://code.claude.com/docs/en/claude-apps-gateway-spend-limits", "status": "success", "path": "en/docs/claude-code/claude-apps-gateway-spend-limits.md", - "sha256": "a2bda61b7f6247f80d7da0bae7611f2e3d085a031dbac0e8614955d1f8a9d1fb", - "size": 15368 + "sha256": "870ddf46c8650164e152da46b608d1deaa0c9e976051a80d1f491010f902ee4b", + "size": 15433 }, { "url": "https://code.claude.com/docs/en/claude-code-on-the-web", "status": "success", "path": "en/docs/claude-code/claude-code-on-the-web.md", - "sha256": "5da364f62f0fb114cbda749bb4674ad3cd31c2b51b92e089149879458f9f017b", - "size": 65275 + "sha256": "79ca3e98fae9f1d80f7017947bebf4dfef417e0cae9588d32903903a606fa4f1", + "size": 65505 }, { "url": "https://code.claude.com/docs/en/claude-directory", "status": "success", "path": "en/docs/claude-code/claude-directory.md", - "sha256": "50520c5275485cce00539944ff0513c98327b61b207fb777c09d63194fbb72dd", - "size": 87685 + "sha256": "9ca98b493874566d360cbe5d937a1d83be55fe6b4b1dfcf6d876cf9597aceb89", + "size": 88005 }, { "url": "https://code.claude.com/docs/en/claude-platform-on-aws", "status": "success", "path": "en/docs/claude-code/claude-platform-on-aws.md", - "sha256": "5a8e5ef4daf836b8afee983f4312c25cf473b05042d07e2aeaca437ceba70226", - "size": 18583 + "sha256": "2954025b78406bd6f20b4104397f7a8595f7365ed3aef4af27b3296de6a7204d", + "size": 18638 }, { "url": "https://code.claude.com/docs/en/cli-reference", "status": "success", "path": "en/docs/claude-code/cli-reference.md", - "sha256": "54350aebfb326f91095051d297ee6d59b85cb9e9c32e4e78422365a9b2ffbb76", - "size": 104178 + "sha256": "ffab33516822c955bfeba62e66ab71ee9cd8cc6c7979900b887688577451bde1", + "size": 104653 }, { "url": "https://code.claude.com/docs/en/code-review", "status": "success", "path": "en/docs/claude-code/code-review.md", - "sha256": "880d20f1e76ee356e88722dd5da481619570bccfd53f8deb3e180c2dbbb49385", - "size": 25068 + "sha256": "c2b991212c90f32e62a772cce12c9b829674c4a64c09a2e5f2f197748a5173b5", + "size": 25143 }, { "url": "https://code.claude.com/docs/en/commands", "status": "success", "path": "en/docs/claude-code/commands.md", - "sha256": "f29d49f4005662fbf05616985b858a439c55803906ed938c651820d7ae078c45", - "size": 184620 + "sha256": "d819bbe42a7a9d98c6f56d297375acfb8072cfb3443ab743969528a2be23e9ea", + "size": 185215 }, { "url": "https://code.claude.com/docs/en/common-workflows", "status": "success", "path": "en/docs/claude-code/common-workflows.md", - "sha256": "1fbbc61ca6c9c2094c363a3a6f2dfc8867e292346975d49378b9343890f80e39", - "size": 18444 + "sha256": "1ebdc7f9889596a169fe4b715b6f4b0dcd2baff10e75765a193ac15b8c8f0580", + "size": 18539 }, { "url": "https://code.claude.com/docs/en/communications-kit", "status": "success", "path": "en/docs/claude-code/communications-kit.md", - "sha256": "5b916679049ada750674617ad366ea7a4fdb7364129996a60fd1f3d88bf32c7a", - "size": 25415 + "sha256": "f6ac574c477e0aa9eaf61c4009015434c99a023fc48663942e5adcf7535a7105", + "size": 25465 }, { "url": "https://code.claude.com/docs/en/computer-use", "status": "success", "path": "en/docs/claude-code/computer-use.md", - "sha256": "426f0be07e0e4b26385021b886b705ddf998c789d34de3b1841388b8ad1f0cfe", - "size": 12432 + "sha256": "0f59cf0fd91094e1ff373c4f00769055e871154714c4fff1c42938bb44f79d60", + "size": 12482 }, { "url": "https://code.claude.com/docs/en/context-window", "status": "success", "path": "en/docs/claude-code/context-window.md", - "sha256": "9342d125641bbbce8755fee05dc5ac47b60e484eeacefc991b1115d91fe7d54e", - "size": 57868 + "sha256": "5c64e9d8f40b78f059a174ffe6402f5e4b313a8743eb8bd0cfea124a3bd4b445", + "size": 57948 }, { "url": "https://code.claude.com/docs/en/corporate-launcher", "status": "success", "path": "en/docs/claude-code/corporate-launcher.md", - "sha256": "d1dde83d2f53f74ee5df19073d8d5031605f93ac90ae2846748dab2dc0890358", - "size": 13409 + "sha256": "27380fc97bf9eefcbc6d0fa5f108a0ad91ed5382d953ee6efd6bbe27b8750bc3", + "size": 13509 }, { "url": "https://code.claude.com/docs/en/costs", "status": "success", "path": "en/docs/claude-code/costs.md", - "sha256": "7bd1f7705f024b296ddb2bd702086ace406d310865bebb154c4b384c551e31a3", - "size": 28662 + "sha256": "362da7bec045633785fcc5643938b4e9bb3c114bfc4d51caa00e3cd67018bfce", + "size": 28822 }, { "url": "https://code.claude.com/docs/en/data-usage", "status": "success", "path": "en/docs/claude-code/data-usage.md", - "sha256": "b8c0eab985c5e4212a106fd46779038a0358963fb2ce13b1a754c6a15b7751b4", - "size": 20049 + "sha256": "716ab8bb72232372eaf66b1e7fd2b725f6fa4ca2bf55415842e89929772dcc7e", + "size": 20159 }, { "url": "https://code.claude.com/docs/en/debug-your-config", "status": "success", "path": "en/docs/claude-code/debug-your-config.md", - "sha256": "64a6192b55ce10e9f5d6096587c58c357b56d367d1c73c7030aed6d0ba37a103", - "size": 21059 + "sha256": "1c0ad5e80406b44cca2b891bcf281eb306834a3d8182fc44ff7c4dd7e4f771d8", + "size": 21229 }, { "url": "https://code.claude.com/docs/en/deep-links", "status": "success", "path": "en/docs/claude-code/deep-links.md", - "sha256": "150cb36d510ee0ea6ba4a4c16a9efc1109034dd5ae5353ce6cc211c873ee951d", - "size": 14810 + "sha256": "980c5ffa86c376c1a1a0451b2bcf64d2b736333ee2f7142043c9c6941a94400d", + "size": 14835 }, { "url": "https://code.claude.com/docs/en/desktop", "status": "success", "path": "en/docs/claude-code/desktop.md", - "sha256": "419c8cff4567eaf8eb6ccea5a9ec275b3c16a853a5f26dd3fb62fab4ad758353", - "size": 86009 + "sha256": "0ac2c25bdc096bb378b5ac0b75d5dc66303cea77edd589f3a1ca22eea2d3e800", + "size": 86454 }, { "url": "https://code.claude.com/docs/en/desktop-linux", "status": "success", "path": "en/docs/claude-code/desktop-linux.md", - "sha256": "6bdfd9f7520951b3bf21114f13d987db82775efc949c36bc100d40fb4c6ca46d", - "size": 7107 + "sha256": "1562b9e7017b302fee5a38e356caf600111ad9243d8fd55c1b441008f989040f", + "size": 7142 }, { "url": "https://code.claude.com/docs/en/desktop-quickstart", "status": "success", "path": "en/docs/claude-code/desktop-quickstart.md", - "sha256": "c41fc802c86452fd070af0eb8351ce8785beab102b886fe55c57b5ba83737afd", - "size": 11086 + "sha256": "8d683b6f5b1e8330e9fc72a55e2929277b989300bb2f435a257a0508cfb4c76a", + "size": 11261 }, { "url": "https://code.claude.com/docs/en/desktop-scheduled-tasks", "status": "success", "path": "en/docs/claude-code/desktop-scheduled-tasks.md", - "sha256": "8f6e792ac76e9b40405de15f4c10e83d37ec525b9caef600843491690dd7e44f", - "size": 11492 + "sha256": "e7dfc9343bd66f2fff2bc770c7b6571875f3f132db7df6b7403c5c7cacd2ed04", + "size": 11567 }, { "url": "https://code.claude.com/docs/en/desktop-wsl", "status": "success", "path": "en/docs/claude-code/desktop-wsl.md", - "sha256": "1de311e2c470a9ad1b1f152f7ea822cb4a8fc8e0687110aff24b830e830a5248", - "size": 2957 + "sha256": "71129cad01b0efe4c2bd3fa3e60617895d13189b1c063b830cf9543044dcfad3", + "size": 2962 }, { "url": "https://code.claude.com/docs/en/devcontainer", "status": "success", "path": "en/docs/claude-code/devcontainer.md", - "sha256": "49fd24b5bc9cc078bf3f4f75f6e068d4b4f61443c9f4c744cb34eb31ea342288", - "size": 17385 + "sha256": "c72fed7e75b5b57c88b6e25b51efecafd399575c44aec3136fdb22f9251ace8c", + "size": 17535 }, { "url": "https://code.claude.com/docs/en/discover-plugins", "status": "success", "path": "en/docs/claude-code/discover-plugins.md", - "sha256": "006e0bdcb18b30960aa220c7f8a1dace7f42d494ea3204d367ef90b69540f54c", - "size": 27234 + "sha256": "491d5b388939665d19c44493a4c8d1dcb036cc101844eff6631eedc01beda135", + "size": 27384 }, { "url": "https://code.claude.com/docs/en/env-vars", "status": "success", "path": "en/docs/claude-code/env-vars.md", - "sha256": "ee1188d564510380817025e386d2b7499ade25f89ea6cea7344d2cc98fbfa7e6", - "size": 325227 + "sha256": "3317158bbf686df0db9699b97b9caeaefc53233a71722563086a03162ba92c09", + "size": 329716 }, { "url": "https://code.claude.com/docs/en/errors", "status": "success", "path": "en/docs/claude-code/errors.md", - "sha256": "eef1d9628c543082d45b70a44f7ec7e3410fbcfcd5f62a0334ea6752630545eb", - "size": 119759 + "sha256": "36c0e2806a3a5fc703fa555eb965395763a855ecfc9a0806a1195d3fe67f386c", + "size": 138316 }, { "url": "https://code.claude.com/docs/en/fast-mode", "status": "success", "path": "en/docs/claude-code/fast-mode.md", - "sha256": "bc519fafc9d7d8573c6e614790d72883039d92401676e0dffa7dc954862e9db0", - "size": 14094 + "sha256": "3602237bd3e2cd25c8d4484e59ab884777df741c0fe3fdde5c1bfa7febf6b903", + "size": 14184 }, { "url": "https://code.claude.com/docs/en/feature-availability", "status": "success", "path": "en/docs/claude-code/feature-availability.md", - "sha256": "ca70421ad8c82c7168848472d45ca4421a764bc6ae6b07cb29adfc1aab969b41", - "size": 19902 + "sha256": "c636fcb3c1e6edc333e47901a13ce8727c51b5b74612397eceb56fa87e2b6cd7", + "size": 20542 }, { "url": "https://code.claude.com/docs/en/features-overview", "status": "success", "path": "en/docs/claude-code/features-overview.md", - "sha256": "eda9ecce85038815bdd9531e132f406c52d31078fd8f291fc3d5b7afdeadb9f2", - "size": 30699 + "sha256": "a28bea6880c6b1f9416dd8ea4d07b01aee53a84a985ff3739687a869a32c1748", + "size": 30974 }, { "url": "https://code.claude.com/docs/en/fullscreen", "status": "success", "path": "en/docs/claude-code/fullscreen.md", - "sha256": "7a0c1020d87c10498597589430092351b1e44a4d5ca32d6441f49d881b6c30d8", - "size": 22728 + "sha256": "7026e72db712a59d7cf0d55138ff3b141eb5b6a8023b9de53d9680ab6fdf772f", + "size": 23112 }, { "url": "https://code.claude.com/docs/en/gateways", "status": "success", "path": "en/docs/claude-code/gateways.md", - "sha256": "53d8b01706f7b3ce64d83c039de3c76011a4a24c3f8f741b72e0355f875668d5", - "size": 8674 + "sha256": "c1ba759a8681856b8b6cdf623a66e8ab1841d2a44b2cf113c133f92b32db6cb9", + "size": 8784 }, { "url": "https://code.claude.com/docs/en/github-actions", "status": "success", "path": "en/docs/claude-code/github-actions.md", - "sha256": "d8fd2c65e27a9d184b74bcc5214179883ace40fef2fa50bc212e00bef714292b", - "size": 29695 + "sha256": "b5b839f60ff45816372ac4cc13336b97c9d7abcf18571cbea7c18a49c6f51fe4", + "size": 29725 }, { "url": "https://code.claude.com/docs/en/github-enterprise-server", "status": "success", "path": "en/docs/claude-code/github-enterprise-server.md", - "sha256": "052b12a0f3ac7bfad4ade2d0776b0637d1084ef565b7f63754396535cb30e980", - "size": 19576 + "sha256": "05b3a96047c1a1f237a603139f2d2cca77065cf2cdbebc1f6ea6a592da8251f7", + "size": 19676 }, { "url": "https://code.claude.com/docs/en/gitlab-ci-cd", "status": "success", "path": "en/docs/claude-code/gitlab-ci-cd.md", - "sha256": "50229363728305765045171e92e03bf69dcf7195fda93b7a4e1e1885beb270aa", - "size": 18279 + "sha256": "6cf18fe767e2d99cf1e8b47ac9c18effac7b6c43891ddb1b4d26190ec0fa44b5", + "size": 18284 }, { "url": "https://code.claude.com/docs/en/glossary", "status": "success", "path": "en/docs/claude-code/glossary.md", - "sha256": "b9727fd21482ebf2d067d1d4895693ee931cbf7507f9d7f0154cf6697f6b8eb8", - "size": 22650 + "sha256": "7124719893dffc0f6e837ffb097ed66f3abbda98760de2f0a01731e4d3db098c", + "size": 22940 }, { "url": "https://code.claude.com/docs/en/goal", "status": "success", "path": "en/docs/claude-code/goal.md", - "sha256": "eb3bc9ebe201a9cdc9eab2c6a1a8f0d98395a2791d30f4b673c303ecd0e383eb", - "size": 9000 + "sha256": "56a76edc6bf3e8e52f184385c7223a075d520b0565ad8a3eb5b07d04b811e076", + "size": 9080 }, { "url": "https://code.claude.com/docs/en/google-vertex-ai", "status": "success", "path": "en/docs/claude-code/google-vertex-ai.md", - "sha256": "a78a92cdbe85faefa7c364bed918f9ec6240d84b911cd232feaf211cfad8fc97", - "size": 20338 + "sha256": "f091b073b283884bcc772fa881681b1b9f9239a89c1b9dd852913bd4ec0241a5", + "size": 20393 }, { "url": "https://code.claude.com/docs/en/headless", "status": "success", "path": "en/docs/claude-code/headless.md", - "sha256": "8d39e87fadc1c6f9905b6fd7df31343065e110848edeff3fc1a2103d9a95693e", - "size": 22311 + "sha256": "ca8efba5e0acf19dd9d85d4aed108c949189cb3f11404483e7f52a30f3e2530f", + "size": 22486 }, { "url": "https://code.claude.com/docs/en/hooks", "status": "success", "path": "en/docs/claude-code/hooks.md", - "sha256": "509b8036585918b0f121b09d0e4ce41952ab778bad794e75c961c14b628a0fca", - "size": 232221 + "sha256": "c2d0b1e77e9d229299102dbb8d71b71787fa9b9996c60a091972b6a63202fef9", + "size": 236614 }, { "url": "https://code.claude.com/docs/en/hooks-guide", "status": "success", "path": "en/docs/claude-code/hooks-guide.md", - "sha256": "8cefb47b77ee1b88233882a17b6350e5b33533673bab22bfd24cd81e2d0ef902", - "size": 60072 + "sha256": "c05b5f5d57712475dd0e79d998799a0cae1957222840ab52850e1cd3707cc898", + "size": 60390 }, { "url": "https://code.claude.com/docs/en/how-claude-code-works", "status": "success", "path": "en/docs/claude-code/how-claude-code-works.md", - "sha256": "c3e93b24c01d1624301dddb578d73d2c839a1d087bf03391dfdbdc12f27e2cf7", - "size": 19002 + "sha256": "6a204ff447e0b6b7a5710861b79bbdb7e67799ae5ef05ebf6e0ce8f4906680b3", + "size": 19217 }, { "url": "https://code.claude.com/docs/en/interactive-mode", "status": "success", "path": "en/docs/claude-code/interactive-mode.md", - "sha256": "1f3decfcc74cb9d53bd060aa2cb7dd339732728cabf6ccca5257de0bed9d1bef", - "size": 49693 + "sha256": "c18e034ce5e356b5bae60f9336a1e2e9203944ba374531d891c61b2864c25a85", + "size": 49893 }, { "url": "https://code.claude.com/docs/en/jetbrains", "status": "success", "path": "en/docs/claude-code/jetbrains.md", - "sha256": "71bebb6268a21a59070720bb6ede2af8d10bae0c21fb1d30f8c8324b7b626a61", - "size": 11936 + "sha256": "a5f0360ca8b0e7be9ebd885cd4c80347d2329f9703e7a26f22946d53a959d2b7", + "size": 11981 }, { "url": "https://code.claude.com/docs/en/keybindings", "status": "success", "path": "en/docs/claude-code/keybindings.md", - "sha256": "d450c6c3b1f253d4acdb42c833bc3ef0aceee7e46fd1437a8c7f7b7fe455085e", - "size": 30508 + "sha256": "21e693e3bb174f0e56af8b16f9d707424584cb3a2ca1e186e0e20b513470d134", + "size": 30563 }, { "url": "https://code.claude.com/docs/en/large-codebases", "status": "success", "path": "en/docs/claude-code/large-codebases.md", - "sha256": "bad37d06c688384e918d055f1f2c20eb819e47548f19c2f17b0d58d260250b1c", - "size": 34515 + "sha256": "ff3698115ed40ec008eb55db10a59f82474e1b0a793022394d4ac03c4d2c39f6", + "size": 35112 }, { "url": "https://code.claude.com/docs/en/legal-and-compliance", "status": "success", "path": "en/docs/claude-code/legal-and-compliance.md", - "sha256": "69c673bb776250e503d4e8e65020c3b6e180a1d70aa78294ff4aad4fe9bd28eb", - "size": 3576 + "sha256": "d8814ea0d3a0ae8294d2e274ace1c5f7ab09353bae9da78a5050308828b4100f", + "size": 3591 }, { "url": "https://code.claude.com/docs/en/llm-gateway", "status": "success", "path": "en/docs/claude-code/llm-gateway.md", - "sha256": "899d5ef807239e660ef54ca238433388467de2a8b1ddcf3d44988179dc1799b8", - "size": 6056 + "sha256": "e77e458089836ea04a710510e2710c000599f0c08c12cc75b686ad8a0a8f795a", + "size": 6166 }, { "url": "https://code.claude.com/docs/en/llm-gateway-connect", "status": "success", "path": "en/docs/claude-code/llm-gateway-connect.md", - "sha256": "6fca0caa12b3424c561332f48fce031cb0d647e973c9d04315460c5dafc0a67a", - "size": 48500 + "sha256": "0c5329b87a1715f1bfcf4e8b07d83a96fbd88d0fe5289a10c6d2449ef524157d", + "size": 48720 }, { "url": "https://code.claude.com/docs/en/llm-gateway-protocol", "status": "success", "path": "en/docs/claude-code/llm-gateway-protocol.md", - "sha256": "94f190c456ebd38ec5091ec998d2d808cfb36081c9ee2dec6fa5f09a444eaec1", - "size": 27206 + "sha256": "4fa26b575a2e55a255247886c4fc1e2f0644a79aedfcb4ae07e330dba9818fc8", + "size": 27351 }, { "url": "https://code.claude.com/docs/en/llm-gateway-rollout", "status": "success", "path": "en/docs/claude-code/llm-gateway-rollout.md", - "sha256": "7b28f3c6466338c8bc031acb674980c4654da610eb5f6af0a22269c74a36cee8", - "size": 32015 + "sha256": "a01da98187066d09bd41725e15dcee436850dc9ec863e15b84c4954dfc9ad35f", + "size": 32215 }, { "url": "https://code.claude.com/docs/en/managed-mcp", "status": "success", "path": "en/docs/claude-code/managed-mcp.md", - "sha256": "7ee78a104a01c1e9f817280938d628997fb712c2a344e866f30dba153eb9b8a7", - "size": 29122 + "sha256": "40daa2cd7c091499777e6c5fe3c01d24052f0f43d2aafb7b975381bcdd25486c", + "size": 29262 }, { "url": "https://code.claude.com/docs/en/mcp", "status": "success", "path": "en/docs/claude-code/mcp.md", - "sha256": "52e110820b2be82dd8ced2ea6adebac330ddaaa987be836f67780fb27ee39214", - "size": 73291 + "sha256": "6f88dc3977c6c5eb6c74757609aabb342d6ba1541f2d1cd96407e1796ce3cd22", + "size": 73536 }, { "url": "https://code.claude.com/docs/en/mcp-quickstart", "status": "success", "path": "en/docs/claude-code/mcp-quickstart.md", - "sha256": "0687e5239dd59936a78e24bcd390d3969846d5b4b5fbed8c7fa109cdf832215c", - "size": 24122 + "sha256": "1a53a8b765220bebd775af1213fdf966c5ba1d9eb37b22baa204e47eb79f194c", + "size": 24212 }, { "url": "https://code.claude.com/docs/en/memory", "status": "success", "path": "en/docs/claude-code/memory.md", - "sha256": "9a6b9bc5f0e7e848c9311736f6d318bb51415caf3c71d2df2b38ccbbc77995fd", - "size": 32753 + "sha256": "5fd9c0dc12334ae2595022699d7224fd90832912e9e82da6c4115024199009e8", + "size": 34020 }, { "url": "https://code.claude.com/docs/en/microsoft-foundry", "status": "success", "path": "en/docs/claude-code/microsoft-foundry.md", - "sha256": "38de3512a0ce6c572f98f78e475bab6ed145e8cfa3f76de796be8b60cd975665", - "size": 10407 + "sha256": "07478cd337fe7050bc746e2424dd796b1ad90d8cb23b2ec6dce93ef97d5f4605", + "size": 10417 }, { "url": "https://code.claude.com/docs/en/mobile", "status": "success", "path": "en/docs/claude-code/mobile.md", - "sha256": "3a9f0740c3fa9545e4e61baa63676c5c24fbff0ae32b0de059557b299f842ac4", - "size": 8395 + "sha256": "356cb30546e3cd38167e4a69670c131c2ac9fc301a80751607b24f68189eafb5", + "size": 8510 }, { "url": "https://code.claude.com/docs/en/model-config", "status": "success", "path": "en/docs/claude-code/model-config.md", - "sha256": "62f788133e4c991a53bf1580ed9f861600b8bfeb4de98ade03be0705b87e6b85", - "size": 79870 + "sha256": "32650ab02f3379301de30f3d1abbf477814a41fd8ac486b142b651cc7ddbad70", + "size": 80325 }, { "url": "https://code.claude.com/docs/en/monitoring-usage", "status": "success", "path": "en/docs/claude-code/monitoring-usage.md", - "sha256": "8eed27eef358562909bbaae9ca2558372c3183901d8b700d5638b1c0bf7588e5", - "size": 106627 + "sha256": "37a5bd417f54a08064d418dcdbcd68184991687de72e1975839248428ba146aa", + "size": 107090 }, { "url": "https://code.claude.com/docs/en/network-config", "status": "success", "path": "en/docs/claude-code/network-config.md", - "sha256": "b585de969c4afa01c8feb9073bdd8ca802f7b655cf7e740f785cfd18040605a2", - "size": 16275 + "sha256": "9802e513a385c0e43bf61b20ff2290f9ca417c961e90d038d64d3772e83b151b", + "size": 18753 }, { "url": "https://code.claude.com/docs/en/output-styles", "status": "success", "path": "en/docs/claude-code/output-styles.md", - "sha256": "54d1fb85f0e20fa083e9cc76b29b37b674dd6c447499c1a62645ebae47155953", - "size": 9836 + "sha256": "dc40e5d4c4870730f7b6b654194788f9bd23f3e4b6d95f40db8fcf41d343eb39", + "size": 9911 }, { "url": "https://code.claude.com/docs/en/overview", "status": "success", "path": "en/docs/claude-code/overview.md", - "sha256": "fb1c6eac271eacf0bce114aa383a2b8920ac101733c6d6b8993147a9d45b8bc0", - "size": 16476 + "sha256": "6ab249f4af57b42d07e7a035d3f44a4fe61bd90b92c4e6fee6ba69d166e168fa", + "size": 16251 }, { "url": "https://code.claude.com/docs/en/permission-modes", "status": "success", "path": "en/docs/claude-code/permission-modes.md", - "sha256": "d3569e2b3d25777570280e99e849566aadd06aec28e0286018f09637f70ce2bc", - "size": 46503 + "sha256": "5a270f37044270e12a00e355c91a4f60e95a3f75c2fb6ac66223119f312948fd", + "size": 47076 }, { "url": "https://code.claude.com/docs/en/permissions", "status": "success", "path": "en/docs/claude-code/permissions.md", - "sha256": "4115fbedd0b8e5b5f33d8e9a7b53220e66d2d32dfe89993ea7182b80f3b82d10", - "size": 55942 + "sha256": "8c4e746e0c5be9793a79005dc8b63611ce306728530f6934d992f4403a999e9e", + "size": 57529 }, { "url": "https://code.claude.com/docs/en/platforms", "status": "success", "path": "en/docs/claude-code/platforms.md", - "sha256": "564d01882d0dc397747a8830b6f1c55305e473f1a042b05c0479326d2c46a4bd", - "size": 10878 + "sha256": "be30c00ee12f41c09ea9f31d21f11027be15cd46f8b6c27a7e3c1899cb0ebd35", + "size": 11138 }, { "url": "https://code.claude.com/docs/en/plugin-dependencies", "status": "success", "path": "en/docs/claude-code/plugin-dependencies.md", - "sha256": "5cdc588c46d96cd8a4236d79f64abef268fad765be6e439caeb870f52885bbdb", - "size": 21330 + "sha256": "4630a5fb450f9a96aa3b7d9f5e21ca0c124ff58e4b77b10c30baee9158d37898", + "size": 21370 }, { "url": "https://code.claude.com/docs/en/plugin-hints", "status": "success", "path": "en/docs/claude-code/plugin-hints.md", - "sha256": "02d5a1ce0d92f37ccb5795f5805376f79116d4e21bd49af7cc92658341920c60", - "size": 9717 + "sha256": "7a15a21fd6fb24e30524ff5b690959f72d4ed3f7487e079fc7479299e5e5cd81", + "size": 9767 }, { "url": "https://code.claude.com/docs/en/plugin-marketplaces", "status": "success", "path": "en/docs/claude-code/plugin-marketplaces.md", - "sha256": "dc8db9af5a1c3b1ed70f458b976a845fb4fd8e9740fcd1498c75aec625d715df", - "size": 71688 + "sha256": "c55f3461e5367188de1f3d67e61ba273bf3881b280066f32537dc2dca9bb3b23", + "size": 71868 }, { "url": "https://code.claude.com/docs/en/plugin-relevance", "status": "success", "path": "en/docs/claude-code/plugin-relevance.md", - "sha256": "f179430bc747f6a7eeaa57e63d71145a39bf1691d1ce980024afee4b0ccb65d4", - "size": 16181 + "sha256": "0920d33f7d634ca0f88bc7d5761e04f0aba45966afb97093593fd8df5f62d32b", + "size": 16226 }, { "url": "https://code.claude.com/docs/en/plugins", "status": "success", "path": "en/docs/claude-code/plugins.md", - "sha256": "e3b1cd68802a7951441e76fbfda0b8a2c69c93e077d9c7e480d40673dc71b364", - "size": 26079 + "sha256": "73f85ab77e79b493e8f1f37fdc0b83b98d23d5c6bbff4ff4d74fe181f54ff444", + "size": 26244 }, { "url": "https://code.claude.com/docs/en/plugins-reference", "status": "success", "path": "en/docs/claude-code/plugins-reference.md", - "sha256": "5bbcde15fdea0eb938c9cfb6f404910eff2e9eb3f71987b1dc3dd45a63179aa7", - "size": 87822 + "sha256": "049e382c1972913032d000cf18426cf97c7cc498b07bcfcaeac38e9075e0bd5d", + "size": 88062 }, { "url": "https://code.claude.com/docs/en/prompt-caching", "status": "success", "path": "en/docs/claude-code/prompt-caching.md", - "sha256": "617c5f9beae3a7a09a81e0a086567bcc9702607754c21b9571fd5545c2f3e57d", - "size": 28657 + "sha256": "e8caadf2ec8733025c7217efb0c159616491d1660815407b7fd6c611945c982a", + "size": 28907 }, { "url": "https://code.claude.com/docs/en/prompt-library", "status": "success", "path": "en/docs/claude-code/prompt-library.md", - "sha256": "b39d81a97db424ed2d641fc4481b80911410a1cba6d77fbeb557ce96c297f46a", - "size": 56097 + "sha256": "922e064e2fb8742b8af2ed8b48cff81d9839e0ce998245dfb97417b93dd27ac9", + "size": 56232 }, { "url": "https://code.claude.com/docs/en/quickstart", "status": "success", "path": "en/docs/claude-code/quickstart.md", - "sha256": "51b023fe4f64611ca6a697d7a5190a56980e06bfa924aad5ccca97ff463de8a0", - "size": 12751 + "sha256": "cf01823d70f5c9eb0187e98a447a76f9153ab1472d2c8e2a582d38c98962d18d", + "size": 12866 }, { "url": "https://code.claude.com/docs/en/remote-control", "status": "success", "path": "en/docs/claude-code/remote-control.md", - "sha256": "9c522c030e6311c7c6a860dab7fd7d5d83e0887c5a93898becdcf078e0ee96d5", - "size": 39538 + "sha256": "f7ebd427b00a7f4f2607c292a2b0b3bd16d3164522efc375d1e18033af6cfd64", + "size": 39961 }, { "url": "https://code.claude.com/docs/en/routines", "status": "success", "path": "en/docs/claude-code/routines.md", - "sha256": "0e907e04942a326d6f3df8d9551929e5c87fc42156201e0b72fa57169c725752", - "size": 31162 + "sha256": "7dedf65d4eaca373e08f8f3874d718c364b85f3d54ea08fd63ae4ce97d1d817e", + "size": 31920 }, { "url": "https://code.claude.com/docs/en/sandbox-environments", "status": "success", "path": "en/docs/claude-code/sandbox-environments.md", - "sha256": "6f05f1c348769af573f09443f13da33015a958471e5a0b852af6ba6a591f0584", - "size": 15802 + "sha256": "2324628f2ff29aed71ef7cd1e28a018757c4420d6e3f0315acea3ad1fbb51035", + "size": 15967 }, { "url": "https://code.claude.com/docs/en/sandboxing", "status": "success", "path": "en/docs/claude-code/sandboxing.md", - "sha256": "e4cd766b8a14fd3371f6e22ec9c5f3b8e1364f2a12a1268554be4687cc258ae5", - "size": 42441 + "sha256": "51f127a3c5a38011fa8eb8e4a2c57c657743581ff78097a426160bf7e08ef932", + "size": 42881 }, { "url": "https://code.claude.com/docs/en/scheduled-tasks", "status": "success", "path": "en/docs/claude-code/scheduled-tasks.md", - "sha256": "c069da72de29cece78a9f3a69638218028b4fe0fec38bca092a4ad989f2f72cf", - "size": 16644 + "sha256": "61724a3069bb0d4313de1506462f79ec8207b6f91811509edd14ef9de4d37a10", + "size": 16764 }, { "url": "https://code.claude.com/docs/en/security", "status": "success", "path": "en/docs/claude-code/security.md", - "sha256": "0886d228bdc321fff528389fe4de20cff477666458dbebd5f0f442fcb177a16b", - "size": 11042 + "sha256": "15a3393a1af5d6e640beac172e0831058c4150ed29a0f2911e0137415980daaa", + "size": 11152 }, { "url": "https://code.claude.com/docs/en/security-guidance", "status": "success", "path": "en/docs/claude-code/security-guidance.md", - "sha256": "cb9341aaf69abf3602f67f27cc38307b4701f2e6d39e1f47c3bfdf07a7f56442", - "size": 19853 + "sha256": "3536885c6fca282c58ef05a2b48e85cbe35a66b518e882ce3dd9d10f5095c792", + "size": 19928 }, { "url": "https://code.claude.com/docs/en/server-managed-settings", "status": "success", "path": "en/docs/claude-code/server-managed-settings.md", - "sha256": "e9f02b5eff23a6872cee72285abab39c7971d1bf59efff59a7993f4b4297053d", - "size": 26038 + "sha256": "4f5a7dc0607843767d64018b71a3f2ef2a45c6101a3fbfaaf19afcede18cfe96", + "size": 26203 }, { "url": "https://code.claude.com/docs/en/sessions", "status": "success", "path": "en/docs/claude-code/sessions.md", - "sha256": "c5947f605c202c2093862579a94dc9448be5b53375846f29765e124fa369e843", - "size": 18144 + "sha256": "09bf7cd3c7e48084bc3547ad204883cbf3b94f7d402c2402e30d71ebe8266576", + "size": 18319 }, { "url": "https://code.claude.com/docs/en/settings", "status": "success", "path": "en/docs/claude-code/settings.md", - "sha256": "c27ec03a0e6f5f61295945a3913e6e08d6ec7816dbc33fe3e2767cdfd58412a1", - "size": 265322 + "sha256": "72b3190a648921155fa0a69edd638c4a20f910145a45918995ada29c6b1b9546", + "size": 266332 }, { "url": "https://code.claude.com/docs/en/setup", "status": "success", "path": "en/docs/claude-code/setup.md", - "sha256": "49a3f9e72501f3862313263ad9c871490a81de364365438c0d47debab8f1567d", - "size": 30078 + "sha256": "cc4acc3af867749f1aa819a17dce2a606dcad8df70aaed1c7fe2869c46b045d6", + "size": 30263 }, { "url": "https://code.claude.com/docs/en/skills", "status": "success", "path": "en/docs/claude-code/skills.md", - "sha256": "209e5766f78ff65935b0a12364069f493c595388be72095f58ac13ed6f922527", - "size": 68067 + "sha256": "399a80e12c7bd12753a4fe0487579d052c1cd84d1f909a6dd555f02d7e8110c0", + "size": 69493 }, { "url": "https://code.claude.com/docs/en/slack", "status": "success", "path": "en/docs/claude-code/slack.md", - "sha256": "8c8449a9d4c2e517ce49a1eac3cead877f3a2cbed7f821128152e960fd09ab64", - "size": 14701 + "sha256": "9207433e566af55ee6120b12b888ad0caeb7e0117b838838a0772846a6818eff", + "size": 14716 }, { "url": "https://code.claude.com/docs/en/statusline", "status": "success", "path": "en/docs/claude-code/statusline.md", - "sha256": "a986e67cd5e92c6ae5fc8a5a607e55a96ffbf99f058d64a4c8eac2f65b9e295e", - "size": 63354 + "sha256": "813a29d86b487d9bcf3f8ffeda01dede78a1eb39c7d91d479b7a49cf9d034647", + "size": 64327 }, { "url": "https://code.claude.com/docs/en/sub-agents", "status": "success", "path": "en/docs/claude-code/sub-agents.md", - "sha256": "7bb05454a8259d336d8af51afdb06b09270607d15049e4d8f3fe127603a1804b", - "size": 85505 + "sha256": "7e6df1218a9cdc36518cb099640d93cac725e779d367e46f379ef2b285f523ab", + "size": 86170 }, { "url": "https://code.claude.com/docs/en/terminal-config", "status": "success", "path": "en/docs/claude-code/terminal-config.md", - "sha256": "5b3a6aa54cfe4353cd60d9cac62558df2aae0fc57c920e42521efcdfa9c01b8b", - "size": 21312 + "sha256": "73c7bf8f7f11ef90619e3615626effe8fad8d2f2abcba2648e9af9b819443369", + "size": 21735 }, { "url": "https://code.claude.com/docs/en/third-party-integrations", "status": "success", "path": "en/docs/claude-code/third-party-integrations.md", - "sha256": "cabac7526593b42f919da10044d78171f3f52b08c5c6bd90a4faa1992d39bacd", - "size": 16216 + "sha256": "8994dc0e9ea790781be5a2061cc5e25f5fd39caed4d8274455bd8f5e0efa4f95", + "size": 16356 }, { "url": "https://code.claude.com/docs/en/tools-reference", "status": "success", "path": "en/docs/claude-code/tools-reference.md", - "sha256": "1b9dbdf6abe7266114c016c5873da84555d350b7b463c3c43343514ceca9f352", - "size": 85425 + "sha256": "f2ee06121556907cff00fb306b5605340c1f13388ec052de61c293eeefbe8085", + "size": 90998 }, { "url": "https://code.claude.com/docs/en/troubleshoot-install", "status": "success", "path": "en/docs/claude-code/troubleshoot-install.md", - "sha256": "dcb61584c214fbe397e78277a1de9d51fbad328713f227ea4db1806e6bf39278", - "size": 48699 + "sha256": "541e9c2c4de6ea3f053ba003da8114f33e4ec3f7b6730857d31e861869c916da", + "size": 50351 }, { "url": "https://code.claude.com/docs/en/troubleshooting", "status": "success", "path": "en/docs/claude-code/troubleshooting.md", - "sha256": "b4687d055a34f36535401bbe8681a0eb1ae3acf0d48df5c8016f4d8662e88ef2", - "size": 10304 + "sha256": "90cc239e41097219bcec924daf692aeec2c2bbb1655871dd65a9ee8d54870a54", + "size": 10379 }, { "url": "https://code.claude.com/docs/en/ultraplan", "status": "success", "path": "en/docs/claude-code/ultraplan.md", - "sha256": "733aa4ced90941295d05c507e10e5a125a400d30446fbaed24903ed2527fb70a", - "size": 6142 + "sha256": "86f52e8cf34071ba765580433b9b570011ab44375e7ce45b4006467cd42b3c33", + "size": 6192 }, { "url": "https://code.claude.com/docs/en/ultrareview", "status": "success", "path": "en/docs/claude-code/ultrareview.md", - "sha256": "64e99c10a0cf6fafeafc01bf72376c3b527e9b29cf7041b616d7badaf6158aee", - "size": 11240 + "sha256": "890036c3052b4adaacc1c43bed3b4ab20709e51c9316a6ff602afa94c4989811", + "size": 11836 }, { "url": "https://code.claude.com/docs/en/voice-dictation", "status": "success", "path": "en/docs/claude-code/voice-dictation.md", - "sha256": "e7438dcc66ce1dc3b9167d6f485b8f8452c223b808ebf6cb42f57d3b2e0c2a21", - "size": 16274 + "sha256": "94244b70c4ea6bab2106e3404f9284d28aa41855e69616ff9c4f53f96ee03fdc", + "size": 16339 }, { "url": "https://code.claude.com/docs/en/vs-code", "status": "success", "path": "en/docs/claude-code/vs-code.md", - "sha256": "2334bcd39f74f951b24abb7c9d2ca37cf1b58d51754045465cdcf737643a2e6d", - "size": 48227 + "sha256": "6c94727fa1e0739aebe41bce94d412069ba9e9d184e81a1884a46999a5a8962e", + "size": 48427 }, { "url": "https://code.claude.com/docs/en/web-quickstart", "status": "success", "path": "en/docs/claude-code/web-quickstart.md", - "sha256": "470630e470ab15fc18af4816a082a79f092b912b5243f85825aa35f0b1156cde", - "size": 19355 + "sha256": "830305638102f760ed200ec85f22b99aab24119a470c462ea1ab15b0ed3bfc77", + "size": 19515 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w13", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w13.md", - "sha256": "e7a5c4b655864da37137de29d32d1b778db324a6eab8b7e3028e313861cc5141", - "size": 7976 + "sha256": "80f582261194b25a95822fc6eaa851fdf22931c963c7ecf036be004986f15ab2", + "size": 8016 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w14", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w14.md", - "sha256": "7c851b215ac027823ab67bfcc76827cc9f4e547e4815373434f886a780c45f5e", - "size": 7879 + "sha256": "53c72ca1de9878100f54de050a5d3432f3a2fdc8a9b0b0f6ab10bb5460eef413", + "size": 7914 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w15", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w15.md", - "sha256": "1b58ef2dac2a5a5cc78a0e2bef54fdd549bf39a4c3688111d1ce738b4c6922db", - "size": 7304 + "sha256": "b45286ab29c3c3d8653de0fd3830cfbc435774286dd890464c9a1c6a7bcb43fd", + "size": 7344 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w16", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w16.md", - "sha256": "ee973f20e71707ed5725fb3e220de5567370e4635fda06b24716f387d542a4b7", - "size": 8927 + "sha256": "cd6cc611efc53cf7319f2bab020f13d733affe6a6b7a5c270a2f218289fa64dc", + "size": 8982 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w17", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w17.md", - "sha256": "b4752e008a8f934d42a90f4c78793b8e240505c3994e6a02295332e877c6897e", - "size": 7267 + "sha256": "f578e85cf3fa449fd7c61e930970cd8fd70af9c2204d319b0c1a30daf99611df", + "size": 7337 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w18", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w18.md", - "sha256": "aff79e0150683f82a7ba56bfbc6b249d822a51778cc4c08e5ef9e093280da5bb", - "size": 6687 + "sha256": "a567c98c444a4e80c35d2090bf2cf756b1ce097c75cb57364d8ab5ddf155051e", + "size": 6722 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w19", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w19.md", - "sha256": "a4d69d75fd04c89569a31f3d1b97d37eb9014327018befb546c03e1f4d1b6ef3", - "size": 4604 + "sha256": "3d6247b5b48579fcd46e23c5286b90c390fd2e7c1f4b66aebb12578622a95e14", + "size": 4624 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w20", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w20.md", - "sha256": "5b093d94b4a1cf58b790326f2a35925d9f8d0ab3c25f5255f3487e21f67c76fc", - "size": 6853 + "sha256": "22e8189f0d539e07f0c6dbf0dac1de0145d64211cd0465acea97d0183e922525", + "size": 6878 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w21", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w21.md", - "sha256": "a89417b667f9c1b5fb1947507740059944d803d6b31711dfe7fc41d0cac54bab", - "size": 4328 + "sha256": "e17d2c9c819da2cd57397894b79b7355f2b590be6b023c41f1fe322fa51b6d22", + "size": 4353 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w22", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w22.md", - "sha256": "fd36195414fd4a8e69a4c4a5f5e0f92d2e379f3e3ac6d43aece75a21d1105101", - "size": 7311 + "sha256": "67a1e21ef37f6148bded984e7fec0868aef45f82803bfc3ff8ad6222810b1c52", + "size": 7341 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w23", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w23.md", - "sha256": "4f8ddbf899e5fbabe459481487a8d206b26049d1b1c0b8b1965425ab840d7985", - "size": 6581 + "sha256": "b018b3971c54c267f4b4623676eedb339e2728b9a4c413cd07a0428739110dc4", + "size": 6621 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w24", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w24.md", - "sha256": "35652ba87f7fc2569e9ff18ac642944a7a2cd7ded9a76f1ffb3053e3180202e6", - "size": 5342 + "sha256": "0baa5b3c36e72a00777e837bb379dc69d05cb73460a155d5ed29eddb5d6363c2", + "size": 5372 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w25", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w25.md", - "sha256": "4d7e48a048106e0041d47a8428d42c7e8f6d80b7853b007f452bc4f5939ad205", - "size": 5338 + "sha256": "a96708c857c757827143fe90043c37ba48a178317a49fa27c778f99654446336", + "size": 5363 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w26", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w26.md", - "sha256": "d03a60398668f09b9ae483686d3c57d93eac1a95f70cfe62d3b3fcb51239bf2f", - "size": 4451 + "sha256": "3b2140a178f36250ee1a24643bdc12b83b4f1534e1a1f8a94d40556a4c7e0625", + "size": 4471 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w27", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w27.md", - "sha256": "31229875e10a45c8a2a8cc769f5ef3b3718621f7cd74aa8940caa149ae8f8fdb", - "size": 6799 + "sha256": "f6386207819b3af9e7bb379d3e5df84dabf7f87b90fe9a5fc58e91d5c2ed5fe2", + "size": 6844 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w28", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w28.md", - "sha256": "6e1c2f542e6f08f0c00d58f96fd312e7b4ade1cbdaa9c51fcfbf1e8b10b22242", - "size": 4304 + "sha256": "d42c311bdf5932bcb49f8a765ada6d39073bfaa98d16fdf12807b7074463b83f", + "size": 4324 }, { "url": "https://code.claude.com/docs/en/whats-new/2026-w29", "status": "success", "path": "en/docs/claude-code/whats-new/2026-w29.md", - "sha256": "d816334042c3416f189ca4c0b4559e5e6b914cd1d4d4c1abd124f724001a3416", - "size": 5433 + "sha256": "e0563fc80db055d3df06848a14803783f8c06f05815f0374cf8c3d3b657ce93a", + "size": 5468 }, { "url": "https://code.claude.com/docs/en/whats-new/index", "status": "success", "path": "en/docs/claude-code/whats-new/index.md", - "sha256": "5ccb0e79c18799db3932a72f56efedb9c48d68356bcdfd2d923d07bca19e2e92", - "size": 11148 + "sha256": "9f66cbe31eb348a6d14df057a1b0bfa8dc0ee89e9e40db163a1be1723d518d34", + "size": 11238 }, { "url": "https://code.claude.com/docs/en/workflows", "status": "success", "path": "en/docs/claude-code/workflows.md", - "sha256": "59c8ed9aba62e803388fd396be3c12e118ad523e2a6cca72878da933dac5fd28", - "size": 28114 + "sha256": "6bb58ee28f984aefebed12f66f588be8cb94e68bcf57d94a2586d7a1573450b7", + "size": 28224 }, { "url": "https://code.claude.com/docs/en/worktrees", "status": "success", "path": "en/docs/claude-code/worktrees.md", - "sha256": "12f1917a8901171ecf23234432a043edff8f71162432eec761340b1c6a99424e", - "size": 18187 + "sha256": "4e50a52477834746b29ba307586fde7769f719fd4ebef18b91df6f8357277dae", + "size": 18332 }, { "url": "https://code.claude.com/docs/en/zero-data-retention", "status": "success", "path": "en/docs/claude-code/zero-data-retention.md", - "sha256": "e382d357acadabf66f3bdc49c25ba134d96e4829fa8ce87305cf6e40016bef16", - "size": 8683 + "sha256": "987003e69d73fb0de7161e222a7d660d42d839b9aaba073dea2074e6d20f127b", + "size": 8728 }, { "url": "https://modelcontextprotocol.io/community/antitrust", @@ -13953,8 +14366,8 @@ "url": "https://modelcontextprotocol.io/seps/2243-http-standardization", "status": "success", "path": "mcp/seps/2243-http-standardization.md", - "sha256": "81efee298669e8cd1d3062ba4ef9ec34783bc0c7159b4eb10d403a2c134f5a6f", - "size": 45821 + "sha256": "e8f5285ce3900dfe4f5829a4227d14bf49c4a2255f8b2289fbdd09d17b6076a7", + "size": 46551 }, { "url": "https://modelcontextprotocol.io/seps/2260-Require-Server-requests-to-be-associated-with-Client-requests", @@ -13988,8 +14401,8 @@ "url": "https://modelcontextprotocol.io/seps/2549-TTL-for-list-results", "status": "success", "path": "mcp/seps/2549-TTL-for-list-results.md", - "sha256": "fc726a9008ecf09c43cf1d22846a4a5328bc8e41a961c8d56e9ba21704570774", - "size": 15302 + "sha256": "3b6a40318ff92fe647e3e43d8b367af5305bee3ac140e972a40cf52a4933fc2a", + "size": 15306 }, { "url": "https://modelcontextprotocol.io/seps/2567-sessionless-mcp", @@ -14002,7 +14415,7 @@ "url": "https://modelcontextprotocol.io/seps/2575-stateless-mcp", "status": "success", "path": "mcp/seps/2575-stateless-mcp.md", - "sha256": "d35195f59f257050a71d60bc242b9fbcf7d548d4bbabe18041b301463c16d9b6", + "sha256": "6aabcf4ded1b0aa95889072261eeede1a3face8ddb4d0d7a62044242ad9364a3", "size": 35620 }, { @@ -14471,8 +14884,8 @@ "url": "https://modelcontextprotocol.io/specification/2025-06-18/server/tools", "status": "success", "path": "mcp/specification/2025-06-18/server/tools.md", - "sha256": "b500aee8a25af2bca9a8212922f3d31461503137c2ff10f58ab29152656cce77", - "size": 10705 + "sha256": "ecb4ed3880d10e8506dc586f646c754cdd82151b7531a2798f42c8250cc05d94", + "size": 10860 }, { "url": "https://modelcontextprotocol.io/specification/2025-06-18/server/utilities/completion", @@ -14625,8 +15038,8 @@ "url": "https://modelcontextprotocol.io/specification/2025-11-25/server/tools", "status": "success", "path": "mcp/specification/2025-11-25/server/tools.md", - "sha256": "8ed43301d44f1b15317963b8553eb025c3bd72f433ee873565338e727b106b55", - "size": 13906 + "sha256": "b372c48eea58744850f5684b796438f32643402533b24084d82b49b580721aff", + "size": 14061 }, { "url": "https://modelcontextprotocol.io/specification/2025-11-25/server/utilities/completion", @@ -14835,8 +15248,8 @@ "url": "https://modelcontextprotocol.io/specification/draft/server/tools", "status": "success", "path": "mcp/specification/draft/server/tools.md", - "sha256": "43e4cb92a6e6bcb64c771fa55687ce0f2da88e51f8cb87492892f44a40bd94a9", - "size": 24074 + "sha256": "d4a91750a9b352f74ffd11a69fafac45fa6ca2e6ad196c61cfb6477197c0e998", + "size": 24229 }, { "url": "https://modelcontextprotocol.io/specification/draft/server/utilities/caching", @@ -14947,8 +15360,8 @@ "url": "https://support.claude.com/en/articles/8114491-get-started-with-claude", "status": "success", "path": "support/8114491-get-started-with-claude.md", - "sha256": "531bf0756d1f63c1c11310bd024731c6835f59781515d76a101284d9588f46c5", - "size": 5300 + "sha256": "107046c6dd426305db5496de88dd968676f5c8718fd3584d67972e5628666962", + "size": 5302 }, { "url": "https://support.claude.com/en/articles/8114494-how-up-to-date-is-claude-s-training-data", @@ -15031,8 +15444,8 @@ "url": "https://support.claude.com/en/articles/8230524-how-can-i-delete-or-rename-a-conversation", "status": "success", "path": "support/8230524-how-can-i-delete-or-rename-a-conversation.md", - "sha256": "e3f60a592a36627f85157c19f4ab00a9a48834ed0b61b8a79d8d08ef2afd86c6", - "size": 1301 + "sha256": "a98814f9d650d8bb98b3653d4bcc4ce46c0e84e0e17f9a497fbabbf13cd0ec32", + "size": 1299 }, { "url": "https://support.claude.com/en/articles/8241126-upload-files-to-claude", @@ -15073,8 +15486,8 @@ "url": "https://support.claude.com/en/articles/8287232-verify-your-phone-number", "status": "success", "path": "support/8287232-verify-your-phone-number.md", - "sha256": "952a6cc6deab83c5116508805e0afa696e6c680746dfb9235ae052e188439a32", - "size": 3979 + "sha256": "ce39f08177dc2bd2bd183a700600bf162e3ae96f9ffbeba0ce1a80b6561703fa", + "size": 3981 }, { "url": "https://support.claude.com/en/articles/8325606-what-is-the-pro-plan", @@ -15101,8 +15514,8 @@ "url": "https://support.claude.com/en/articles/8325618-paid-plan-billing-faqs", "status": "success", "path": "support/8325618-paid-plan-billing-faqs.md", - "sha256": "0d96c649233980d0c6cbc231fe432039cf40536a98d00c9cf3c082f497a75e18", - "size": 4551 + "sha256": "172e535197ab43f0c393f4966c577cef03ddb2abd02ecefff2797daf1b323b5a", + "size": 4553 }, { "url": "https://support.claude.com/en/articles/8325621-i-would-like-to-input-sensitive-data-into-my-chats-with-claude-who-can-view-my-conversations", @@ -15143,7 +15556,7 @@ "url": "https://support.claude.com/en/articles/8606378-how-do-i-use-the-workbench", "status": "success", "path": "support/8606378-how-do-i-use-the-workbench.md", - "sha256": "e77a17f167054efd057ea4cc4899ffcb8979ac9eca21e76a8c511eae2eb4d799", + "sha256": "6c3d5ff41598bd7f05964b57acbc2f51fabc15cc81c701f855a1d2a582822004", "size": 9651 }, { @@ -15171,7 +15584,7 @@ "url": "https://support.claude.com/en/articles/8887527-customizing-your-appearance-settings", "status": "success", "path": "support/8887527-customizing-your-appearance-settings.md", - "sha256": "84307341ed5a78c349e772c192eec97472b982964e6ff8b7db2b3b5b6c7d5b75", + "sha256": "6eeec94bc1e04e5645752b81f592a5737292dcb8ddb9d07b580dedd30584e210", "size": 1868 }, { @@ -15227,8 +15640,8 @@ "url": "https://support.claude.com/en/articles/9028421-how-can-i-delete-my-claude-account", "status": "success", "path": "support/9028421-how-can-i-delete-my-claude-account.md", - "sha256": "dafedabec360eb94b6a3580911798a378d2b68225dc0e6becff51fce1f686fd9", - "size": 2025 + "sha256": "18e6b328a6d00be902c294537255589e6babf22c189d8e4b3da448904b12f6ca", + "size": 2019 }, { "url": "https://support.claude.com/en/articles/9035075-law-enforcement-requests", @@ -15346,8 +15759,8 @@ "url": "https://support.claude.com/en/articles/9267400-move-your-personal-claude-account-to-a-team-or-enterprise-organization", "status": "success", "path": "support/9267400-move-your-personal-claude-account-to-a-team-or-enterprise-organization.md", - "sha256": "beaf5048814206bcaecc62f254b332a3e302cba9dbd36b7d3d07d10342c93dc8", - "size": 4809 + "sha256": "f8e732b0cf51fa528ff60968a5d8ac6c596f0fa4983faad3c56050bd4420e7f0", + "size": 4811 }, { "url": "https://support.claude.com/en/articles/9301722-updates-to-our-acceptable-use-policy-now-usage-policy-consumer-terms-of-service-and-privacy-policy", @@ -15395,15 +15808,15 @@ "url": "https://support.claude.com/en/articles/9519177-how-can-i-create-and-manage-projects", "status": "success", "path": "support/9519177-how-can-i-create-and-manage-projects.md", - "sha256": "2a6761d66ca635b1a54be8f1c77fb3094351f2623b8b772d79d9f90a27984d38", - "size": 11172 + "sha256": "26f9f73fb4c1dc74a4aec78c2fefbbf8a8fdbcfb634e799db6e1b46f9cae2ee5", + "size": 11186 }, { "url": "https://support.claude.com/en/articles/9519189-manage-project-visibility-and-sharing", "status": "success", "path": "support/9519189-manage-project-visibility-and-sharing.md", - "sha256": "bb3404a479f2aeab793e27f5692f492d9a8b6584cea0f38f430e38cd42e44cdf", - "size": 6807 + "sha256": "e2f381b6da8826a167b5ea56d51e5014a2c2d722da71d3cf926786d60b4e26ac", + "size": 6809 }, { "url": "https://support.claude.com/en/articles/9519291-what-is-anthropic-s-policy-for-handling-governmental-requests-for-user-information", @@ -15423,15 +15836,15 @@ "url": "https://support.claude.com/en/articles/9534590-cost-and-usage-reporting-in-the-claude-console", "status": "success", "path": "support/9534590-cost-and-usage-reporting-in-the-claude-console.md", - "sha256": "ec803e5ddad4b21e0f8eccf18beb79ed3de12fdb0d555f558e8cc9515c9f3882", - "size": 5108 + "sha256": "203839c9cb7a7dc4aab01c9936500b3466d16dc9ebece4ac001992bc8cab5518", + "size": 5104 }, { "url": "https://support.claude.com/en/articles/9547008-publish-and-share-artifacts", "status": "success", "path": "support/9547008-publish-and-share-artifacts.md", - "sha256": "64c6d3d5138e4f4b5d3cc271ec4774a75e8c3e4d9a721616efc96a373942c2cc", - "size": 6642 + "sha256": "2b58c975fcdb20d2ca94078f72d9f662eba649f4f14b5a310027b4cc5de6be41", + "size": 6636 }, { "url": "https://support.claude.com/en/articles/9612887-install-claude-for-android", @@ -15507,8 +15920,8 @@ "url": "https://support.claude.com/en/articles/9927533-disable-public-projects-for-your-organization", "status": "success", "path": "support/9927533-disable-public-projects-for-your-organization.md", - "sha256": "1f4287e74003988e6e795032f7a66a77238c8c7acc9bf537d73f0d12e1a6fc96", - "size": 2580 + "sha256": "0302444bdf6ba503e078048ecf484b0e9180acd971f53c5ea7b1494fdb200739", + "size": 2578 }, { "url": "https://support.claude.com/en/articles/9927624-add-or-update-your-team-plan-s-tax-or-vat-id", @@ -15647,14 +16060,14 @@ "url": "https://support.claude.com/en/articles/10310342-how-do-i-log-out-of-all-active-sessions", "status": "success", "path": "support/10310342-how-do-i-log-out-of-all-active-sessions.md", - "sha256": "ccc922debde1f9685425b93ddc232c3b03942bfd506cdf1f64f26125a922d7c5", - "size": 2480 + "sha256": "65f9b3b88c885470c170d760d5018129a304cbc887ca7e7d949b85278bcd603c", + "size": 2484 }, { "url": "https://support.claude.com/en/articles/10366376-how-can-i-delete-my-claude-console-account", "status": "success", "path": "support/10366376-how-can-i-delete-my-claude-console-account.md", - "sha256": "915c50406aa004e63a9124e462abd548965eecc574081a01cd97ed8a8d301b53", + "sha256": "7eafe9e9309821843998ea93550c54555aeed214cb4ab644aa4a75951e01291c", "size": 3165 }, { @@ -15703,14 +16116,14 @@ "url": "https://support.claude.com/en/articles/10504844-manage-user-feedback-settings-on-team-and-enterprise-plans", "status": "success", "path": "support/10504844-manage-user-feedback-settings-on-team-and-enterprise-plans.md", - "sha256": "e80ce4db95205b00be7566daf32f2e60649c1fdaf8eed67cce193d64b675758e", - "size": 1036 + "sha256": "c2bed923c8029f6bd076bf45156ec70fb3b2dad7180ea97d86cae60fe056eb1c", + "size": 1034 }, { "url": "https://support.claude.com/en/articles/10504853-manage-user-feedback-settings-on-claude-console", "status": "success", "path": "support/10504853-manage-user-feedback-settings-on-claude-console.md", - "sha256": "369be87edd08a0c39442bc79405d540cea09ca09df7c9382fe96d7dcd28662f6", + "sha256": "e4e847922e46d359840b0c10edccdff8d5145187bbfc79f0324d20689205aa74", "size": 997 }, { @@ -15724,15 +16137,15 @@ "url": "https://support.claude.com/en/articles/10593882-share-and-unshare-chats", "status": "success", "path": "support/10593882-share-and-unshare-chats.md", - "sha256": "843aef8bafee621580f642c482e16d72833f065cc82ca2b36917b97106759aa5", - "size": 4020 + "sha256": "408eb51e85eed15f09a9f3050e6e02b35591a51d0972740251b7212bbc53b617", + "size": 4012 }, { "url": "https://support.claude.com/en/articles/10684626-enable-and-use-web-search", "status": "success", "path": "support/10684626-enable-and-use-web-search.md", - "sha256": "04e011414cc30599eec51a5396bd9eff7ae4930c716ef35f098a6a6860e20985", - "size": 6364 + "sha256": "e742f1b40ba68e952308ac5cdf2d9d3740c1c1f7acca0828b62b7e1cd1102c66", + "size": 6358 }, { "url": "https://support.claude.com/en/articles/10684638-reporting-blocking-and-removing-content-from-claude", @@ -15745,8 +16158,8 @@ "url": "https://support.claude.com/en/articles/10722177-sharing-prompts-in-the-claude-console", "status": "success", "path": "support/10722177-sharing-prompts-in-the-claude-console.md", - "sha256": "e907a3955fccf911e8b5362c2ad339c51afa4c71e96f971d7ef553ac39485a75", - "size": 4531 + "sha256": "8c3b58524df9475b8b729d8716d5be44b717ba45da41f657ad27b0adf7ee56a1", + "size": 4527 }, { "url": "https://support.claude.com/en/articles/10769299-how-to-use-claude-in-your-preferred-language", @@ -15759,7 +16172,7 @@ "url": "https://support.claude.com/en/articles/10949351-getting-started-with-local-mcp-servers-on-claude-desktop", "status": "success", "path": "support/10949351-getting-started-with-local-mcp-servers-on-claude-desktop.md", - "sha256": "1394c353afd00becdd4ed7c960b6284be15588a488492e1340effd49a825895d", + "sha256": "6e73ca11cc01f76808eb528653daacd9fe64c5ad2ac116e4dc544ef13c80bf1b", "size": 8269 }, { @@ -15801,8 +16214,8 @@ "url": "https://support.claude.com/en/articles/11101966-use-voice-mode", "status": "success", "path": "support/11101966-use-voice-mode.md", - "sha256": "7042580c377f093bdd0e1539c01a8690639f3624eee94afd20153ac9997393cf", - "size": 8418 + "sha256": "d9ceb017499b22751c874edfdadead4cec7ade8dc3fb7f205867cdf530312f84", + "size": 8408 }, { "url": "https://support.claude.com/en/articles/11107691-why-is-a-coupon-or-promotion-not-available-for-my-account", @@ -15906,8 +16319,8 @@ "url": "https://support.claude.com/en/articles/11506255-get-started-with-claude-in-slack", "status": "success", "path": "support/11506255-get-started-with-claude-in-slack.md", - "sha256": "52feae0e6e0678dde95d7fa6a1c80d78130ee39773c72b7405a587170a43f941", - "size": 10406 + "sha256": "e7444c2e3b3d5810cc3c439f156319b4bd886b632d61fedd1d4951056296e043", + "size": 10412 }, { "url": "https://support.claude.com/en/articles/11526368-how-am-i-billed-for-my-enterprise-plan", @@ -15948,8 +16361,8 @@ "url": "https://support.claude.com/en/articles/11725453-set-up-the-claude-lti-in-canvas-by-instructure", "status": "success", "path": "support/11725453-set-up-the-claude-lti-in-canvas-by-instructure.md", - "sha256": "ccd9f9284cfb0e26fd1b0d58bc43b72612e3ff2544d84e5ad016a3bdc6e93d92", - "size": 2750 + "sha256": "495a9f856735caa0a32e03969c75b9ea8572a69771eeecafc0b5c5e141495737", + "size": 2748 }, { "url": "https://support.claude.com/en/articles/11732894-who-owns-and-manages-the-data-of-my-claude-for-education-account", @@ -15962,15 +16375,15 @@ "url": "https://support.claude.com/en/articles/11817273-use-claude-s-chat-search-and-memory-to-build-on-previous-context", "status": "success", "path": "support/11817273-use-claude-s-chat-search-and-memory-to-build-on-previous-context.md", - "sha256": "1a8a84d822a9add7ab0ec9c236889a114a4ff5a8d55fd646b569c6e46817db12", - "size": 21522 + "sha256": "f5e45aed182f948fa32624b82ec139653c5f54cb40db6b3b6654b1080cda70e6", + "size": 21514 }, { "url": "https://support.claude.com/en/articles/11818288-why-am-i-being-asked-to-verify-my-payment-method", "status": "success", "path": "support/11818288-why-am-i-being-asked-to-verify-my-payment-method.md", - "sha256": "0c32346be3487b605aee6b69b1834581d1bcc92ce71013fcea7b17feef159300", - "size": 818 + "sha256": "5fff6d863c24990435fe90e2ba9dfd9e2f60af09603eb5bd423195d5b9bbb7bc", + "size": 816 }, { "url": "https://support.claude.com/en/articles/11825384-how-to-update-claude-for-ios", @@ -16004,8 +16417,8 @@ "url": "https://support.claude.com/en/articles/11869629-use-claude-with-android-apps", "status": "success", "path": "support/11869629-use-claude-with-android-apps.md", - "sha256": "ec1b52b45229f34e53eed9841af8538ae7281c592b6e44374836fc8722572dec", - "size": 14141 + "sha256": "bb78574cd72e03f544c3e97af8a49c7bb43c03bdbdf06d2e9534b93cc75a3a01", + "size": 14139 }, { "url": "https://support.claude.com/en/articles/11932705-automated-security-reviews-in-claude-code", @@ -16039,15 +16452,15 @@ "url": "https://support.claude.com/en/articles/12005970-manage-usage-credits-for-team-and-seat-based-enterprise-plans", "status": "success", "path": "support/12005970-manage-usage-credits-for-team-and-seat-based-enterprise-plans.md", - "sha256": "4de315895024f8d08bb234e8617a4a3e69067199b4463579d68ae2c049c55c78", - "size": 9642 + "sha256": "a1ed8c78deecb6be6fb2301506cf9f28a46e1624ef3c3b3f7785225367da360b", + "size": 9638 }, { "url": "https://support.claude.com/en/articles/12012173-get-started-with-claude-in-chrome", "status": "success", "path": "support/12012173-get-started-with-claude-in-chrome.md", - "sha256": "adb94f2095c1d95483ef12570f0668de2027e3f7366ec054c1359fed0e62794c", - "size": 12674 + "sha256": "f9d5529ec9dd62e2e476f05642dc70e2967e3764b84b0aca027cf555e2b10fd3", + "size": 12676 }, { "url": "https://support.claude.com/en/articles/12053672-what-happens-to-a-user-s-data-when-they-are-removed-from-a-team-or-enterprise-organization", @@ -16074,8 +16487,8 @@ "url": "https://support.claude.com/en/articles/12111783-create-and-edit-files-with-claude", "status": "success", "path": "support/12111783-create-and-edit-files-with-claude.md", - "sha256": "b59b1ff3dcd8b4c3063e7c85bd3a9e832e4ea5b7f2cd5e9a0120f4a09335d439", - "size": 18409 + "sha256": "60b5fdf19e488f13b2ec16635af02e6b5725fd4055e1e666338567b58cc364cd", + "size": 18405 }, { "url": "https://support.claude.com/en/articles/12119250-model-safety-bug-bounty-program", @@ -16102,22 +16515,22 @@ "url": "https://support.claude.com/en/articles/12157520-claude-code-usage-analytics", "status": "success", "path": "support/12157520-claude-code-usage-analytics.md", - "sha256": "3fa94c606a10ae48bdcfa0c40ed032af0f5d80d72a4e16aeb295f4812f4c0b04", - "size": 6429 + "sha256": "199cc8dcadf2d47af087d26b4e354c11c909b3994196f4eedeb60b7b0fbf1891", + "size": 6431 }, { "url": "https://support.claude.com/en/articles/12260368-use-incognito-chats", "status": "success", "path": "support/12260368-use-incognito-chats.md", - "sha256": "514a8c22f161c1e9dae2a1f71e92cffae8c9af4efb6638a4c4518a5abf6c631a", - "size": 3596 + "sha256": "c91796d6daea91918f460fb83b1e26ca694228eb4a50b788e2a32e7c4fd7662c", + "size": 3594 }, { "url": "https://support.claude.com/en/articles/12293051-use-claude-in-xcode", "status": "success", "path": "support/12293051-use-claude-in-xcode.md", - "sha256": "38a0dac6d3a849df6c78b82596656d790166ca5fb2845317153ca305d204b955", - "size": 1907 + "sha256": "a491f8ecec05792484a995e4df78f5e9c4b9319bb96d7703b0072b6ac8ade438", + "size": 1905 }, { "url": "https://support.claude.com/en/articles/12304248-manage-api-key-environment-variables-in-claude-code", @@ -16158,22 +16571,22 @@ "url": "https://support.claude.com/en/articles/12429409-manage-usage-credits-for-paid-claude-plans", "status": "success", "path": "support/12429409-manage-usage-credits-for-paid-claude-plans.md", - "sha256": "267674da3100363ffa49894b5d3bd7b1258ff9ef0fa1a989c99f336df0be1700", - "size": 6073 + "sha256": "53852f65c0c797de1bb8181c2712b67772ddcf9dad1a2af4d17d7a97beb9a2db", + "size": 6069 }, { "url": "https://support.claude.com/en/articles/12461605-use-claude-in-slack", "status": "success", "path": "support/12461605-use-claude-in-slack.md", - "sha256": "1ae184c8bb3e848fbbfb4fb1edab9de23330cc8f4b0ff7cd3dad0683b97b5042", - "size": 9971 + "sha256": "ccbb0f6103e7ef00c2d96d9a6f0804611337e08ec6d7bced992f8bb8821f3bc5", + "size": 9973 }, { "url": "https://support.claude.com/en/articles/12466728-troubleshoot-claude-error-messages", "status": "success", "path": "support/12466728-troubleshoot-claude-error-messages.md", - "sha256": "2f04e606d6d40313cdf09e98ae49454a045025109bebf86977876b5544dbe346", - "size": 4234 + "sha256": "1136f977785e954d5188e4394515831ab51293e8173cd44e78e9f81476a8dbae", + "size": 4232 }, { "url": "https://support.claude.com/en/articles/12489464-use-enterprise-search", @@ -16193,8 +16606,8 @@ "url": "https://support.claude.com/en/articles/12512180-use-skills-in-claude", "status": "success", "path": "support/12512180-use-skills-in-claude.md", - "sha256": "bf0427e32b138eae330a32e5dea4f6d3953d8b46db1e744e51182c8bbd4c9d0c", - "size": 13285 + "sha256": "25f3f5bd7899e7438324889f0e50ae948bc14def363cfdb523d650370a16f9a4", + "size": 13283 }, { "url": "https://support.claude.com/en/articles/12512198-how-to-create-custom-skills", @@ -16214,8 +16627,8 @@ "url": "https://support.claude.com/en/articles/12592343-enabling-and-using-the-desktop-extension-allowlist", "status": "success", "path": "support/12592343-enabling-and-using-the-desktop-extension-allowlist.md", - "sha256": "fd93faf4d5112457447c9d40a5014f9dc13f733626fcace8343fcb1404e8ac47", - "size": 5712 + "sha256": "fcdd91d739f4cb3fe61b63f4796bcf84a5645f4becd19f1c5250a21419e120c2", + "size": 5708 }, { "url": "https://support.claude.com/en/articles/12611117-deploy-claude-desktop-for-macos", @@ -16228,8 +16641,8 @@ "url": "https://support.claude.com/en/articles/12618689-claude-code-on-the-web", "status": "success", "path": "support/12618689-claude-code-on-the-web.md", - "sha256": "ab32eaa81cfa67a84522ffced34aadaaff69aea35b6e2d4c269bab928e19549d", - "size": 10962 + "sha256": "83b6369cf4c368071743dada323c5e4adc5e3b1db485008f482dc11349ce84d5", + "size": 10964 }, { "url": "https://support.claude.com/en/articles/12622667-enterprise-configuration-for-claude-desktop", @@ -16249,14 +16662,14 @@ "url": "https://support.claude.com/en/articles/12626668-use-quick-entry-with-claude-desktop-on-mac", "status": "success", "path": "support/12626668-use-quick-entry-with-claude-desktop-on-mac.md", - "sha256": "722240ac6955f86f806029543ec50f895195be36e9f57888f3d7905b21f4911f", + "sha256": "066f19718260bc3226e1cea49061743c64258b17ed514de7e66e37906bedfaec", "size": 5970 }, { "url": "https://support.claude.com/en/articles/12650343-use-claude-for-excel", "status": "success", "path": "support/12650343-use-claude-for-excel.md", - "sha256": "afc26c1c15bc30145ec4241e1a8a85f3747aab334ba1d586d6ba982e33867f65", + "sha256": "1949f41e5c0b47f7883247c9b7a66e2d2b21a1630b135fbe65e33b31be96b682", "size": 20225 }, { @@ -16291,15 +16704,15 @@ "url": "https://support.claude.com/en/articles/12883420-view-usage-analytics-for-team-and-enterprise-plans", "status": "success", "path": "support/12883420-view-usage-analytics-for-team-and-enterprise-plans.md", - "sha256": "d109d0c191f247b97012397a1db6025e9a931267e9995841b3d4db9f96afac28", - "size": 12266 + "sha256": "1d9aad92df5f2922b5bdb9410993bb57a313960c8cad56f42018306da94e6297", + "size": 12262 }, { "url": "https://support.claude.com/en/articles/12893767-getting-started-with-claude-for-nonprofits", "status": "success", "path": "support/12893767-getting-started-with-claude-for-nonprofits.md", - "sha256": "c080c1c422b8f634d06922a8966308b915097a4764245f722031bae31bc3043d", - "size": 510559 + "sha256": "ef0ded1d24a1dd381ca8029b4c2004626e2622436c7285afb3d495c971ad780e", + "size": 523497 }, { "url": "https://support.claude.com/en/articles/12902405-claude-in-chrome-troubleshooting", @@ -16319,8 +16732,8 @@ "url": "https://support.claude.com/en/articles/12902446-claude-in-chrome-permissions-guide", "status": "success", "path": "support/12902446-claude-in-chrome-permissions-guide.md", - "sha256": "2aeb47a5837e587c57ff8f49574ea309755678cfd6ee371aeacfc797e5aff2c4", - "size": 8090 + "sha256": "37899373e857f7d9b629825fe08e28ad99055c5ee67be93101eaf2fa6c5721af", + "size": 8092 }, { "url": "https://support.claude.com/en/articles/12922490-remote-mcp-server-submission-guide", @@ -16347,36 +16760,36 @@ "url": "https://support.claude.com/en/articles/12923221-using-the-blackbaud-connector-in-claude", "status": "success", "path": "support/12923221-using-the-blackbaud-connector-in-claude.md", - "sha256": "0601c70ec3e4d37d3a536860c32a8940ca44ea9c437403a2a4c36c3cb42ab306", - "size": 508743 + "sha256": "6ad6e87de41ef3f68ad0c239d5de9fddba2c426295d82e60c8c889a0a0fae406", + "size": 521681 }, { "url": "https://support.claude.com/en/articles/12923227-using-the-benevity-connector-in-claude", "status": "success", "path": "support/12923227-using-the-benevity-connector-in-claude.md", - "sha256": "a4d186d5a428f192c6208b257491c85d5bad123b83db71beaacc6da2e2b4b130", - "size": 507373 + "sha256": "2f80622876f0bedbabd8f9dfc55e5753343f586a156f79292dc11a47b81ab62b", + "size": 520311 }, { "url": "https://support.claude.com/en/articles/12923235-using-the-candid-connector-in-claude", "status": "success", "path": "support/12923235-using-the-candid-connector-in-claude.md", - "sha256": "8823b4b10c7e82754961d0da92a6aede9ae5c345076ffc9fe744f7b49a719edc", - "size": 510308 + "sha256": "9edfd16af3532b679f61e644033faf4cbf7a79167a7294cad7cd1dc9a9eaff15", + "size": 523246 }, { "url": "https://support.claude.com/en/articles/12923668-claude-for-nonprofits-partnership-success-guide-for-admins", "status": "success", "path": "support/12923668-claude-for-nonprofits-partnership-success-guide-for-admins.md", - "sha256": "bed12821003373a660ad5b1d3c766d9d45918027b4278cfeec1841af9b3bac3d", - "size": 509376 + "sha256": "fdfef9dba7893940d7cdbc254381e6d080a2946d8bc50b9189166b78e0083877", + "size": 522314 }, { "url": "https://support.claude.com/en/articles/12923901-claude-for-nonprofits-partnership-guide-for-all-users", "status": "success", "path": "support/12923901-claude-for-nonprofits-partnership-guide-for-all-users.md", - "sha256": "143747934aec6b65edca2d7b56cbe76e463255d740468471d90d59fd7ce7f1e8", - "size": 508700 + "sha256": "6b1fc7dd2dd0aabe3486ba6875252700b91cd792cf8d5cca758d018db71a06ba", + "size": 521638 }, { "url": "https://support.claude.com/en/articles/12938627-how-to-gift-a-claude-subscription", @@ -16403,8 +16816,8 @@ "url": "https://support.claude.com/en/articles/12997503-team-plan-billing-faqs", "status": "success", "path": "support/12997503-team-plan-billing-faqs.md", - "sha256": "1e366e63a9b0347810ad84d59ca3ddda590c07745eaaf80534faf235537a136d", - "size": 3926 + "sha256": "0c99156388f5d95a51d82a6f3820efeaf13e43a20a8913afc11901b80ba0be74", + "size": 3922 }, { "url": "https://support.claude.com/en/articles/13015708-access-the-compliance-api", @@ -16452,15 +16865,15 @@ "url": "https://support.claude.com/en/articles/13132885-set-up-single-sign-on-sso", "status": "success", "path": "support/13132885-set-up-single-sign-on-sso.md", - "sha256": "4c232e0f291119a50505fb488dad4e685882896e700fa2fefea324bafe9e5b35", - "size": 12319 + "sha256": "34f8f4b5235c76ff2d21b30f6cd33d1a1ca7a3eeb70e71e36b34bbbeb8a47c12", + "size": 12315 }, { "url": "https://support.claude.com/en/articles/13133195-set-up-jit-or-scim-provisioning", "status": "success", "path": "support/13133195-set-up-jit-or-scim-provisioning.md", - "sha256": "7c9cc663f252c41b1866aefe7f14f8f14ac6fa2ba8498c01b99cc0f2053715b0", - "size": 16604 + "sha256": "58ceefdcc8fc717b810c6c5145730275c79380aef2bab7b49e5708961c24fe6d", + "size": 16612 }, { "url": "https://support.claude.com/en/articles/13133750-manage-members-on-team-and-enterprise-plans", @@ -16487,8 +16900,8 @@ "url": "https://support.claude.com/en/articles/13163631-configuring-session-security-settings", "status": "success", "path": "support/13163631-configuring-session-security-settings.md", - "sha256": "46df1acb20474ff880888eda10c8ba3db122e3358ab6e5e327cb56cd07acc778", - "size": 3700 + "sha256": "a72f08f9d26b0e71164c46ebb6067444ce0ced7e5f3af74315de6bf2bb556f04", + "size": 3698 }, { "url": "https://support.claude.com/en/articles/13163666-holiday-2025-usage-promotion", @@ -16508,8 +16921,8 @@ "url": "https://support.claude.com/en/articles/13189465-log-in-to-your-claude-account", "status": "success", "path": "support/13189465-log-in-to-your-claude-account.md", - "sha256": "f879459af66ea65050b504510befcde38e320909564b04d7b6b2b547428b7bdc", - "size": 7036 + "sha256": "a7df4c48662102f232101fe82923a04bc10b167b0c785ad9d5f425aace1c14dc", + "size": 7038 }, { "url": "https://support.claude.com/en/articles/13198485-enforce-network-level-access-control-with-tenant-restrictions", @@ -16536,21 +16949,21 @@ "url": "https://support.claude.com/en/articles/13325567-account-management-faqs", "status": "success", "path": "support/13325567-account-management-faqs.md", - "sha256": "b6a5c42db1ba5bf75768403fd7e55cf02b0500656ebb39d54615bf4d832d7e2f", - "size": 2637 + "sha256": "6ba5182a459490189f0c62ed530d5af3adb239955f3546c411c84c2400ceb360", + "size": 2641 }, { "url": "https://support.claude.com/en/articles/13345190-get-started-with-claude-cowork", "status": "success", "path": "support/13345190-get-started-with-claude-cowork.md", - "sha256": "d19039098c5078eecb8c92de2bb30f28a8527dc3d2fe330506247d5e2c1a3056", - "size": 20039 + "sha256": "1c37afe2348623a96752a31d78d4b1894c715f329f167ac585f45d90e413f7a2", + "size": 20037 }, { "url": "https://support.claude.com/en/articles/13346458-customizing-your-console-appearance-settings", "status": "success", "path": "support/13346458-customizing-your-console-appearance-settings.md", - "sha256": "8366a3c3d0c74c676128974bc2121968e8623be3b9725aeb57987dc53de715c5", + "sha256": "b1d351abce17b2d3c909cb8b34f543237dce9fcaf80352d3877b5117e80deca9", "size": 605 }, { @@ -16571,8 +16984,8 @@ "url": "https://support.claude.com/en/articles/13371040-logging-in-to-your-console-account", "status": "success", "path": "support/13371040-logging-in-to-your-console-account.md", - "sha256": "a84223df0af9e54dbeacae7db224ddc940a0f8461986da546b15840076cff1dc", - "size": 4598 + "sha256": "0c32ff18b1628a19d3f379b057ff568e9e51a1999a5ff5287cd2cf24e21157f7", + "size": 4596 }, { "url": "https://support.claude.com/en/articles/13393991-purchase-and-manage-seats-on-enterprise-plans", @@ -16634,7 +17047,7 @@ "url": "https://support.claude.com/en/articles/13641943-visual-and-interactive-content", "status": "success", "path": "support/13641943-visual-and-interactive-content.md", - "sha256": "a38773904e301ae652c152b6fc2979b50e91a46c106008b20d96ca092d7f3edc", + "sha256": "8b3b2dc8de19a0784d4a457b8497ea24c4edf20ba53489e49754294efdfc924d", "size": 6511 }, { @@ -16662,8 +17075,8 @@ "url": "https://support.claude.com/en/articles/13756069-public-sector-faqs", "status": "success", "path": "support/13756069-public-sector-faqs.md", - "sha256": "ed99188441a0dadc80fc2ba0709412c359185a8d62ab9ab48caee815d297a4ab", - "size": 8380 + "sha256": "33b6da6fe563dac526797fc807b55702e196a763359e5fe29d7374278c1c46b3", + "size": 8376 }, { "url": "https://support.claude.com/en/articles/13776697-join-an-organization-via-invite-link", @@ -16690,29 +17103,29 @@ "url": "https://support.claude.com/en/articles/13837433-manage-plugins-for-your-organization", "status": "success", "path": "support/13837433-manage-plugins-for-your-organization.md", - "sha256": "f406b167ce1e1b32c1e0a1f66160966148df213dd264036cfdf63862d5e92298", - "size": 19757 + "sha256": "f7815ffe54ad5ad1ec7e09ff0159caa2855721cf6aef76cea054d70ed1451222", + "size": 19759 }, { "url": "https://support.claude.com/en/articles/13837440-use-plugins-in-claude", "status": "success", "path": "support/13837440-use-plugins-in-claude.md", - "sha256": "4c57d65e863ebc2512a57db6d93d788004e21db0c7b328cf683c8755c5a73460", + "sha256": "a95f0410944487604c1a962b582bd019b86ddd506cd8c9ae9e29b4fbc2a8c004", "size": 6360 }, { "url": "https://support.claude.com/en/articles/13854387-schedule-recurring-tasks-in-claude-cowork", "status": "success", "path": "support/13854387-schedule-recurring-tasks-in-claude-cowork.md", - "sha256": "f37a27591135bedec2f2b88ef51cd27c7f0d7f0910bc902e0addd04da388f359", - "size": 4698 + "sha256": "69337584b3ca43c6fce378537ff42889ba4b1da8080816a686721b73c080c429", + "size": 4704 }, { "url": "https://support.claude.com/en/articles/13892150-work-across-microsoft-365-apps", "status": "success", "path": "support/13892150-work-across-microsoft-365-apps.md", - "sha256": "b9f2cf98fd1e4ec1aba5c7ea12ba3bfba57c5d2f8796ca4bd9c18db8f1c85480", - "size": 6344 + "sha256": "08f3a72a7da7887d25a53ef9c0b84bac0d46441b1ddda249d2a663cce96f56ad", + "size": 6338 }, { "url": "https://support.claude.com/en/articles/13917817-google-workspace-sso-scim-email-mismatch", @@ -16795,8 +17208,8 @@ "url": "https://support.claude.com/en/articles/13930458-set-up-role-based-permissions-on-enterprise-plans", "status": "success", "path": "support/13930458-set-up-role-based-permissions-on-enterprise-plans.md", - "sha256": "ab9cf62a977b829c4847a91aa66ce0d7028143e01974bcfd915338d55a455bb8", - "size": 40970 + "sha256": "c33cc04f1028a1619b96da2eed70ca36f266e0eb6795362e06850176f18d05d5", + "size": 40978 }, { "url": "https://support.claude.com/en/articles/13945233-use-claude-for-microsoft-365-with-third-party-platforms", @@ -16809,8 +17222,8 @@ "url": "https://support.claude.com/en/articles/13947068-assign-tasks-from-anywhere-in-claude-cowork", "status": "success", "path": "support/13947068-assign-tasks-from-anywhere-in-claude-cowork.md", - "sha256": "2fe142b7d4d2495d462b5630113b80f7b14eeea4ab741d652b85c867de74c2ac", - "size": 8276 + "sha256": "620dea7d175ffc4a6e86e1822eb0168974c4f95a809817074a37d31974dddd01", + "size": 8278 }, { "url": "https://support.claude.com/en/articles/13979539-custom-visuals-in-chat-and-cowork", @@ -16830,15 +17243,15 @@ "url": "https://support.claude.com/en/articles/14116274-organize-your-tasks-with-projects-in-claude-cowork", "status": "success", "path": "support/14116274-organize-your-tasks-with-projects-in-claude-cowork.md", - "sha256": "b9a607b425446a6f065b534fa69b6a8c51ebba0e2f03d1a0cfe600be03023f52", - "size": 5712 + "sha256": "a98d2bcc5aa8c0b053890d943a9da254400edd4f8334283d95f4818b35400c45", + "size": 5704 }, { "url": "https://support.claude.com/en/articles/14128542-let-claude-use-your-computer-in-cowork", "status": "success", "path": "support/14128542-let-claude-use-your-computer-in-cowork.md", - "sha256": "18cc3a290e71b6329399fd10e07fa23bda00624b218039e5eac6fedfaaeb3e6e", - "size": 8282 + "sha256": "5c832776c41e9fb46e3a43903f5140fec0410b43a924dfe3145265e854d1c6b5", + "size": 8284 }, { "url": "https://support.claude.com/en/articles/14128775-claude-code-on-console-to-enterprise-migration", @@ -16914,8 +17327,8 @@ "url": "https://support.claude.com/en/articles/14499648-how-scim-sync-works-for-enterprise-organizations", "status": "success", "path": "support/14499648-how-scim-sync-works-for-enterprise-organizations.md", - "sha256": "51d9d9a52cdf8a1afff1cd6c395bdcdef36b8d0980106ce12e0e721aeb17d31e", - "size": 7437 + "sha256": "fb0a0492938227316f950d4235140daf9441a27165fd82e8b991b2ca925cabf6", + "size": 7441 }, { "url": "https://support.claude.com/en/articles/14503520-available-beta-and-research-preview-features", @@ -16935,15 +17348,15 @@ "url": "https://support.claude.com/en/articles/14503613-sso-login", "status": "success", "path": "support/14503613-sso-login.md", - "sha256": "ca30b5f9aa8520455d742b97b7275b52b8c3e95f2c66b85a8af5d786c935a688", + "sha256": "8333febf2a7026f2ef18c1b2fd2fdc829c59790642b07651b4399932290c0130", "size": 6690 }, { "url": "https://support.claude.com/en/articles/14503643-set-up-scim-in-claude-for-government", "status": "success", "path": "support/14503643-set-up-scim-in-claude-for-government.md", - "sha256": "e1559b56c3356cf484f9248953a24cd6b05baea39be02104dadfe3296656b034", - "size": 6404 + "sha256": "99d069f38beb751ff41f5316b129711ee8e7270d786d28a1c453590a3b5d2906", + "size": 6412 }, { "url": "https://support.claude.com/en/articles/14503675-organization-instructions-in-claude-for-government", @@ -16970,7 +17383,7 @@ "url": "https://support.claude.com/en/articles/14503775-mcp-web-search", "status": "success", "path": "support/14503775-mcp-web-search.md", - "sha256": "037637a48e5040cf46d885cb850d30f9152a0cc1b5898f91cf9cae3b6ece6e7b", + "sha256": "a1824252d63629143cb69f909a6c9bf36c16a79cc1dd8e2767bc645cacdd8d93", "size": 4675 }, { @@ -17047,8 +17460,8 @@ "url": "https://support.claude.com/en/articles/14554922-claude-code-user-faq", "status": "success", "path": "support/14554922-claude-code-user-faq.md", - "sha256": "49aed0b1014f469c53149001e77d8621543720c5c6bddd7787f1c15e52b1a5be", - "size": 28483 + "sha256": "35c7d398a86b863673e676dd5b695e8750b8e8ebbe6ec415bcd12bc3ce0f17b3", + "size": 27153 }, { "url": "https://support.claude.com/en/articles/14555399-claude-code-champion-kit", @@ -17068,22 +17481,22 @@ "url": "https://support.claude.com/en/articles/14604397-set-up-your-design-system-in-claude-design", "status": "success", "path": "support/14604397-set-up-your-design-system-in-claude-design.md", - "sha256": "2f907ed419a552f848def45421777194166ba9eaf582831e8462fae8d71c637b", - "size": 4401 + "sha256": "d6ff1611f7504ad53ddd2b25c7eb7e5d6f4c81db2c9ca4bf2a11b6ccfd31cc48", + "size": 4393 }, { "url": "https://support.claude.com/en/articles/14604406-claude-design-admin-guide-for-team-and-enterprise-plans", "status": "success", "path": "support/14604406-claude-design-admin-guide-for-team-and-enterprise-plans.md", - "sha256": "604f8aaa56a0c181af3334d23fdf31830e65753ae2c53e5f647a1bb29706d7e6", - "size": 12242 + "sha256": "6e3ea0e2b16ca9df15e32ff7acfce6d0ff3cfea4de36135cd6bedf7e86685597", + "size": 12238 }, { "url": "https://support.claude.com/en/articles/14604416-get-started-with-claude-design", "status": "success", "path": "support/14604416-get-started-with-claude-design.md", - "sha256": "ca63b917c5501cfd2f97267c500a18d4453e8bd7254b2ddc355f5bac33f2acb2", - "size": 11128 + "sha256": "9efb0b4cebfbb79bcbb7d607be4d9ca440dd7c64a56de1057209834703888c98", + "size": 11130 }, { "url": "https://support.claude.com/en/articles/14604842-real-time-cyber-safeguards-on-claude-opus-and-sonnet", @@ -17180,8 +17593,8 @@ "url": "https://support.claude.com/en/articles/15167101-get-started-with-claude-compliance-api-integrations", "status": "success", "path": "support/15167101-get-started-with-claude-compliance-api-integrations.md", - "sha256": "c76bdf864359fe53813a00b2dfc3e618b555e4e0814344e350e4195b831b7643", - "size": 33018 + "sha256": "4064e9b32027746fdf129d5a64d249e1dc9b2c63b5cfb0179a4f1e1bbc48351c", + "size": 44121 }, { "url": "https://support.claude.com/en/articles/15171100-age-assurance-on-claude", @@ -17215,8 +17628,8 @@ "url": "https://support.claude.com/en/articles/15330088-set-a-default-model-for-your-organization", "status": "success", "path": "support/15330088-set-a-default-model-for-your-organization.md", - "sha256": "d09b106b667d4ceea92ea5985085a2e77bf04dc027bbb7acda9d084bab7b50df", - "size": 5727 + "sha256": "59486ba574cd70363be6ce362e7e6ceb16be415aa4b1d1286ae30820c3473fca", + "size": 5731 }, { "url": "https://support.claude.com/en/articles/15330651-claude-enterprise-admin-api-reference-guide", @@ -17327,8 +17740,8 @@ "url": "https://support.claude.com/en/articles/15694740-manage-model-access-for-your-organization", "status": "success", "path": "support/15694740-manage-model-access-for-your-organization.md", - "sha256": "409341290810705e80c5c447999e32b43c5249267f7b9965fa47a96768895b4d", - "size": 8494 + "sha256": "b46e56c4f43fc00cded106ca26b730acde575c2f334c5aec1e5467d2b65578a2", + "size": 8498 }, { "url": "https://support.claude.com/en/articles/15707726-using-claude-for-legal-work-privilege-confidentiality-and-how-to-think-about-configuration", @@ -17355,8 +17768,8 @@ "url": "https://support.claude.com/en/articles/15936181-get-started-with-1password-for-claude", "status": "success", "path": "support/15936181-get-started-with-1password-for-claude.md", - "sha256": "9f2735277cccb4f24766bef3e6061c23e29b21f869ce7a4ac9ad36e8cfb78077", - "size": 5060 + "sha256": "328bdd38d0a599ef2c9e8a1a9f421626ed8bf6dde0e58a082633dcd4a7312365", + "size": 5058 }, { "url": "https://raw.githubusercontent.com/anthropics/claude-cookbooks/main/.claude/agents/code-reviewer.md", @@ -19378,8 +19791,8 @@ "url": "https://raw.githubusercontent.com/anthropics/claude-plugins-official/main/.claude-plugin/marketplace.json", "status": "success", "path": "github/claude-plugins-official/.claude-plugin/marketplace.json", - "sha256": "89649dbc0329a60265bb484481f08936c04c5761e29b7078511e38e3ecaecb14", - "size": 149584 + "sha256": "5cbaadbc182eb9a23002140842dd25127ec0b0722c985e70563cac89fcafdb03", + "size": 149976 }, { "url": "https://raw.githubusercontent.com/anthropics/claude-plugins-official/main/.github/policy/prompt.md", @@ -21205,8 +21618,8 @@ "url": "https://raw.githubusercontent.com/anthropics/claude-plugins-official/main/plugins/receipts/README.md", "status": "success", "path": "github/claude-plugins-official/plugins/receipts/README.md", - "sha256": "f951f4b2e729ab9ac963d19f5375341cfb72cbfdf163a6854f4686f4958aef00", - "size": 6231 + "sha256": "9a495b3c48f9123f73475986c84400923e495b2ff49e3519d28a4864c1ae80f2", + "size": 3648 }, { "url": "https://raw.githubusercontent.com/anthropics/claude-plugins-official/main/plugins/receipts/skills/receipts/SKILL.md", @@ -22689,8 +23102,8 @@ "url": "https://raw.githubusercontent.com/anthropics/anthropic-sdk-typescript/main/CHANGELOG.md", "status": "success", "path": "github/anthropic-sdk-typescript/CHANGELOG.md", - "sha256": "876371edac71b4f39887dd9d4f147ad9c5963cd27d49a37e6510e414fcb4e68e", - "size": 193700 + "sha256": "bdf82347f622f47f6d42ad095555dad5a5e054c1ac892c26b8d7b08b808c38a4", + "size": 194295 }, { "url": "https://raw.githubusercontent.com/anthropics/anthropic-sdk-typescript/main/CLAUDE.md", @@ -22752,8 +23165,8 @@ "url": "https://raw.githubusercontent.com/anthropics/anthropic-sdk-typescript/main/packages/aws-sdk/CHANGELOG.md", "status": "success", "path": "github/anthropic-sdk-typescript/packages/aws-sdk/CHANGELOG.md", - "sha256": "e3f582b6b0a0b34601c49ffc34dc92728b5c22a223c291d023b3081e5ca518d5", - "size": 6137 + "sha256": "b64eaf2843428c5014e55e8ddddd3ffb251f19b825a604e2dd918274dce86b99", + "size": 6838 }, { "url": "https://raw.githubusercontent.com/anthropics/anthropic-sdk-typescript/main/packages/aws-sdk/README.md", @@ -22856,8 +23269,8 @@ ], "failures": [], "summary": { - "total": 3264, - "downloaded": 3264, + "total": 3323, + "downloaded": 3323, "skipped": 0, "failed": 0, "success_rate": 100.0 diff --git a/content/CHANGELOG.md b/content/CHANGELOG.md index dccd1b049..835feec72 100644 --- a/content/CHANGELOG.md +++ b/content/CHANGELOG.md @@ -1,5 +1,48 @@ # Changelog +## 2.1.216 + +- Added `sandbox.filesystem.disabled` setting to skip filesystem isolation while keeping network egress control +- Fixed a slowdown in long sessions where message normalization cost grew quadratically with the number of turns, causing multi-second stalls and slow resumes +- Fixed auto mode denying commands with "HTTP 401" classifier errors after the OAuth token expired or rotated mid-session +- Fixed AskUserQuestion telling Claude to continue even when your answer asked it to wait or explain first — free-text answers now get neutral wording +- Fixed Claude Code on the web re-asking the same question and dropping your answer after the session sat idle for a few minutes +- Fixed @-mentions silently attaching nothing after file-modifying hooks, vim dot-repeat of `c`-operators and paste, statusline running twice on resume, and resume-picker hangs on failure +- Fixed resumed background agent sessions reverting to the default agent: the agent's prompt and tool restrictions are now restored +- Fixed worktree-isolated subagents redirecting git into the shared checkout via `git -C`, `--git-dir`, or `GIT_DIR`/`GIT_WORK_TREE` +- Fixed worktree sessions landing in another project's leftover worktree when the working directory did not match the selected project +- Fixed background sessions whose worktree has no git repository being undeletable +- Fixed `claude daemon stop --any` potentially terminating an unrelated process via a stale legacy daemon lockfile +- Fixed Esc-Esc at an idle prompt not opening the rewind picker in long-running sessions with background tasks +- Fixed Bash command permission checking for compound statements with redirects inside `&&` lists or negations +- Fixed pressing Ctrl+X twice in the agent list failing to delete a session, and deleted sessions reappearing when their background worker had died +- Fixed background subagents getting cancelled when a high-priority message arrives during their startup window +- Fixed mouse and focus garbage in the terminal while a GUI editor from `/memory`, `/plan`, `/keybindings`, or Ctrl+G is open; `/memory` no longer waits for the editor to close +- Fixed Claude-in-Chrome 403-looping on reconnect when the session's OAuth token lacks a required scope +- Fixed workflow saves and scheduled-task writes following a symlink at `.claude`, which could redirect writes outside the project +- Fixed MCP re-authenticate revoking working credentials before the new sign-in succeeds, and the reconnect needs-auth message in background sessions pointing at an unusable command +- Fixed read-only commands on Windows accessing network paths without a permission prompt +- Fixed Bash command parsing of non-ASCII characters to match real shell word boundaries +- Fixed PowerShell tool permission validation of commands containing invisible Unicode characters +- Fixed dialogs in fullscreen mode stretching past the right-hand edge of their panel +- Fixed the `/config` settings list in fullscreen mode clipping its keyboard-hint footer +- Fixed the transcript-mode (Ctrl+O) footer hint wrapping on terminals narrower than 104 columns +- Fixed the Prometheus metrics endpoint (`OTEL_METRICS_EXPORTER=prometheus`) emitting invalid `# UNIT` lines +- Fixed skills and commands changed during a session not appearing in the slash menu until restart +- Fixed plugin skills with a `name` frontmatter field losing their plugin prefix in slash-command autocomplete +- Fixed telemetry misreporting permission denials: failed permission-prompt requests no longer count as user rejections, and user interrupts are now reported as user aborts instead of rejections +- Improved the `/fork` confirmation to one line with the new session's name, `claude attach` id, and a note when the copy shares your checkout +- Improved validation of `git` and `gh` command arguments in the PowerShell tool +- Improved the `/ultrareview` diff-too-large error to show configured limits, measured diff size, and largest contributing files +- Improved `/code-review ultra` empty-diff message to name the exact base ref and suggest passing an explicit base +- Improved the spend limit adjustment prompt to show the server's reason when a spend limit change is rejected +- `/context` now shows an explicit warning when the conversation exceeds the context window, and a failed `/compact` displays as an error +- `/rewind` no longer restores or deletes files through symlinks or hard links at tracked paths and reports how many paths it skipped +- Background sessions: `/mcp` and `/install-github-app` now park a "needs input" request in the agent view when no client is attached +- Updated the bundled dataviz skill: reordered the default chart palette and fixed guidance that suggested direct labels for four-series charts +- [VSCode] Fixed right-to-left text (Arabic, Hebrew, Persian) rendering in the wrong order when mixed with English or code +- Fixed cloud sessions dropping the in-flight message when the session's container restarts mid-turn — the interrupted turn now re-runs on resume instead of leaving the session unresponsive + ## 2.1.215 - Claude no longer runs the `/verify` and `/code-review` skills on its own; invoke them with `/verify` or `/code-review` when you want them diff --git a/content/claude-code-manifest.json b/content/claude-code-manifest.json index 59d55ec6b..802d1fdca 100644 --- a/content/claude-code-manifest.json +++ b/content/claude-code-manifest.json @@ -1,12 +1,12 @@ { "name": "@anthropic-ai/claude-code", - "version": "2.1.215", + "version": "2.1.216", "author": { "name": "Anthropic", "email": "support@anthropic.com" }, "license": "SEE LICENSE IN README.md", - "_id": "@anthropic-ai/claude-code@2.1.215", + "_id": "@anthropic-ai/claude-code@2.1.216", "maintainers": [ { "name": "zak-anthropic", @@ -69,20 +69,20 @@ "claude": "bin/claude.exe" }, "dist": { - "shasum": "c349d13afa43e6cf6f147f36bec121dc9824ed89", - "tarball": "https://registry.npmjs.org/@anthropic-ai/claude-code/-/claude-code-2.1.215.tgz", + "shasum": "85e15d4dfcc3e044c2769cec5866a703af780adb", + "tarball": "https://registry.npmjs.org/@anthropic-ai/claude-code/-/claude-code-2.1.216.tgz", "fileCount": 7, - "integrity": "sha512-lsWBvyMyBqg/rOZ06o/HEhlpOzHsH8IBf9RH4u5gzizQ1LWS/oXAh5nqWNUu3hn2d8QkyYg0aseNzMDlGD+3qg==", + "integrity": "sha512-UH4XncIO6ceNZMq2HxphEn9YR7uu5NGVJMcogGSPvu0nXRw8ubEjbu5cD3Dhvx3vXxizwFA82tmk6mVXjhr/dw==", "signatures": [ { - "sig": "MEUCIDMhAnHnwQOkVShX93Vrlg1V0yWFmKoxXZEO2vaxoF2JAiEAiIPDb+5n1XvwFBJcQP3COvZVkZ8qfqzEFP0W4VmUOjk=", + "sig": "MEQCICeuyK6OFbcmHd1c+dAT+C+gS0irexcb6PurkSrLPowbAiAWx0f6pI6RcLM5QQEqB5nkdHtZQjM+8CLc5RDrPnExfA==", "keyid": "SHA256:DhQ8wR5APBvFHLF/+Tc+AYvPOdTpcIDqOhxsBHRwC7U" } ], - "unpackedSize": 165282 + "unpackedSize": 165315 }, "type": "module", - "_from": "file:staged-npm/anthropic-ai-claude-code-2.1.215.tgz", + "_from": "file:staged-npm/anthropic-ai-claude-code-2.1.216.tgz", "engines": { "node": ">=22.0.0" }, @@ -94,8 +94,8 @@ "name": "wolffiex", "email": "wolffiex@anthropic.com" }, - "_resolved": "/home/runner/work/claude-cli-internal/claude-cli-internal/staged-npm/anthropic-ai-claude-code-2.1.215.tgz", - "_integrity": "sha512-lsWBvyMyBqg/rOZ06o/HEhlpOzHsH8IBf9RH4u5gzizQ1LWS/oXAh5nqWNUu3hn2d8QkyYg0aseNzMDlGD+3qg==", + "_resolved": "/home/runner/work/claude-cli-internal/claude-cli-internal/staged-npm/anthropic-ai-claude-code-2.1.216.tgz", + "_integrity": "sha512-UH4XncIO6ceNZMq2HxphEn9YR7uu5NGVJMcogGSPvu0nXRw8ubEjbu5cD3Dhvx3vXxizwFA82tmk6mVXjhr/dw==", "_npmVersion": "11.16.0", "description": "Use Claude, Anthropic's AI assistant, right from your terminal. Claude can understand your codebase, edit files, run terminal commands, and handle entire workflows for you.", "directories": {}, @@ -104,17 +104,17 @@ "_hasShrinkwrap": false, "readmeFilename": "README.md", "optionalDependencies": { - "@anthropic-ai/claude-code-linux-x64": "2.1.215", - "@anthropic-ai/claude-code-win32-x64": "2.1.215", - "@anthropic-ai/claude-code-darwin-x64": "2.1.215", - "@anthropic-ai/claude-code-linux-arm64": "2.1.215", - "@anthropic-ai/claude-code-win32-arm64": "2.1.215", - "@anthropic-ai/claude-code-darwin-arm64": "2.1.215", - "@anthropic-ai/claude-code-linux-x64-musl": "2.1.215", - "@anthropic-ai/claude-code-linux-arm64-musl": "2.1.215" + "@anthropic-ai/claude-code-linux-x64": "2.1.216", + "@anthropic-ai/claude-code-win32-x64": "2.1.216", + "@anthropic-ai/claude-code-darwin-x64": "2.1.216", + "@anthropic-ai/claude-code-linux-arm64": "2.1.216", + "@anthropic-ai/claude-code-win32-arm64": "2.1.216", + "@anthropic-ai/claude-code-darwin-arm64": "2.1.216", + "@anthropic-ai/claude-code-linux-x64-musl": "2.1.216", + "@anthropic-ai/claude-code-linux-arm64-musl": "2.1.216" }, "_npmOperationalInternal": { - "tmp": "tmp/claude-code_2.1.215_1784422417038_0.7779043961753329", + "tmp": "tmp/claude-code_2.1.216_1784578777312_0.3890477303128479", "host": "s3://npm-registry-packages-npm-production" } } \ No newline at end of file diff --git a/content/en/about-claude/glossary.md b/content/en/about-claude/glossary.md index 7948287db..0294820ea 100644 --- a/content/en/about-claude/glossary.md +++ b/content/en/about-claude/glossary.md @@ -1,26 +1,26 @@ # Glossary -These concepts are not unique to Anthropic’s language models, but we present a brief summary of key terms below. +These concepts are not unique to Anthropic’s language models, but this page presents a brief summary of key terms. --- ## Context window -The "context window" refers to the amount of text a language model can look back on and reference when generating new text. This is different from the large corpus of data the language model was trained on, and instead represents a "working memory" for the model. A larger context window allows the model to understand and respond to more complex and lengthy prompts, while a smaller context window may limit the model's ability to handle longer prompts or maintain coherence over extended conversations. +The "context window" refers to the amount of text a language model can look back on and reference when generating new text. This is different from the large corpus of data the language model was trained on, and instead represents a "working memory" for the model. A larger context window allows the model to process and respond to more complex and lengthy prompts, while a smaller context window may limit the model's ability to handle longer prompts or maintain coherence over extended conversations. -See our [guide to understanding context windows](/docs/en/build-with-claude/context-windows) to learn more. +See the [guide to understanding context windows](/docs/en/build-with-claude/context-windows) to learn more. ## Fine-tuning -Fine-tuning is the process of further training a pretrained language model using additional data. This causes the model to start representing and mimicking the patterns and characteristics of the fine-tuning dataset. Claude is not a bare language model; it has already been fine-tuned to be a helpful assistant. Our API does not currently offer fine-tuning, but please ask your Anthropic contact if you are interested in exploring this option. Fine-tuning can be useful for adapting a language model to a specific domain, task, or writing style, but it requires careful consideration of the fine-tuning data and the potential impact on the model's performance and biases. +Fine-tuning is the process of further training a pretrained language model using additional data. This causes the model to start representing and mimicking the patterns and characteristics of the fine-tuning dataset. Claude is not a bare language model; it has already been fine-tuned to be a helpful assistant. The Claude API does not currently offer fine-tuning, but ask your Anthropic contact if you are interested in exploring this option. Fine-tuning can be useful for adapting a language model to a specific domain, task, or writing style, but it requires careful consideration of the fine-tuning data and the potential impact on the model's performance and biases. ## HHH These three H's represent Anthropic's goals in ensuring that Claude is beneficial to society: -- A **helpful** AI will attempt to perform the task or answer the question posed to the best of its abilities, providing relevant and useful information. -- An **honest** AI will give accurate information, and not hallucinate or confabulate. It will acknowledge its limitations and uncertainties when appropriate. -- A **harmless** AI will not be offensive or discriminatory, and when asked to aid in a dangerous or unethical act, the AI should politely refuse and explain why it cannot comply. +* A **helpful** AI will attempt to perform the task or answer the question posed to the best of its abilities, providing relevant and useful information. +* An **honest** AI will give accurate information, and not hallucinate or confabulate. It will acknowledge its limitations and uncertainties when appropriate. +* A **harmless** AI will not be offensive or discriminatory, and when asked to aid in a dangerous or unethical act, the AI should politely refuse and explain why it cannot comply. ## Latency @@ -32,19 +32,19 @@ Large language models (LLMs) are AI language models with many parameters that ar ## MCP (Model Context Protocol) -Model Context Protocol (MCP) is an open protocol that standardizes how applications provide context to LLMs. Like a USB-C port for AI applications, MCP provides a unified way to connect AI models to different data sources and tools. MCP enables AI systems to maintain consistent context across interactions and access external resources in a standardized manner. See our [MCP documentation](/docs/en/mcp) to learn more. +Model Context Protocol (MCP) is an open protocol that standardizes how applications provide context to LLMs. Like a USB-C port for AI applications, MCP provides a unified way to connect AI models to different data sources and tools. MCP enables AI systems to maintain consistent context across interactions and access external resources in a standardized manner. See the [MCP documentation](/docs/en/mcp) to learn more. ## MCP connector -The MCP connector is a feature that allows API users to connect to MCP servers directly from the Messages API without building an MCP client. This enables seamless integration with MCP-compatible tools and services through the Claude API. The MCP connector supports features like tool calling and is available in beta. See the [MCP connector documentation](/docs/en/agents-and-tools/mcp-connector) to learn more. +The MCP connector is a feature that allows API users to connect to MCP servers directly from the Messages API without building an MCP client. This enables seamless integration with MCP-compatible tools and services through the Claude API. The MCP connector supports features such as tool calling and is available in beta. See the [MCP connector](/docs/en/agents-and-tools/mcp-connector) documentation to learn more. ## Pretraining -Pretraining is the initial process of training language models on a large unlabeled corpus of text. In Claude's case, autoregressive language models (like Claude's underlying model) are pretrained to predict the next word, given the previous context of text in the document. These pretrained models are not inherently good at answering questions or following instructions, and often require deep skill in prompt engineering to elicit desired behaviors. Fine-tuning and RLHF are used to refine these pretrained models, making them more useful for a wide range of tasks. +Pretraining is the initial process of training language models on a large unlabeled corpus of text. In Claude's case, autoregressive language models (such as Claude's underlying model) are pretrained to predict the next word, given the previous context of text in the document. These pretrained models are not inherently good at answering questions or following instructions, and often require deep skill in prompt engineering to elicit desired behaviors. Fine-tuning and RLHF are used to refine these pretrained models, making them more useful for a wide range of tasks. ## RAG (Retrieval augmented generation) -Retrieval augmented generation (RAG) is a technique that combines information retrieval with language model generation to improve the accuracy and relevance of the generated text, and to better ground the model's response in evidence. In RAG, a language model is augmented with an external knowledge base or a set of documents that is passed into the context window. The data is retrieved at run time when a query is sent to the model, although the model itself does not necessarily retrieve the data (but can with [tool use](/docs/en/agents-and-tools/tool-use/overview) and a retrieval function). When generating text, relevant information first must be retrieved from the knowledge base based on the input prompt, and then passed to the model along with the original query. The model uses this information to guide the output it generates. This allows the model to access and utilize information beyond its training data, reducing the reliance on memorization and improving the factual accuracy of the generated text. RAG can be particularly useful for tasks that require up-to-date information, domain-specific knowledge, or explicit citation of sources. However, the effectiveness of RAG depends on the quality and relevance of the external knowledge base and the knowledge that is retrieved at runtime. +Retrieval augmented generation (RAG) is a technique that combines information retrieval with language model generation to improve the accuracy and relevance of the generated text, and to better ground the model's response in evidence. In RAG, a language model is augmented with an external knowledge base or a set of documents that is passed into the context window. The data is retrieved at runtime when a query is sent to the model, although the model itself does not necessarily retrieve the data (but can with [tool use](/docs/en/agents-and-tools/tool-use/overview) and a retrieval function). When generating text, relevant information first must be retrieved from the knowledge base based on the input prompt, and then passed to the model along with the original query. The model uses this information to guide the output it generates. This allows the model to access and use information beyond its training data, reducing the reliance on memorization and improving the factual accuracy of the generated text. RAG can be particularly useful for tasks that require up-to-date information, domain-specific knowledge, or explicit citation of sources. However, the effectiveness of RAG depends on the quality and relevance of the external knowledge base and the knowledge that is retrieved at runtime. ## RLHF @@ -62,4 +62,4 @@ Time to First Token (TTFT) is a performance metric that measures the time it tak ## Tokens -Tokens are the smallest individual units of a language model, and can correspond to words, subwords, characters, or even bytes (in the case of Unicode). For Claude, a token approximately represents 3.5 English characters, though the exact number can vary depending on the language used. Tokens are typically hidden when interacting with language models at the "text" level but become relevant when examining the exact inputs and outputs of a language model. When Claude is provided with text to evaluate, the text (consisting of a series of characters) is encoded into a series of tokens for the model to process. Larger tokens enable data efficiency during inference and pretraining (and are utilized when possible), while smaller tokens allow a model to handle uncommon or never-before-seen words. The choice of tokenization method can impact the model's performance, vocabulary size, and ability to handle out-of-vocabulary words. \ No newline at end of file +Tokens are the smallest individual units of a language model, and can correspond to words, subwords, characters, or even bytes (in the case of Unicode). For Claude, a token approximately represents 3.5 English characters, though the exact number can vary depending on the language used. Tokens are typically hidden when interacting with language models at the "text" level but become relevant when examining the exact inputs and outputs of a language model. When Claude is provided with text to evaluate, the text (consisting of a series of characters) is encoded into a series of tokens for the model to process. Larger tokens enable data efficiency during inference and pretraining (and are used when possible), while smaller tokens allow a model to handle uncommon or never-before-seen words. The choice of tokenization method can impact the model's performance, vocabulary size, and ability to handle out-of-vocabulary words. diff --git a/content/en/about-claude/model-deprecations.md b/content/en/about-claude/model-deprecations.md index fe5c34d6c..b8c6f006e 100644 --- a/content/en/about-claude/model-deprecations.md +++ b/content/en/about-claude/model-deprecations.md @@ -1,5 +1,7 @@ # Model deprecations +See which Claude models are active, deprecated, or retired, and find retirement dates and recommended replacements for models and API parameters. + --- As safer and more capable models launch, Anthropic regularly retires older ones. Applications relying on Anthropic models may need occasional updates to keep working. Impacted customers will always be notified by email and in the documentation. @@ -9,15 +11,18 @@ This page lists all API deprecations, along with recommended replacements. ## Overview Anthropic uses the following terms to describe the model lifecycle: -- **Active**: The model is fully supported and recommended for use. -- **Legacy**: The model will no longer receive updates and may be deprecated in the future. -- **Deprecated**: The model is still functional but no longer recommended. Anthropic provides a recommended replacement and assigns a retirement date. -- **Retired**: The model is no longer available for use. Requests to retired models will fail. + +* **Active:** The model is fully supported and recommended for use. +* **Legacy:** The model will no longer receive updates and may be deprecated in the future. +* **Deprecated:** The model is still functional but no longer recommended. Anthropic provides a recommended replacement and assigns a retirement date. +* **Retired:** The model is no longer available for use. Requests to retired models will fail. -Deprecated models are likely to be less reliable than active models. Move workloads to active models to maintain the highest level of support and reliability. + Deprecated models are likely to be less reliable than active models. Move workloads to active models to maintain the highest level of support and reliability. +The dates on this page apply to Anthropic-operated platforms: the Claude API, [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws), and [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry). Partner-operated platforms (Amazon Bedrock and Google Cloud) set their own retirement schedules, so a model's lifecycle status and dates can differ. See the [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock#supported-models), [Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy#api-model-ids), and [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai#api-model-ids) model tables. + ## Migrating to replacements Once a model is deprecated, migrate all usage to a suitable replacement before the retirement date. Requests to models past the retirement date will fail. @@ -28,15 +33,15 @@ For specific instructions on migrating to the latest Claude models, see the [Mig ## Notifications -Anthropic notifies customers with active deployments for models with upcoming retirements, providing at least 60 days notice before model retirement for publicly released models. +Anthropic notifies customers with active deployments for models with upcoming retirements, providing at least 60 days' notice before model retirement for publicly released models. ## Auditing model usage To help identify usage of deprecated models, customers can access an audit of their API usage. Follow these steps: -1. Go to the [Usage](/usage) page in Claude Console -2. Click the "Export" button -3. Review the downloaded CSV to see usage broken down by API key and model +1. Go to the [Usage](/usage) page in Claude Console. +2. Click **Export**. +3. Review the downloaded CSV to see usage broken down by API key and model. This audit will help you locate any instances where your application is still using deprecated models, allowing you to prioritize updates to newer models before the retirement date. @@ -50,133 +55,163 @@ This audit will help you locate any instances where your application is still us ## Deprecation downsides and mitigations Anthropic currently deprecates and retires models to ensure capacity for new model releases. This comes with downsides: -- Users who value specific models must migrate to new versions -- Researchers lose access to models for ongoing and comparative studies -- Model retirement introduces safety- and model welfare-related risks + +* Users who value specific models must migrate to new versions +* Researchers lose access to models for ongoing and comparative studies +* Model retirement introduces safety- and model welfare-related risks At some point, Anthropic hopes to make past models publicly available again. In the meantime, Anthropic has committed to long-term preservation of model weights and other measures to help mitigate these impacts. For more details, see [Commitments on Model Deprecation and Preservation](https://www.anthropic.com/research/deprecation-commitments). ## Model status + + [Claude Mythos Preview](https://anthropic.com/glasswing) (`claude-mythos-preview`) will be retired on July 21, 2026. To migrate to [Claude Mythos 5](https://anthropic.com/glasswing) (`claude-mythos-5`), see the [migration guide](/docs/en/about-claude/models/migration-guide#migrating-from-claude-mythos-preview). + + Current and recently retired models are listed in the following table with their status: -| API model name | Current state | Deprecated | Tentative retirement date | -|:----------------------------|:--------------------|:------------------|:-------------------------| -| `claude-opus-4-7` | Active | N/A | Not sooner than April 16, 2027 | -| `claude-opus-4-6` | Active | N/A | Not sooner than February 5, 2027 | -| `claude-opus-4-5-20251101` | Active | N/A | Not sooner than November 24, 2026 | -| `claude-opus-4-1-20250805` | Active | N/A | Not sooner than August 5, 2026 | -| `claude-opus-4-20250514` | Deprecated | April 14, 2026 | June 15, 2026 | -| `claude-sonnet-4-6` | Active | N/A | Not sooner than February 17, 2027 | -| `claude-sonnet-4-5-20250929`| Active | N/A | Not sooner than September 29, 2026 | -| `claude-sonnet-4-20250514` | Deprecated | April 14, 2026 | June 15, 2026 | -| `claude-3-7-sonnet-20250219`| Retired | October 28, 2025 | February 19, 2026 | -| `claude-haiku-4-5-20251001` | Active | N/A | Not sooner than October 15, 2026 | -| `claude-3-5-haiku-20241022` | Retired | December 19, 2025 | February 19, 2026 | -| `claude-3-haiku-20240307` | Retired | February 19, 2026 | April 20, 2026 | +| API model name | Current state | Deprecated | Tentative retirement date | +| -------------------------- | ------------- | ----------------- | ---------------------------------- | +| claude-fable-5 | Active | N/A | Not sooner than June 9, 2027 | +| claude-opus-4-8 | Active | N/A | Not sooner than May 28, 2027 | +| claude-opus-4-7 | Active | N/A | Not sooner than April 16, 2027 | +| claude-opus-4-6 | Active | N/A | Not sooner than February 5, 2027 | +| claude-opus-4-5-20251101 | Active | N/A | Not sooner than November 24, 2026 | +| claude-opus-4-1-20250805 | Deprecated | June 5, 2026 | August 5, 2026 | +| claude-opus-4-20250514 | Retired | April 14, 2026 | June 15, 2026 | +| claude-sonnet-5 | Active | N/A | Not sooner than June 30, 2027 | +| claude-sonnet-4-6 | Active | N/A | Not sooner than February 17, 2027 | +| claude-sonnet-4-5-20250929 | Active | N/A | Not sooner than September 29, 2026 | +| claude-sonnet-4-20250514 | Retired | April 14, 2026 | June 15, 2026 | +| claude-3-7-sonnet-20250219 | Retired | October 28, 2025 | February 19, 2026 | +| claude-haiku-4-5-20251001 | Active | N/A | Not sooner than October 15, 2026 | +| claude-3-5-haiku-20241022 | Retired | December 19, 2025 | February 19, 2026 | +| claude-3-haiku-20240307 | Retired | February 19, 2026 | April 20, 2026 | ## Deprecation history -All deprecations are listed below, with the most recent announcements at the top. +All deprecations are listed in the following sections, with the most recent announcements first. + +### 2026-06-05: Claude Opus 4.1 model + +On June 5, 2026, Anthropic notified developers using Claude Opus 4.1 of its upcoming retirement on the Claude API. + +| Retirement date | Deprecated model | Recommended replacement | +| --------------- | -------------------------- | ----------------------- | +| August 5, 2026 | `claude-opus-4-1-20250805` | `claude-opus-4-8` | ### 2026-04-14: Claude Sonnet 4 and Claude Opus 4 models + + These models were retired June 15, 2026. + + On April 14, 2026, Anthropic notified developers using Claude Sonnet 4 and Claude Opus 4 models of their upcoming retirement on the Claude API. -| Retirement date | Deprecated model | Recommended replacement | -|:----------------------------|:----------------------------|:--------------------------------| -| June 15, 2026 | `claude-sonnet-4-20250514` | `claude-sonnet-4-6` | -| June 15, 2026 | `claude-opus-4-20250514` | `claude-opus-4-7` | +| Retirement date | Deprecated model | Recommended replacement | +| --------------- | -------------------------- | ----------------------- | +| June 15, 2026 | `claude-sonnet-4-20250514` | `claude-sonnet-4-6` | +| June 15, 2026 | `claude-opus-4-20250514` | `claude-opus-4-8` | ### 2026-02-19: Claude Haiku 3 model -This model was retired April 20, 2026. + This model was retired April 20, 2026. On February 19, 2026, Anthropic notified developers using Claude Haiku 3 model of its upcoming retirement on the Claude API. -| Retirement date | Deprecated model | Recommended replacement | -|:----------------------------|:----------------------------|:--------------------------------| -| April 20, 2026 | `claude-3-haiku-20240307` | `claude-haiku-4-5-20251001` | +| Retirement date | Deprecated model | Recommended replacement | +| --------------- | ------------------------- | --------------------------- | +| April 20, 2026 | `claude-3-haiku-20240307` | `claude-haiku-4-5-20251001` | ### 2025-12-19: Claude Haiku 3.5 model -This model was retired February 19, 2026. + This model was retired February 19, 2026. On December 19, 2025, Anthropic notified developers using Claude Haiku 3.5 model of its upcoming retirement on the Claude API. -| Retirement date | Deprecated model | Recommended replacement | -|:----------------------------|:----------------------------|:--------------------------------| -| February 19, 2026 | `claude-3-5-haiku-20241022` | `claude-haiku-4-5-20251001` | +| Retirement date | Deprecated model | Recommended replacement | +| ----------------- | --------------------------- | --------------------------- | +| February 19, 2026 | `claude-3-5-haiku-20241022` | `claude-haiku-4-5-20251001` | ### 2025-10-28: Claude Sonnet 3.7 model -This model was retired February 19, 2026. + This model was retired February 19, 2026. On October 28, 2025, Anthropic notified developers using Claude Sonnet 3.7 model of its upcoming retirement on the Claude API. -| Retirement date | Deprecated model | Recommended replacement | -|:----------------------------|:----------------------------|:--------------------------------| -| February 19, 2026 | `claude-3-7-sonnet-20250219`| `claude-sonnet-4-6` | +| Retirement date | Deprecated model | Recommended replacement | +| ----------------- | ---------------------------- | ----------------------- | +| February 19, 2026 | `claude-3-7-sonnet-20250219` | `claude-sonnet-4-6` | ### 2025-08-13: Claude Sonnet 3.5 models -These models were retired October 28, 2025. + These models were retired October 28, 2025. On August 13, 2025, Anthropic notified developers using Claude Sonnet 3.5 models of their upcoming retirement. -| Retirement date | Deprecated model | Recommended replacement | -|:----------------------------|:----------------------------|:--------------------------------| -| October 28, 2025 | `claude-3-5-sonnet-20240620`| `claude-sonnet-4-6` | -| October 28, 2025 | `claude-3-5-sonnet-20241022`| `claude-sonnet-4-6` | +| Retirement date | Deprecated model | Recommended replacement | +| ---------------- | ---------------------------- | ----------------------- | +| October 28, 2025 | `claude-3-5-sonnet-20240620` | `claude-sonnet-4-6` | +| October 28, 2025 | `claude-3-5-sonnet-20241022` | `claude-sonnet-4-6` | ### 2025-06-30: Claude Opus 3 model -This model was retired January 5, 2026. + This model was retired January 5, 2026. On June 30, 2025, Anthropic notified developers using Claude Opus 3 model of its upcoming retirement. -| Retirement date | Deprecated model | Recommended replacement | -|:----------------------------|:----------------------------|:--------------------------------| -| January 5, 2026 | `claude-3-opus-20240229` | `claude-opus-4-7` | +| Retirement date | Deprecated model | Recommended replacement | +| --------------- | ------------------------ | ----------------------- | +| January 5, 2026 | `claude-3-opus-20240229` | `claude-opus-4-8` | ### 2025-01-21: Claude 2, Claude 2.1, and Claude Sonnet 3 models -These models were retired July 21, 2025. + These models were retired July 21, 2025. On January 21, 2025, Anthropic notified developers using Claude 2, Claude 2.1, and Claude Sonnet 3 models of their upcoming retirements. -| Retirement date | Deprecated model | Recommended replacement | -|:----------------------------|:----------------------------|:--------------------------------| -| July 21, 2025 | `claude-2.0` | `claude-opus-4-7` | -| July 21, 2025 | `claude-2.1` | `claude-opus-4-7` | -| July 21, 2025 | `claude-3-sonnet-20240229` | `claude-sonnet-4-6` | +| Retirement date | Deprecated model | Recommended replacement | +| --------------- | -------------------------- | ----------------------- | +| July 21, 2025 | `claude-2.0` | `claude-opus-4-8` | +| July 21, 2025 | `claude-2.1` | `claude-opus-4-8` | +| July 21, 2025 | `claude-3-sonnet-20240229` | `claude-sonnet-4-6` | ### 2024-09-04: Claude 1 and Instant models -These models were retired November 6, 2024. + These models were retired November 6, 2024. On September 4, 2024, Anthropic notified developers using Claude 1 and Instant models of their upcoming retirements. -| Retirement date | Deprecated model | Recommended replacement | -|:----------------------------|:--------------------------|:---------------------------| -| November 6, 2024 | `claude-1.0` | `claude-haiku-4-5-20251001`| -| November 6, 2024 | `claude-1.1` | `claude-haiku-4-5-20251001`| -| November 6, 2024 | `claude-1.2` | `claude-haiku-4-5-20251001`| -| November 6, 2024 | `claude-1.3` | `claude-haiku-4-5-20251001`| -| November 6, 2024 | `claude-instant-1.0` | `claude-haiku-4-5-20251001`| -| November 6, 2024 | `claude-instant-1.1` | `claude-haiku-4-5-20251001`| -| November 6, 2024 | `claude-instant-1.2` | `claude-haiku-4-5-20251001`| \ No newline at end of file +| Retirement date | Deprecated model | Recommended replacement | +| ---------------- | -------------------- | --------------------------- | +| November 6, 2024 | `claude-1.0` | `claude-haiku-4-5-20251001` | +| November 6, 2024 | `claude-1.1` | `claude-haiku-4-5-20251001` | +| November 6, 2024 | `claude-1.2` | `claude-haiku-4-5-20251001` | +| November 6, 2024 | `claude-1.3` | `claude-haiku-4-5-20251001` | +| November 6, 2024 | `claude-instant-1.0` | `claude-haiku-4-5-20251001` | +| November 6, 2024 | `claude-instant-1.1` | `claude-haiku-4-5-20251001` | +| November 6, 2024 | `claude-instant-1.2` | `claude-haiku-4-5-20251001` | + +## API parameter deprecations + +Anthropic occasionally deprecates request parameters that no longer apply to current models. Deprecated parameters remain in the SDK request types so existing code continues to type-check, but their behavior changes per model. + +| Parameter | Status | Behavior | Recommended replacement | +| ------------------------------- | -------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------- | +| `temperature`, `top_p`, `top_k` | Deprecated (Claude Opus 4.7 and later) | Returns a 400 error when set to a non-default value on Claude Opus 4.7 and later, including Claude Opus 4.8, and on Claude Sonnet 5. | Omit and use [prompting](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) to guide model behavior. | + +For migration steps, see the [migration guide](/docs/en/about-claude/models/migration-guide). diff --git a/content/en/about-claude/models/choosing-a-model.md b/content/en/about-claude/models/choosing-a-model.md index 7275ba3c4..64ae21e5f 100644 --- a/content/en/about-claude/models/choosing-a-model.md +++ b/content/en/about-claude/models/choosing-a-model.md @@ -7,9 +7,11 @@ Selecting the optimal Claude model for your application involves balancing three ## Establish key criteria When choosing a Claude model, consider first evaluating these factors: -- **Capabilities:** What specific features or capabilities will you need the model to have in order to meet your needs? -- **Speed:** How quickly does the model need to respond in your application? For Claude Opus 4.6, [fast mode](/docs/en/build-with-claude/fast-mode) (beta: research preview) can provide up to 2.5x higher output speed at premium pricing. -- **Cost:** What's your budget for both development and production usage? + +* **Capabilities:** What specific features or capabilities will you need the model to have to meet your needs? +* **Speed:** How quickly does the model need to respond in your application? Claude Opus 4.8 and Claude Opus 4.7 support [fast mode](/docs/en/build-with-claude/fast-mode) (research preview), which delivers up to 2.5x higher output speed at premium pricing. Fast mode on Claude Opus 4.7 is deprecated, with removal on July 24, 2026. +* **Cost:** What's your budget for both development and production usage? +* **Effort:** Recent Opus and Sonnet models support an [effort parameter](/docs/en/build-with-claude/effort) that trades intelligence for latency and cost within a single model. Tuning effort is often a better lever than switching models. On Claude Opus 4.8 and Claude Opus 4.7, the `xhigh` effort level, between `high` and `max`, is the best setting for most coding and agentic use cases. Knowing these answers in advance will make narrowing down and deciding which model to use much easier. @@ -23,64 +25,85 @@ There are two general approaches you can use to start testing which Claude model For many applications, starting with a faster, more cost-effective model like Claude Haiku 4.5 can be the optimal approach: -1. Begin implementation with Claude Haiku 4.5 -2. Test your use case thoroughly -3. Evaluate if performance meets your requirements -4. Upgrade only if necessary for specific capability gaps +1. Begin implementation with Claude Haiku 4.5. +2. Test your use case thoroughly. +3. Evaluate if performance meets your requirements. +4. Upgrade only if necessary for specific capability gaps. This approach allows for quick iteration, lower development costs, and is often sufficient for many common applications. This approach is best for: -- Initial prototyping and development -- Applications with tight latency requirements -- Cost-sensitive implementations -- High-volume, straightforward tasks + +* Initial prototyping and development +* Applications with tight latency requirements +* Cost-sensitive implementations +* High-volume, straightforward tasks ### Option 2: Start with the most capable model For complex tasks where intelligence and advanced capabilities are paramount, you may want to start with the most capable model and then consider optimizing to more efficient models down the line: -1. Implement with Claude Opus 4.7 -2. Optimize your prompts for these models -3. Evaluate if performance meets your requirements -4. Consider increasing efficiency by downgrading intelligence over time with greater workflow optimization +1. Implement with Claude Opus 4.8. +2. Optimize your prompts for these models. +3. Evaluate if performance meets your requirements. +4. Consider increasing efficiency by lowering [effort](/docs/en/build-with-claude/effort) or downgrading models over time with greater workflow optimization. This approach is best for: -- Complex reasoning tasks -- Scientific or mathematical applications -- Tasks requiring nuanced understanding -- Applications where accuracy outweighs cost considerations -- Advanced coding + +* Complex reasoning tasks +* Scientific or mathematical applications +* Tasks requiring nuanced understanding +* Applications where accuracy outweighs cost considerations +* Advanced coding and high-autonomy agentic work + + + The [effort parameter](/docs/en/build-with-claude/effort) defaults to `high` on Claude Opus 4.8 across all surfaces, including Claude Code and the Messages API. Use `xhigh` for coding, high-autonomy work, and the most intelligence-demanding tasks. + + +**Claude Fable 5** (`claude-fable-5`) is Anthropic's most capable widely released model, delivering next-generation intelligence for long-running agents. **Claude Mythos 5** (`claude-mythos-5`) is available through [Project Glasswing](https://anthropic.com/glasswing). Both models support a 1M token context window by default, up to 128k output tokens, and always-on [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). See [Introducing Claude Fable 5 and Claude Mythos 5](/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5) for launch details. + +Claude Fable 5 and Claude Mythos 5 are priced at $10 per million input tokens and $50 per million output tokens. ## Model selection matrix -| When you need... | Consider starting with... | Example use cases | -|------------------|-------------------|-------------------| -| Anthropic's most capable generally available model for complex reasoning and agentic coding, with a step-change jump over Claude Opus 4.6 | Claude Opus 4.7 | Long-horizon agentic coding, large-scale refactoring, complex systems engineering, advanced research, multi-hour autonomous tasks | -| Frontier intelligence at scale, built for coding, agents, and enterprise workflows | Claude Sonnet 4.6 | Code generation, data analysis, content creation, visual understanding, agentic tool use | -| Near-frontier performance with lightning-fast speed and extended thinking at the most economical price point | Claude Haiku 4.5 | Real-time applications, high-volume intelligent processing, cost-sensitive deployments needing strong reasoning, sub-agent tasks | +| When you need... | Consider starting with... | Example use cases | +| ------------------------------------------------------------------------------------------------------------ | ------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Complex agentic coding and enterprise work | Claude Opus 4.8 | Multihour autonomous coding agents, large-scale refactoring, complex systems engineering, advanced research, knowledge work, vision-heavy workflows, computer use | +| Frontier intelligence at scale, built for coding, agents, and enterprise workflows | Claude Sonnet 5 | Code generation, data analysis, content creation, visual understanding, agentic tool use | +| Near-frontier performance with lightning-fast speed and extended thinking at the most economical price point | Claude Haiku 4.5 | Real-time applications, high-volume intelligent processing, cost-sensitive deployments needing strong reasoning, sub-agent tasks | *** ## Decide whether to upgrade or change models To determine if you need to upgrade or change models, you should: -1. [Create benchmark tests](/docs/en/test-and-evaluate/develop-tests) specific to your use case - having a good evaluation set is the most important step in the process -2. Test with your actual prompts and data + +1. [Create benchmark tests](/docs/en/test-and-evaluate/develop-tests) specific to your use case - having a good evaluation set is the most important step in the process. + +2. Test with your actual prompts and data. + 3. Compare performance across models for: - - Accuracy of responses - - Response quality - - Handling of edge cases -4. Weigh performance and cost tradeoffs + + * Accuracy of responses + * Response quality + * Handling of edge cases + +4. Weigh performance and cost tradeoffs. ## Next steps - + See detailed specifications and pricing for the latest Claude models - - Explore the latest improvements in Claude Opus 4.7 + + + Explore the latest improvements in Claude Opus 4.8 + + + The best combination of speed and intelligence + + Get started with your first API call - \ No newline at end of file + diff --git a/content/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5.md b/content/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5.md new file mode 100644 index 000000000..de8fb94dc --- /dev/null +++ b/content/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5.md @@ -0,0 +1,137 @@ +# Introducing Claude Fable 5 and Claude Mythos 5 + +Claude Fable 5 and Claude Mythos 5 capabilities, API changes, and availability. + +--- + + + Access to Claude Fable 5 and Claude Mythos 5 has been restored. See [our statement](https://www.anthropic.com/news/redeploying-fable-5) for more information. + + +Claude Fable 5 is Anthropic's most capable widely released model, built for the most demanding reasoning and long-horizon agentic work. Claude Mythos 5 shares the same capabilities and is available only in limited release through [Project Glasswing](https://anthropic.com/glasswing). + +The headline change for integrations: Claude Fable 5 includes safety classifiers that can decline requests. Claude Mythos 5 does not include these classifiers. If your integration calls Claude Fable 5, plan for three changes: new response handling for refusals, fallback options for retrying on another Claude model, and new billing rules. [Refusals, fallback, and billing on Claude Fable 5](#refusals-fallback-and-billing-on-claude-fable-5) summarizes all three. + +## Models + +| Model | API model ID | Description | +| --------------- | ----------------- | --------------------------------------------------------------------------------------------------------------------------------------------- | +| Claude Fable 5 | `claude-fable-5` | Anthropic's most capable widely released model, for the most demanding reasoning and long-horizon agentic work | +| Claude Mythos 5 | `claude-mythos-5` | Shares Claude Fable 5's capabilities without the safety classifiers. Available through Project Glasswing. Successor to Claude Mythos Preview. | + +Claude Fable 5 and Claude Mythos 5 share the same specs and pricing: + +* **Context window and output:** a [1M token context window](/docs/en/build-with-claude/context-windows) by default, and up to 128k output tokens per request. +* **Pricing:** $10 USD per million input tokens and $50 USD per million output tokens. + +For specs across all current models, see the [models overview](/docs/en/about-claude/models/overview). + +## Refusals, fallback, and billing on Claude Fable 5 + +Claude Fable 5 includes safety classifiers that can decline certain requests. Claude Mythos 5 does not include these classifiers, so this section applies to Claude Fable 5 only. The following sections summarize what refusals mean for your integration; each links to the full guide. + +### Refusals + +When Claude Fable 5 declines a request, the Messages API returns `stop_reason: "refusal"` as a successful HTTP 200 response, not an error. The response also reports which classifier declined the request. See [Refusals and fallback](/docs/en/build-with-claude/refusals-and-fallback) for response shapes and handling guidance. + +### Fallback + +A request that Claude Fable 5 refuses can usually be served by another Claude model. There are three ways to retry: + +* **Server-side:** Pass the `fallbacks` parameter to have the API retry for you (in beta on the Claude API and Claude Platform on AWS). See [Server-side fallback](/docs/en/build-with-claude/refusals-and-fallback#server-side-fallback). +* **Client-side:** Use the [SDK middleware](/docs/en/cli-sdks-libraries/middleware) to retry from the client on any platform. See [Client-side fallback](/docs/en/build-with-claude/refusals-and-fallback#client-side-fallback). +* **Manual:** Build the retry yourself, on any platform and in any language. See [Fallback credit](/docs/en/build-with-claude/fallback-credit). + +### Billing + +You are not billed for a request that is refused before any output is generated. When you retry on another model, [fallback credit](/docs/en/build-with-claude/fallback-credit) refunds the prompt-cache cost of switching, so you avoid paying that cost twice. + +## Availability + +Claude Fable 5 and Claude Mythos 5 both become available on June 9, 2026: + +* **Claude Fable 5** is generally available on the Claude API, [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock), [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws), [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai), and [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry). +* **Claude Mythos 5** is not generally available: it is offered in limited availability to approved customers in [Project Glasswing](https://anthropic.com/glasswing). For access, contact your Anthropic, AWS, or Google Cloud account team. Customers without access to Claude Mythos 5 can use Claude Fable 5, which is generally available and offers the same capabilities. + +Claude Fable 5 and Claude Mythos 5 carry 30-day data retention and are not available under zero data retention: both are designated [Covered Models](https://support.claude.com/en/articles/15425695). See [Model-specific data retention requirements](/docs/en/manage-claude/api-and-data-retention#model-specific-data-retention-requirements). + +## Working with Claude Fable 5 and Claude Mythos 5 + +### Prompting + +Claude Fable 5 responds to the same prompting techniques as other Claude models, with a few differences in how to structure long-context prompts and reasoning instructions. See [Prompting Claude Fable 5](/docs/en/build-with-claude/prompt-engineering/prompting-claude-fable-5). + +## Messages API on Claude Fable 5 and Claude Mythos 5 + + + The behaviors in this section are specific to Claude Fable 5 and Claude Mythos 5. The Messages API is unchanged for Opus, Sonnet, and Haiku models. + + +### Adaptive thinking is always on + +[Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) is the only thinking mode on Claude Fable 5 and Claude Mythos 5. It applies whenever the `thinking` parameter is unset. `thinking: {"type": "disabled"}` is not supported. Use the [effort parameter](/docs/en/build-with-claude/effort) to control thinking depth. + +### Raw thinking content is never returned + +The raw chain of thought is never returned on Claude Fable 5 and Claude Mythos 5. The `thinking.display` setting controls what thinking blocks contain instead: + +* `"summarized"` returns thinking blocks with a readable summary of the reasoning. +* `"omitted"` (the default) returns thinking blocks with an empty `thinking` field. + +Pass thinking blocks back unchanged in multi-turn conversations on the same model. See [thinking output on Claude Fable 5 and Claude Mythos 5](/docs/en/build-with-claude/adaptive-thinking#thinking-output-on-claude-fable-5-and-claude-mythos-5) for cross-model handling. + +## Supported features + +At launch, Claude Fable 5 and Claude Mythos 5 support: + +* [Effort](/docs/en/build-with-claude/effort) +* [Task budgets](/docs/en/build-with-claude/task-budgets) (beta: set the `task-budgets-2026-03-13` header) +* The [memory tool](/docs/en/agents-and-tools/tool-use/memory-tool) +* [Code execution](/docs/en/agents-and-tools/tool-use/code-execution-tool) +* [Programmatic tool calling](/docs/en/agents-and-tools/tool-use/programmatic-tool-calling) +* Tool result clearing through [context editing](/docs/en/build-with-claude/context-editing) (beta: set the `context-management-2025-06-27` header) +* [Compaction](/docs/en/build-with-claude/compaction) +* [Vision](/docs/en/build-with-claude/vision) + +## Migrating from earlier models + +Step-by-step instructions live in the migration guide: + +* From Claude Mythos Preview: see [Migrating from Claude Mythos Preview to Claude Mythos 5](/docs/en/about-claude/models/migration-guide#migrating-from-claude-mythos-preview). +* From Claude Opus 4.8: see [Migrating from Claude Opus 4.8 to Claude Fable 5](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-48). + +## Next steps + + + + Step-by-step upgrade instructions from Claude Opus 4.8 and Claude Mythos Preview. + + + + Specs and comparison for all current Claude models. + + + + The only thinking mode on Claude Fable 5 and Claude Mythos 5. + + + + How Claude Fable 5 declines requests, and how to retry on another model. + + + + Avoid paying the prompt-cache cost twice on a retry. + + + + A worked end-to-end example of refusal handling, fallback, and billing. + + + + Control thinking depth and cost on Claude Fable 5 and Claude Mythos 5. + + + + Fable-specific prompting techniques. + + diff --git a/content/en/about-claude/models/migration-guide.md b/content/en/about-claude/models/migration-guide.md index a2114ad09..9946a2501 100644 --- a/content/en/about-claude/models/migration-guide.md +++ b/content/en/about-claude/models/migration-guide.md @@ -1,1467 +1,2424 @@ # Migration guide -Guide for migrating to Claude Opus 4.7 and Claude 4.6 models from previous Claude versions +Guide for migrating to the latest Claude models from previous Claude versions --- -This guide covers migrating [Messages API](/docs/en/build-with-claude/working-with-messages) code. If you use [Claude Managed Agents](/docs/en/managed-agents/overview), no changes beyond updating model name are required. + This guide covers migrating [Messages API](/docs/en/build-with-claude/working-with-messages) code. If you use [Claude Managed Agents](/docs/en/managed-agents/overview), no changes beyond updating the model name are required. -## Migrating to Claude Opus 4.7 + + **Automate your migration with the Claude API skill.** In Claude Code, run `/claude-api migrate` to invoke the bundled [Claude API skill](/docs/en/agents-and-tools/agent-skills/claude-api-skill#migrating-to-a-newer-claude-model). It works for any target model on this page: -Claude Opus 4.7 is our most capable generally available model to date. It is highly autonomous and performs exceptionally well on long-horizon agentic work, knowledge work, vision tasks, and memory tasks. + ```text wrap + /claude-api migrate this project to claude-opus-4-8 + ``` -Claude Opus 4.7 should have strong out-of-the-box performance on existing Claude Opus 4.6 prompts and evals at the same `$5 / $25` per MTok pricing, but there are a handful of behavioral and API changes worth knowing about as you migrate. It supports the same set of features as Claude Opus 4.6, including: + The skill applies the model ID swap and, as needed, breaking parameter changes, prefill replacement, and effort calibration for your target model across your code base, then produces a checklist of items to verify manually. It asks you to confirm the migration scope (entire working directory, a subdirectory, or a specific file list) before editing any files. The skill also detects Amazon Bedrock, Claude Platform on AWS, Google Cloud, and Microsoft Foundry clients and adjusts model ID formats and feature changes for each platform. + -- [1M token context window](/docs/en/build-with-claude/context-windows) at standard API pricing with no long-context premium -- [128k max output tokens](/docs/en/about-claude/models/overview) -- [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) -- [Prompt caching](/docs/en/build-with-claude/prompt-caching) -- [Batch processing](/docs/en/build-with-claude/batch-processing) -- [Files API](/docs/en/build-with-claude/files) -- [PDF support](/docs/en/build-with-claude/pdf-support) -- [Vision](/docs/en/build-with-claude/vision) -- The full set of server-side and client-side [tools](/docs/en/agents-and-tools/tool-use/overview) ([bash](/docs/en/agents-and-tools/tool-use/bash-tool), [code execution](/docs/en/agents-and-tools/tool-use/code-execution-tool), [computer use](/docs/en/agents-and-tools/tool-use/computer-use-tool), [text editor](/docs/en/agents-and-tools/tool-use/text-editor-tool), [web search](/docs/en/agents-and-tools/tool-use/web-search-tool), [web fetch](/docs/en/agents-and-tools/tool-use/web-fetch-tool), [MCP connector](/docs/en/agents-and-tools/mcp-connector), [memory](/docs/en/agents-and-tools/tool-use/memory-tool)) +## Migrating to Claude Mythos 5 - - **Automate this migration with the Claude API skill.** In Claude Code, run `/claude-api migrate` to invoke the bundled [Claude API skill](/docs/en/agents-and-tools/agent-skills/claude-api-skill#migrating-to-a-newer-claude-model): +[Claude Mythos 5](https://anthropic.com/glasswing) is the access-gated model offered in limited availability to approved customers in Project Glasswing. It shares the same specs and pricing as [Claude Fable 5](/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5): a [1M token context window](/docs/en/build-with-claude/context-windows) by default, and up to 128k output tokens per request. - ```text - /claude-api migrate this project to claude-opus-4-7 - ``` +The baseline settings for `claude-mythos-5`: - The skill applies the model ID swap, breaking parameter changes, prefill replacement, and effort calibration described below across your codebase, then produces a checklist of items to verify manually. It asks you to confirm the migration scope (entire working directory, a subdirectory, or a specific file list) before editing any files. - +* **Thinking:** [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) is always on. The model determines when and how much to think on each request, and no `thinking` configuration is required. Both `thinking: {type: "disabled"}` and manual extended thinking (`thinking: {type: "enabled", budget_tokens: N}`) return a 400 error. +* **Prefill:** Prefilling the assistant message returns a 400 error. Use system prompt instructions instead. +* **Data retention:** Claude Mythos 5 requires 30-day data retention and is not available under zero data retention (ZDR) arrangements; it is designated a Covered Model. See [Model-specific data retention requirements](/docs/en/manage-claude/api-and-data-retention#model-specific-data-retention-requirements). + +### Migrating to Claude Mythos 5 from Claude Mythos Preview + +[Claude Mythos 5](https://anthropic.com/glasswing) is the access-gated successor to [Claude Mythos Preview](https://anthropic.com/glasswing), the invitation-only research preview. For a generally available model with the same capabilities, see [Claude Fable 5](/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5). + +Migration is mostly drop-in. Claude Mythos 5 uses the same [Messages API](/docs/en/build-with-claude/working-with-messages) and the same [tool use](/docs/en/agents-and-tools/tool-use/overview) patterns as Claude Mythos Preview, and token counts are roughly unchanged because both models use the same tokenizer. The key changes to check are the features that are no longer available (listed in the next section) and thinking output. -### Update your model name +For the Claude Mythos Preview retirement timeline, see [Model deprecations](/docs/en/about-claude/model-deprecations). + +#### Update your model name ```python -# Opus migration -model = "claude-opus-4-6" # Before -model = "claude-opus-4-7" # After +model = "claude-mythos-preview" # Before +model = "claude-mythos-5" # After ``` -### Breaking changes +#### Features not available on Claude Mythos 5 -1. **Extended thinking removed:** `thinking: {type: "enabled", budget_tokens: N}` is no longer supported on Claude Opus 4.7 or later models and returns a 400 error. Switch to [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) (`thinking: {type: "adaptive"}`) and use the [effort parameter](/docs/en/build-with-claude/effort) to control thinking depth. Adaptive thinking is **off by default** on Claude Opus 4.7: requests with no `thinking` field run without thinking, matching Opus 4.6 behavior. Set `thinking: {type: "adaptive"}` explicitly to enable it. +1. **Extended thinking and thinking token budgets:** Manual extended thinking (`thinking: {type: "enabled", budget_tokens: N}`) is not supported on `claude-mythos-5` and returns a 400 error. [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) is always on: the model determines when and how much to think on each request, and no `thinking` configuration is required. `thinking: {type: "disabled"}` returns an error. `budget_tokens` has no direct replacement: thinking is adaptive, and the [effort parameter](/docs/en/build-with-claude/effort) is a separate output-level control, not a thinking budget. - Before (Claude Opus 4.6): + Before (Claude Mythos Preview): - ```python - client.messages.create( - model="claude-opus-4-6", - max_tokens=64000, - thinking={"type": "enabled", "budget_tokens": 32000}, - messages=[{"role": "user", "content": "..."}], - ) - ``` + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-mythos-preview", + "max_tokens": 16000, + "thinking": { + "type": "enabled", + "budget_tokens": 10000 + }, + "messages": [ + { + "role": "user", + "content": "..." + } + ] + }' + ``` + + ```bash CLI + ant messages create <<'YAML' + model: claude-mythos-preview + max_tokens: 16000 + thinking: + type: enabled + budget_tokens: 10000 + messages: + - role: user + content: "..." + YAML + ``` + + ```python Python + client.messages.create( + model="claude-mythos-preview", + max_tokens=16000, + thinking={"type": "enabled", "budget_tokens": 10000}, + messages=[{"role": "user", "content": "..."}], + ) + ``` + + ```typescript TypeScript + await client.messages.create({ + model: "claude-mythos-preview", + max_tokens: 16000, + thinking: { type: "enabled", budget_tokens: 10000 }, + messages: [{ role: "user", content: "..." }] + }); + ``` + + ```csharp C# + using Anthropic; + using Anthropic.Models.Messages; + + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = "claude-mythos-preview", + MaxTokens = 16000, + Thinking = new ThinkingConfigEnabled(budgetTokens: 10000), + Messages = [new() { Role = Role.User, Content = "..." }] + }; + + var response = await client.Messages.Create(parameters); + Console.WriteLine(response); + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: "claude-mythos-preview", + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamOfEnabled(10000), + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("...")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(response) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model("claude-mythos-preview") + .maxTokens(16000L) + .enabledThinking(10000L) + .addUserMessage("...") + .build(); + + Message response = client.messages().create(params); + IO.println(response); + ``` + + ```php PHP + $client = new Client(); + + $message = $client->messages->create( + maxTokens: 16000, + messages: [['role' => 'user', 'content' => '...']], + model: 'claude-mythos-preview', + thinking: ['type' => 'enabled', 'budget_tokens' => 10000], + ); + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.create( + model: "claude-mythos-preview", + max_tokens: 16000, + thinking: { + type: "enabled", + budget_tokens: 10000 + }, + messages: [ + { role: "user", content: "..." } + ] + ) + ``` + - After (Claude Opus 4.7): + After (Claude Mythos 5): - ```python - client.messages.create( - model="claude-opus-4-7", - max_tokens=64000, - thinking={"type": "adaptive"}, - output_config={"effort": "high"}, # or "max", "xhigh", "medium", "low" - messages=[{"role": "user", "content": "..."}], - ) - ``` + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-mythos-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "..." + } + ] + }' + ``` + + ```bash CLI + ant messages create <<'YAML' + model: claude-mythos-5 + max_tokens: 16000 + messages: + - role: user + content: "..." + YAML + ``` + + ```python Python + client.messages.create( + model="claude-mythos-5", + max_tokens=16000, + messages=[{"role": "user", "content": "..."}], + ) + ``` + + ```typescript TypeScript + await client.messages.create({ + model: "claude-mythos-5", + max_tokens: 16000, + messages: [{ role: "user", content: "..." }] + }); + ``` + + ```csharp C# + using Anthropic; + using Anthropic.Models.Messages; + + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = "claude-mythos-5", + MaxTokens = 16000, + Messages = [new() { Role = Role.User, Content = "..." }] + }; + + var response = await client.Messages.Create(parameters); + Console.WriteLine(response); + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: "claude-mythos-5", + MaxTokens: 16000, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("...")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(response) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model("claude-mythos-5") + .maxTokens(16000L) + .addUserMessage("...") + .build(); + + Message response = client.messages().create(params); + IO.println(response); + ``` + + ```php PHP + $client = new Client(); + + $message = $client->messages->create( + maxTokens: 16000, + messages: [['role' => 'user', 'content' => '...']], + model: 'claude-mythos-5', + ); + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.create( + model: "claude-mythos-5", + max_tokens: 16000, + messages: [ + { role: "user", content: "..." } + ] + ) + ``` + - Adaptive thinking is steerable through prompting. For guidance on tuning when the model over- or under-thinks, see [Calibrating effort and thinking depth](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#calibrating-effort-and-thinking-depth). +2. **Assistant prefill:** Prefilling the assistant message is not supported on `claude-mythos-5` and returns a 400 error, the same as on Claude Mythos Preview. Use system prompt instructions instead. -2. **Sampling parameters removed:** Setting `temperature`, `top_p`, or `top_k` to any non-default value on Claude Opus 4.7 returns a 400 error. The safest migration path is to omit these parameters entirely from request payloads. Prompting is the recommended way to guide model behavior on Claude Opus 4.7. If you were using `temperature = 0` for determinism, note that it never guaranteed identical outputs on prior models. +3. **Thinking output:** On `claude-mythos-5`, the raw chain of thought is never returned, but thinking blocks still carry readable summarized text when `thinking.display` is set to `summarized`. Pass thinking blocks back unchanged when continuing a conversation on the same model. See [Thinking output on Claude Fable 5 and Claude Mythos 5](/docs/en/build-with-claude/adaptive-thinking#thinking-output-on-claude-fable-5-and-claude-mythos-5). -3. **Thinking content omitted by default:** Thinking blocks still appear in the response stream on Claude Opus 4.7, but their `thinking` field is empty unless you explicitly opt in. This is a silent change from Claude Opus 4.6, where the default was to return summarized thinking text. To restore summarized thinking content on Claude Opus 4.7, set `thinking.display` to `"summarized"`: +#### Token counting and billing - ```python - thinking = { - "type": "adaptive", - "display": "summarized", - } - ``` +`claude-mythos-5` uses the same tokenizer as `claude-mythos-preview` (the tokenizer introduced with Claude Opus 4.7). Token counts are roughly unchanged when migrating from `claude-mythos-preview`. Compared with models before Claude Opus 4.7, the same content can tokenize to roughly 30% more tokens, varying by content and workload shape. - The default is `"omitted"` on Claude Opus 4.7. If your product streams reasoning to users, the new default appears as a long pause before output begins; set `display: "summarized"` to restore visible progress during thinking. See [Extended thinking](/docs/en/build-with-claude/extended-thinking#controlling-thinking-display) for details. +[`/v1/messages/count_tokens`](/docs/en/build-with-claude/token-counting) returns roughly unchanged values for `claude-mythos-5` compared with `claude-mythos-preview`. Re-baseline cost and latency on your own workloads. -4. **Updated token counting:** Claude Opus 4.7 uses a new tokenizer, contributing to its improved performance on a wide range of tasks. The new tokenizer may use roughly 1x to 1.35x as many tokens when processing text compared to previous models (up to ~35% more, varying by content). +#### Migration checklist - [`/v1/messages/count_tokens`](/docs/en/build-with-claude/token-counting) will return a different number of tokens for Claude Opus 4.7 than it did for Claude Opus 4.6. Token efficiency can vary by workload shape. +* Update the model name from `claude-mythos-preview` to `claude-mythos-5`. +* Remove manual extended thinking configuration (`thinking: {type: "enabled", budget_tokens: N}`). Adaptive thinking is always on, and no `thinking` field is required. +* Remove any `thinking: {type: "disabled"}` configuration. Disabling thinking returns an error on `claude-mythos-5`. +* Remove `budget_tokens`. It has no direct replacement: thinking is adaptive, and the `effort` parameter is a separate output-level control, not a thinking budget. +* Verify any code that parses the `thinking` field treats it as display text only and passes thinking blocks back unchanged when continuing on the same model. `thinking.display` defaults to `"omitted"` on `claude-mythos-5`, the same as on Claude Mythos Preview; set `display: "summarized"` to receive readable summaries. See [Thinking output on Claude Fable 5 and Claude Mythos 5](/docs/en/build-with-claude/adaptive-thinking#thinking-output-on-claude-fable-5-and-claude-mythos-5). +* If you replay conversation history on another model, strip `thinking` and `redacted_thinking` blocks from prior assistant turns first. Thinking blocks from `claude-mythos-5` are tied to the model that produced them, and models other than Claude Fable 5 and Claude Mythos 5 silently ignore them. Stripping keeps cross-model requests minimal and uniform. +* Re-baseline token counts and costs on your own workloads. Token counts are roughly unchanged when migrating from `claude-mythos-preview`. - Prompting interventions, `task_budget`, and `effort` can help control costs and ensure appropriate token usage. These controls may trade off model intelligence. We suggest updating your `max_tokens` parameters to give additional headroom, including compaction triggers. Claude Opus 4.7 provides a 1M context window at standard API pricing with no long-context premium. +## Migrating to Claude Fable 5 -5. **Prefill removal (carried over from Opus 4.6):** Prefilling assistant messages returns a 400 error on Claude Opus 4.7. Use [structured outputs](/docs/en/build-with-claude/structured-outputs), system prompt instructions, or `output_config.format` instead. +[Claude Fable 5](/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5) is Anthropic's most capable widely released model, generally available on the Claude API, [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock), [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws), [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai), and [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry). -### Choosing an effort level +Migration is mostly drop-in. Claude Fable 5 uses the same [Messages API](/docs/en/build-with-claude/working-with-messages) and the same [tool use](/docs/en/agents-and-tools/tool-use/overview) patterns as Claude Opus 4.8. It supports the same [1M token context window](/docs/en/build-with-claude/context-windows) by default and the same [128k max output tokens](/docs/en/about-claude/models/overview). Token counts are roughly unchanged because both models use the same tokenizer. -The [effort parameter](/docs/en/build-with-claude/effort) allows you to tune Claude's intelligence vs. token spend, trading off capability for faster speed and lower costs. Start with the new `xhigh` effort level for coding and agentic use cases, and use a minimum of `high` effort for most intelligence-sensitive use cases. Experiment with other effort levels to further tune token usage and intelligence: +The key changes to check are always-on [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking), thinking output, safety classifier refusals, and pricing. [Before you migrate](#before-you-migrate) covers pricing and data retention; [What changed](#what-changed) covers the rest. -- **`max`:** Max effort can deliver performance gains in some use cases, but may show diminishing returns from increased token usage. This setting can also sometimes be prone to overthinking. We recommend testing max effort for intelligence-demanding tasks. -- **`xhigh` (new):** Extra high effort is the best setting for most coding and agentic use cases. -- **`high`:** This setting balances token usage and intelligence. For most intelligence-sensitive use cases, we recommend a minimum of `high` effort. -- **`medium`:** Good for cost-sensitive use cases that need to reduce token usage while trading off intelligence. -- **`low`:** Reserve for short, scoped tasks and latency-sensitive workloads that are not intelligence-sensitive. +### Before you migrate -We expect effort to be more important for this model than for any prior Opus, and recommend experimenting with it actively when you upgrade. +Claude Fable 5 is priced at $10 USD per million input tokens and $50 USD per million output tokens, compared with $5 USD and $25 USD for Claude Opus 4.8. See [Claude pricing](/docs/en/about-claude/pricing) for details. -### Behavior changes +Claude Fable 5 requires 30-day data retention and is not available under zero data retention (ZDR) arrangements; it is designated a Covered Model. On the Claude API, a request from an organization whose data retention configuration does not meet this requirement returns a 400 `invalid_request_error`. Organizations with a ZDR arrangement should contact their Anthropic account team to discuss data retention configuration; Claude Opus 4.8 remains available under ZDR. Alternatively, you can configure data retention per workspace. The 30-day data retention requirement applies on every platform where Claude Fable 5 is offered; see [Model-specific data retention requirements](/docs/en/manage-claude/api-and-data-retention#model-specific-data-retention-requirements) for per-platform details. -Claude Opus 4.7 has several behavioral differences from Claude Opus 4.6 that are not API breaking changes but may require prompt updates or scaffolding removal. + + If your code is on Claude Opus 4.7 or earlier, first apply the relevant [Migrating to Claude Opus 4.8](#migrating-to-claude-opus-4-8) sub-section for your current model. Those sections cover breaking changes (sampling parameters rejected, manual extended thinking rejected, prefill removed, new tokenizer) that this section does not repeat. + -1. **Response length varies by use case:** Claude Opus 4.7 calibrates response length to how complex it judges the task to be, rather than defaulting to a fixed verbosity. This usually means shorter answers on simple lookups and much longer ones on open-ended analysis. +### Migrating to Claude Fable 5 from Claude Opus 4.8 - If your product depends on a certain style or verbosity of output, you may need to tune your prompts. For example, to decrease verbosity, add: "Provide concise, focused responses. Skip non-essential context, and keep examples minimal." If you see specific kinds of over-explaining, add targeted instructions in your prompt to prevent them. +#### Update your model name - Positive examples showing how Claude can communicate with the appropriate level of concision tend to be more effective than negative examples or instructions that tell the model what not to do. +```python +model = "claude-opus-4-8" # Before +model = "claude-fable-5" # After +``` -2. **More literal instruction following:** Claude Opus 4.7 interprets prompts more literally and explicitly than Claude Opus 4.6, particularly at lower effort levels. It will not silently generalize an instruction from one item to another, and it will not infer requests you didn't make. The upside of this literalism is precision and less thrash. It generally performs better for API use cases with carefully tuned prompts, structured extraction, and pipelines where you want predictable behavior. A prompt and harness review may be especially helpful for migration to Claude Opus 4.7. +#### What changed -3. **More direct tone:** As with any new model, prose style on long-form writing may shift. Claude Opus 4.7 is more direct and opinionated, with less validation-forward phrasing and fewer emoji than Claude Opus 4.6's warmer style. If your product relies on a specific voice, re-evaluate style prompts against the new baseline. +The items in this section describe the API and behavior differences worth checking after you swap the model ID. -4. **Built-in progress updates in agentic traces:** Claude Opus 4.7 provides more regular, higher-quality updates to the user throughout long agentic traces. If you've added scaffolding to force interim status messages ("After every 3 tool calls, summarize progress"), try removing it. If you find that the length or contents of Claude Opus 4.7's user-facing updates are not well-calibrated to your use case, explicitly describe what these updates should look like in the prompt and provide examples. +1. **Adaptive thinking is always on:** [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) is the only thinking mode on `claude-fable-5`. The model determines when and how much to think on each request, and no `thinking` configuration is required. `thinking: {type: "disabled"}` returns an error. Use the [effort parameter](/docs/en/build-with-claude/effort) to control thinking depth. -5. **Fewer subagents spawned by default:** Claude Opus 4.7 tends to spawn fewer subagents by default. However, this behavior is steerable through prompting; give Claude Opus 4.7 explicit guidance around when subagents are desirable. + The behavior change to check: on Claude Opus 4.8, requests without a `thinking` field run without thinking; on `claude-fable-5`, those same requests run with adaptive thinking. `max_tokens` remains a hard limit on total output, thinking plus response text, so revisit it for workloads that ran without thinking on Claude Opus 4.8. See [Cost control](/docs/en/build-with-claude/adaptive-thinking#cost-control). -6. **Stricter effort calibration:** Meaningfully changing from Claude Opus 4.6, Claude Opus 4.7 respects [effort levels](/docs/en/build-with-claude/effort) strictly, especially at the low end. At `low` and `medium`, the model scopes its work to what was asked rather than going above and beyond. + Before (Claude Opus 4.8): - This is good for latency and cost, but on moderately complex tasks running at `low` effort there is some risk of under-thinking. If you observe shallow reasoning on complex problems, raise effort to `high` or `xhigh` rather than prompting around it. + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-opus-4-8", + "max_tokens": 16000, + "thinking": { + "type": "adaptive" + }, + "output_config": { + "effort": "high" + }, + "messages": [ + { + "role": "user", + "content": "..." + } + ] + }' + ``` + + ```bash CLI + ant messages create <<'YAML' + model: claude-opus-4-8 + max_tokens: 16000 + thinking: + type: adaptive + output_config: + effort: high + messages: + - role: user + content: "..." + YAML + ``` + + ```python Python + client.messages.create( + model="claude-opus-4-8", + max_tokens=16000, + thinking={"type": "adaptive"}, + output_config={"effort": "high"}, + messages=[{"role": "user", "content": "..."}], + ) + ``` + + ```typescript TypeScript + await client.messages.create({ + model: "claude-opus-4-8", + max_tokens: 16000, + thinking: { type: "adaptive" }, + output_config: { effort: "high" }, + messages: [{ role: "user", content: "..." }] + }); + ``` + + ```csharp C# + using Anthropic; + using Anthropic.Models.Messages; + + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = "claude-opus-4-8", + MaxTokens = 16000, + Thinking = new ThinkingConfigAdaptive(), + OutputConfig = new OutputConfig { Effort = Effort.High }, + Messages = [new() { Role = Role.User, Content = "..." }] + }; + + var response = await client.Messages.Create(parameters); + Console.WriteLine(response); + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: "claude-opus-4-8", + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamUnion{ + OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{}, + }, + OutputConfig: anthropic.OutputConfigParam{ + Effort: anthropic.OutputConfigEffortHigh, + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("...")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(response) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model("claude-opus-4-8") + .maxTokens(16000L) + .thinking(ThinkingConfigAdaptive.builder().build()) + .outputConfig(OutputConfig.builder() + .effort(OutputConfig.Effort.HIGH) + .build()) + .addUserMessage("...") + .build(); + + Message response = client.messages().create(params); + IO.println(response); + ``` + + ```php PHP + $client = new Client(); + + $message = $client->messages->create( + maxTokens: 16000, + messages: [['role' => 'user', 'content' => '...']], + model: 'claude-opus-4-8', + thinking: ['type' => 'adaptive'], + outputConfig: ['effort' => 'high'], + ); + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.create( + model: "claude-opus-4-8", + max_tokens: 16000, + thinking: { + type: "adaptive" + }, + output_config: { + effort: "high" + }, + messages: [ + { role: "user", content: "..." } + ] + ) + ``` + - If you need to keep effort at `low` for latency, add targeted guidance: "This task involves multi-step reasoning. Think carefully through the problem before responding." See [Recommended effort levels for Claude Opus 4.7](/docs/en/build-with-claude/effort#recommended-effort-levels-for-claude-opus-4-7). + After (Claude Fable 5): -7. **Fewer tool calls by default:** Claude Opus 4.7 has a tendency to use tools less often than Claude Opus 4.6 and to use reasoning more. This produces better results in most cases. + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-fable-5", + "max_tokens": 16000, + "output_config": { + "effort": "high" + }, + "messages": [ + { + "role": "user", + "content": "..." + } + ] + }' + ``` + + ```bash CLI + ant messages create <<'YAML' + model: claude-fable-5 + max_tokens: 16000 + output_config: + effort: high + messages: + - role: user + content: "..." + YAML + ``` + + ```python Python + client.messages.create( + model="claude-fable-5", + max_tokens=16000, + output_config={"effort": "high"}, + messages=[{"role": "user", "content": "..."}], + ) + ``` + + ```typescript TypeScript + await client.messages.create({ + model: "claude-fable-5", + max_tokens: 16000, + output_config: { effort: "high" }, + messages: [{ role: "user", content: "..." }] + }); + ``` + + ```csharp C# + using Anthropic; + using Anthropic.Models.Messages; + + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = "claude-fable-5", + MaxTokens = 16000, + OutputConfig = new OutputConfig { Effort = Effort.High }, + Messages = [new() { Role = Role.User, Content = "..." }] + }; + + var response = await client.Messages.Create(parameters); + Console.WriteLine(response); + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: "claude-fable-5", + MaxTokens: 16000, + OutputConfig: anthropic.OutputConfigParam{ + Effort: anthropic.OutputConfigEffortHigh, + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("...")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(response) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model("claude-fable-5") + .maxTokens(16000L) + .outputConfig(OutputConfig.builder() + .effort(OutputConfig.Effort.HIGH) + .build()) + .addUserMessage("...") + .build(); + + Message response = client.messages().create(params); + IO.println(response); + ``` + + ```php PHP + $client = new Client(); + + $message = $client->messages->create( + maxTokens: 16000, + messages: [['role' => 'user', 'content' => '...']], + model: 'claude-fable-5', + outputConfig: ['effort' => 'high'], + ); + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.create( + model: "claude-fable-5", + max_tokens: 16000, + output_config: { + effort: "high" + }, + messages: [ + { role: "user", content: "..." } + ] + ) + ``` + - To increase tool usage, raise the effort setting. `high` or `xhigh` effort settings show substantially more tool usage in agentic search and coding. You can also adjust your prompt to explicitly instruct the model about when and how to properly use its tools. +2. **Extended thinking and thinking budgets (unchanged):** Manual extended thinking (`thinking: {type: "enabled", budget_tokens: N}`) is not supported on `claude-fable-5` and returns a 400 error, the same as on Claude Opus 4.8. `budget_tokens` has no direct replacement: thinking is adaptive, and the [effort parameter](/docs/en/build-with-claude/effort) is a separate output-level control, not a thinking budget. -8. **Real-time cybersecurity safeguards:** Newly added in Claude Opus 4.7, requests that involve prohibited or high-risk topics may lead to refusals. For legitimate security work such as penetration testing, vulnerability research, or red-teaming, apply to the [Cyber Verification Program](https://claude.com/form/cyber-use-case) to request reduced restrictions. See [Safeguards, warnings, and appeals](https://support.claude.com/en/articles/8241253-safeguards-warnings-and-appeals) for background. +3. **Assistant prefill (unchanged):** Prefilling the assistant message is not supported on `claude-fable-5` and returns a 400 error, the same as on Claude Opus 4.8. Use system prompt instructions instead. -9. **High-resolution image support:** Claude Opus 4.7 is the first Claude model with high-resolution image support. Maximum image resolution is 2576 pixels on the long edge, up from 1568 pixels on prior models. This unlocks gains on vision-heavy workloads and is particularly valuable for computer use, screenshot understanding, and document analysis. +4. **Thinking output:** On `claude-fable-5`, the raw chain of thought is never returned, but thinking blocks still carry readable summarized text when `thinking.display` is set to `summarized`. Pass thinking blocks back unchanged when continuing a conversation on the same model. See [Thinking output on Claude Fable 5 and Claude Mythos 5](/docs/en/build-with-claude/adaptive-thinking#thinking-output-on-claude-fable-5-and-claude-mythos-5). - High-resolution support is automatic and requires no beta header or client-side opt-in. Two things to plan for: +5. **Safety classifiers and the `refusal` stop reason:** `claude-fable-5` runs safety classifiers on requests and during response generation. When a classifier declines a request, the Messages API returns `stop_reason: "refusal"` as a successful HTTP 200 response, not an error. The `stop_details.category` field reports which classifier fired, with categories such as `"cyber"`, `"bio"`, and `"reasoning_extraction"`, or `null` when the refusal maps to no named category. See the [refusal category table](/docs/en/build-with-claude/refusals-and-fallback#refusal-response) for the full set. - - Full-resolution images can use up to approximately 3x more image tokens than on prior models (up to 4,784 tokens per image, compared to the previous cap of roughly 1,600 tokens per image). Re-budget `max_tokens` and cost expectations for image-heavy workloads, or downsample before sending if you do not need the additional fidelity. - - Pointing and bounding-box coordinates returned by the model are 1\:1 with actual image pixels on Claude Opus 4.7, so no scale-factor conversion is required. + You are not billed for the input tokens of a request refused before any output is generated. When a classifier fires mid-stream, the input and already-streamed output are billed; discard the partial output. - See [High-resolution image support on Claude Opus 4.7](/docs/en/build-with-claude/vision#high-resolution-image-support-on-claude-opus-4-7) for details. + To re-run refused requests on another model automatically, pass the opt-in `fallbacks` parameter, which is in beta on the Claude API and Claude Platform on AWS. The parameter is not available on the Message Batches API or on Amazon Bedrock, Google Cloud, and Microsoft Foundry; on those three platforms, run the retry client-side or use the SDK refusal-fallback middleware. See [Refusals and fallback](/docs/en/build-with-claude/refusals-and-fallback). -### Recommended changes +6. **Start at `high` effort:** The [effort parameter](/docs/en/build-with-claude/effort) default remains `high`. On Claude Opus 4.8, the recommendation for coding and high-autonomy work is to set `xhigh` explicitly. On `claude-fable-5`, use `high` as the default for most tasks and reserve `xhigh` for the most capability-sensitive workloads. Lower effort settings on `claude-fable-5` still perform well and often exceed `xhigh` performance on prior models. Reduce effort if a task completes but takes longer than necessary. See [Prompting Claude Fable 5](/docs/en/build-with-claude/prompt-engineering/prompting-claude-fable-5#consider-all-effort-levels). -These are not required but will improve your experience: +7. **Lower prompt caching minimum:** The minimum cacheable prompt length on `claude-fable-5` is 512 tokens, lower than the 1,024 tokens on Claude Opus 4.8. Prompts that were too short to cache on Claude Opus 4.8 can now create cache entries, with no code changes required. On Amazon Bedrock, the minimum for `claude-fable-5` is 1,024 tokens. See [Prompt caching](/docs/en/build-with-claude/prompt-caching#cache-limitations) for per-model minimums. -1. **Re-evaluate `max_tokens`:** Because the same text produces a higher token count on Claude Opus 4.7, we suggest updating your `max_tokens` parameters to give additional headroom, including compaction triggers. Prompting interventions, [`task_budget`](/docs/en/build-with-claude/task-budgets), and [`effort`](/docs/en/build-with-claude/effort) can help control costs and ensure appropriate token usage. +#### Migration checklist -2. **Audit token-count expectations:** Any code path that estimates tokens client-side or assumes a fixed token-to-character ratio should be re-tested against Claude Opus 4.7. Use the [Token counting endpoint](/docs/en/build-with-claude/token-counting) to verify. +* If your organization has a zero data retention (ZDR) arrangement, confirm eligibility before migrating. `claude-fable-5` requires 30-day data retention and, on the Claude API, returns a 400 `invalid_request_error` otherwise. See [Model-specific data retention requirements](/docs/en/manage-claude/api-and-data-retention#model-specific-data-retention-requirements). +* Update the model name from `claude-opus-4-8` to `claude-fable-5`. +* Remove any `thinking: {type: "disabled"}` configuration. Disabling thinking returns an error on `claude-fable-5`, and requests without a `thinking` field run with adaptive thinking. +* If you removed manual extended thinking and assistant prefills during earlier migrations, no action is needed: both remain unsupported on `claude-fable-5`. +* Verify any code that parses the `thinking` field treats it as display text only and passes thinking blocks back unchanged when continuing on the same model. `thinking.display` defaults to `"omitted"` on `claude-fable-5`, the same as on Claude Opus 4.8; set `display: "summarized"` to receive readable summaries. See [Thinking output on Claude Fable 5 and Claude Mythos 5](/docs/en/build-with-claude/adaptive-thinking#thinking-output-on-claude-fable-5-and-claude-mythos-5). +* If you replay conversation history on another model, strip `thinking` and `redacted_thinking` blocks from prior assistant turns first. Thinking blocks from `claude-fable-5` are tied to the model that produced them, and models other than Claude Fable 5 and Claude Mythos 5 silently ignore them. Stripping keeps cross-model requests minimal and uniform. The exception is redeeming a [fallback credit](/docs/en/build-with-claude/fallback-credit), which requires the request body echoed under that feature's exact rules. +* Handle `stop_reason: "refusal"` and read the `stop_details.category` field. To re-run refused requests on another model automatically, consider the opt-in `fallbacks` parameter (beta). See [Refusals and fallback](/docs/en/build-with-claude/refusals-and-fallback). +* Re-evaluate your `effort` setting. Start at `high` for most tasks, including workloads that ran at `xhigh` on Claude Opus 4.8. +* Re-baseline cost and latency on your own workloads. Token counts are roughly unchanged when migrating from `claude-opus-4-8`; per-token pricing differs. -3. **Adopt [task budgets](/docs/en/build-with-claude/task-budgets) (beta):** Claude Opus 4.7 introduces task budgets. These budgets let you inform Claude how many tokens it has for a full agentic loop, including thinking, tool calls, tool results, and final output. The model sees a running countdown and uses it to prioritize work and finish the task gracefully as the budget is consumed. To use, set the beta header `task-budgets-2026-03-13` and add the following to your output config: +## Migrating to Claude Opus 4.8 - ```python - output_config = { - "effort": "high", - "task_budget": {"type": "tokens", "total": 128000}, - } - ``` +Claude Opus 4.8 is built for complex agentic coding and enterprise work. These are the baseline settings for `claude-opus-4-8`. The following sub-sections cover the specific changes to make from each earlier Opus model. - You may need to experiment with different task budgets for your use case. If the model is given a task budget that is too restrictive, it may complete the task less thoroughly, referencing its budget as the constraint. +* **Pricing:** see [Claude pricing](/docs/en/about-claude/pricing). +* **Thinking:** [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) (`thinking: {type: "adaptive"}`) is the supported thinking mode and is off by default: requests with no `thinking` field run without thinking. Manual extended thinking (`thinking: {type: "enabled", budget_tokens: N}`) returns a 400 error. +* **Effort:** The [effort parameter](/docs/en/build-with-claude/effort) defaults to `high` across all surfaces. For coding and high-autonomy work, set `xhigh` explicitly. +* **Sampling parameters:** `temperature`, `top_p`, and `top_k` set to a non-default value return a 400 error. Omit them and use prompting to guide the model's behavior. +* **Prefill:** Prefilling the assistant message returns a 400 error. Use [structured outputs](/docs/en/build-with-claude/structured-outputs) or `output_config.format` instead. +* **Context window and output:** The full [1M token context window](/docs/en/build-with-claude/context-windows) is served by default with no beta header and no long-context premium, with [128k max output tokens](/docs/en/about-claude/models/overview). - For open-ended agentic tasks where quality matters more than speed, do not set a task budget. Reserve task budgets for workloads where you need the model to scope its work to a token allowance. The minimum value for a task budget is 20k tokens. +Claude Opus 4.8 also supports [prompt caching](/docs/en/build-with-claude/prompt-caching), [batch processing](/docs/en/build-with-claude/batch-processing), the [Files API](/docs/en/build-with-claude/files), [PDF support](/docs/en/build-with-claude/pdf-support), [vision](/docs/en/build-with-claude/vision), the full set of server-side and client-side [tools](/docs/en/agents-and-tools/tool-use/overview), [mid-conversation system messages](/docs/en/about-claude/models/whats-new-claude-4-8#mid-conversation-system-messages), and [refusal stop details](/docs/en/about-claude/models/whats-new-claude-4-8#refusal-stop-details). - A task budget is not a hard cap; it's a suggestion that the model is aware of. It differs from `max_tokens`: +### Migrating to Claude Opus 4.8 from Claude Opus 4.7 - - **`task_budget`:** an advisory cap across the full agentic loop. The model sees it and uses it to pace itself. - - **`max_tokens`:** a hard per-request ceiling on generated tokens. It is not passed to the model, so the model is not aware of it. +Claude Opus 4.8 builds on Claude Opus 4.7. - Use `task_budget` when you want the model to self-moderate, and `max_tokens` as a hard ceiling to cap usage. +Claude Opus 4.8 should have strong out-of-the-box performance on existing Claude Opus 4.7 prompts and evals. There are no breaking API changes for code already running on Claude Opus 4.7. It supports the same set of features as Claude Opus 4.7, including the [1M token context window](/docs/en/build-with-claude/context-windows), [128k max output tokens](/docs/en/about-claude/models/overview), [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking), [prompt caching](/docs/en/build-with-claude/prompt-caching), [batch processing](/docs/en/build-with-claude/batch-processing), the [Files API](/docs/en/build-with-claude/files), [PDF support](/docs/en/build-with-claude/pdf-support), [vision](/docs/en/build-with-claude/vision), and the full set of server-side and client-side [tools](/docs/en/agents-and-tools/tool-use/overview). It also adds [mid-conversation system messages](/docs/en/about-claude/models/whats-new-claude-4-8#mid-conversation-system-messages) and publicly documents [refusal stop details](/docs/en/about-claude/models/whats-new-claude-4-8#refusal-stop-details). -4. **Set a large `max_tokens` at `max` or `xhigh` effort:** If you are running Claude Opus 4.7 at `max` or `xhigh` effort, set a large max output token budget so the model has room to think and act across its subagents and tool calls. We recommend starting at 64k tokens and tuning from there. + + If your code is on Claude Opus 4.6 or earlier, use [Migrating to Claude Opus 4.8 from Claude Opus 4.6](#migrating-from-claude-opus-46) or [Migrating to Claude Opus 4.8 from Claude Opus 4.5 or earlier](#migrating-from-claude-opus-45) instead. Those sections include breaking changes (sampling parameters rejected, manual extended thinking rejected, new tokenizer) that the upgrade from Claude Opus 4.7 alone does not cover. + -5. **Downsample images if high resolution is unnecessary:** Claude Opus 4.7 supports images up to 2576px / 3.75MP. High-res images use more tokens. If the additional image fidelity is unnecessary, downsample images before sending to Claude to avoid token-usage increases. See [Images and vision](/docs/en/build-with-claude/vision). +#### Update your model name + +```python +# Opus migration +model = "claude-opus-4-7" # Before +model = "claude-opus-4-8" # After +``` + +#### What changed + +These are not breaking changes. Code that runs on Claude Opus 4.7 continues to work unchanged on Claude Opus 4.8. The items below describe behavior differences worth checking after you swap the model ID. + +1. **Sampling parameters (unchanged):** Setting `temperature`, `top_p`, or `top_k` to a non-default value returns a 400 error on Claude Opus 4.8, the same as on Claude Opus 4.7. The SDK request types still define these fields for compatibility with earlier models, so code that sets them type-checks, but the API rejects the request server-side. If you removed these parameters when migrating to Opus 4.7, no further changes are needed. + +2. **Effort default is `high`:** The [effort parameter](/docs/en/build-with-claude/effort) default on Claude Opus 4.8 is `high` across all surfaces, including Claude Code and the Messages API. If you already set effort explicitly, your setting is unchanged. For coding and high-autonomy work, set `xhigh` explicitly. Re-evaluate your effort setting against your latency and cost budget. + +3. **1M context window is the default:** Claude Opus 4.8 serves the full 1M token [context window](/docs/en/build-with-claude/context-windows) by default with no beta header and no long-context premium. If your client passes a context-window beta header for compatibility with older models, you can remove it on Claude Opus 4.8. + +4. **Mid-conversation system messages:** Claude Opus 4.8 accepts `role: "system"` messages immediately after a user turn in the `messages` array (subject to [placement rules](/docs/en/build-with-claude/mid-conversation-system-messages#limitations)). Use the top-level `system` field for instructions that apply from the start. Earlier models, including Claude Opus 4.7, reject `role: "system"` in `messages` with a 400 error. If you maintain code paths that rebuild the full message history to update instructions, you can simplify them and preserve [prompt cache](/docs/en/build-with-claude/prompt-caching) hits on earlier turns. -### Migration checklist +5. **Refusal stop details:** The `stop_details` object on refusal responses (available since Claude Opus 4.7) is now publicly documented. When the model declines a request, it identifies the category of refusal, in addition to the existing `refusal` stop reason. No beta header is required, and there is no opt-out. See [Handling stop reasons](/docs/en/build-with-claude/handling-stop-reasons). -- [ ] Update model name from `claude-opus-4-6` to `claude-opus-4-7` (or update aliases). -- [ ] Remove `temperature`, `top_p`, and `top_k` from request payloads. -- [ ] Replace `thinking: {type: "enabled", budget_tokens: N}` with `thinking: {type: "adaptive"}` plus the [effort parameter](/docs/en/build-with-claude/effort). -- [ ] Remove any assistant-message prefills. -- [ ] If your UI displays thinking content, explicitly opt in to thinking summarization. -- [ ] Re-benchmark end-to-end cost and latency under the updated tokenization. -- [ ] Re-tune `max_tokens` to account for the updated tokenization. -- [ ] Re-test any client-side token-count estimations. -- [ ] If your application sends images, re-budget for [high-resolution image support](/docs/en/build-with-claude/vision#high-resolution-image-support-on-claude-opus-4-7) (up to approximately 3x more image tokens per full-resolution image). Downsample before sending if you do not need the additional fidelity. -- [ ] If you consume pointing or bounding-box coordinates from the model, remove any scale-factor conversion; coordinates are 1\:1 with actual image pixels on Claude Opus 4.7. -- [ ] Review prompts for the behavior changes above (response length, literalism, tone, progress updates, subagents, effort calibration, tool triggering, cyber safeguards, high-resolution image handling). -- [ ] Re-baseline response length with existing length-control prompts removed, then tune explicitly. -- [ ] If using `xhigh` or `max` effort, raise `max_tokens` to at least 64k as a starting point. -- [ ] Consider adopting task budgets (beta) for agentic workflows. -- [ ] If your product does legitimate security work, apply to the [Cyber Verification Program](https://claude.com/form/cyber-use-case) for access to lower restrictions on cyber content. +6. **Lower prompt caching minimum:** The minimum cacheable prompt length on Claude Opus 4.8 is 1,024 tokens, lower than on Claude Opus 4.7. Prompts that were too short to cache on Claude Opus 4.7 can now create cache entries, with no code changes required. See [Prompt caching](/docs/en/build-with-claude/prompt-caching#cache-limitations) for per-model minimums. -## Migrating to Claude Opus 4.7 from Opus 4.5 or earlier +7. **Effort levels recalibrated:** The token allocation behind each effort level changes on Claude Opus 4.8 compared to Claude Opus 4.7: `medium` allows somewhat more thinking, `high` somewhat less, and `xhigh` substantially more. If you tuned an effort level against Claude Opus 4.7 cost or latency, re-baseline at the same level before adjusting it. See [Effort](/docs/en/build-with-claude/effort). -If you are migrating from Claude Opus 4.5, Opus 4.1, or an earlier model directly to Claude Opus 4.7, apply **all of the [Opus 4.7 changes above](#migrating-to-claude-opus-4-7)** plus the cumulative changes in this section that took effect between Opus 4.5 and Opus 4.7. If you are migrating from Opus 4.6, you only need the [Opus 4.7 section above](#migrating-to-claude-opus-4-7). +#### Migration checklist -### Update your model name +* Update model name from `claude-opus-4-7` to `claude-opus-4-8` (or update aliases). +* If you removed sampling parameters during the Opus 4.7 migration, no action is needed. If you re-added them with a 400-retry path, remove that retry path. +* Re-evaluate your `effort` setting. The default is `high` across all surfaces; for coding and high-autonomy work, set `xhigh` explicitly. +* Remove any context-window beta header. The 1M context window is the default on the Claude API, Amazon Bedrock, Google Cloud, and Microsoft Foundry. +* If you rebuild conversation history to update instructions, consider switching to a mid-conversation system message to preserve prompt cache hits. +* Verify your stop-reason handling reads `stop_details` on refusals (available since Claude Opus 4.7; now publicly documented). +* Re-baseline cost and latency at your chosen effort level. + +### Migrating to Claude Opus 4.8 from Claude Opus 4.6 + +Claude Opus 4.8 should have strong out-of-the-box performance on existing Claude Opus 4.6 prompts and evals at the same pricing, but there are a handful of behavioral and API changes worth knowing about as you migrate. These changes took effect in Claude Opus 4.7, and there are no additional breaking API changes between Claude Opus 4.7 and Claude Opus 4.8. It supports the same set of features as Claude Opus 4.6, including: + +* [1M token context window](/docs/en/build-with-claude/context-windows) at standard API pricing with no long-context premium +* [128k max output tokens](/docs/en/about-claude/models/overview) +* [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) +* [Prompt caching](/docs/en/build-with-claude/prompt-caching) +* [Batch processing](/docs/en/build-with-claude/batch-processing) +* [Files API](/docs/en/build-with-claude/files) +* [PDF support](/docs/en/build-with-claude/pdf-support) +* [Vision](/docs/en/build-with-claude/vision) +* The full set of server-side and client-side [tools](/docs/en/agents-and-tools/tool-use/overview) ([bash](/docs/en/agents-and-tools/tool-use/bash-tool), [code execution](/docs/en/agents-and-tools/tool-use/code-execution-tool), [computer use](/docs/en/agents-and-tools/tool-use/computer-use-tool), [text editor](/docs/en/agents-and-tools/tool-use/text-editor-tool), [web search](/docs/en/agents-and-tools/tool-use/web-search-tool), [web fetch](/docs/en/agents-and-tools/tool-use/web-fetch-tool), [MCP connector](/docs/en/agents-and-tools/mcp-connector), [memory](/docs/en/agents-and-tools/tool-use/memory-tool)) + +#### Update your model name ```python # Opus migration -model = "claude-opus-4-5" # Before -model = "claude-opus-4-7" # After +model = "claude-opus-4-6" # Before +model = "claude-opus-4-8" # After ``` -### Breaking changes +#### Breaking changes -1. **Prefill removal** is covered in the [Opus 4.7 breaking changes](#breaking-changes) above. - -2. **Tool parameter quoting:** Claude Opus 4.6 and later models may produce slightly different JSON string escaping in tool call arguments (e.g., different handling of Unicode escapes or forward slash escaping). If you parse tool call `input` as a raw string rather than using a JSON parser, verify your parsing logic. Standard JSON parsers (like `json.loads()` or `JSON.parse()`) handle these differences automatically. +1. **Extended thinking removed:** `thinking: {type: "enabled", budget_tokens: N}` is no longer supported on Claude Opus 4.7 or later models and returns a 400 error. Switch to [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) (`thinking: {type: "adaptive"}`) and use the [effort parameter](/docs/en/build-with-claude/effort) to control thinking depth. Adaptive thinking is **off by default** on Claude Opus 4.7: requests with no `thinking` field run without thinking, matching Opus 4.6 behavior. Set `thinking: {type: "adaptive"}` explicitly to enable it. -### Recommended changes + Before (Claude Opus 4.6): -These changes improve your experience on Opus 4.7. Items marked **(required on Opus 4.7)** were optional recommendations when Opus 4.6 launched but are now mandatory; the rest remain recommended. + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-opus-4-6", + "max_tokens": 16000, + "thinking": { + "type": "enabled", + "budget_tokens": 10000 + }, + "messages": [ + { + "role": "user", + "content": "..." + } + ] + }' + ``` + + ```bash CLI + ant messages create <<'YAML' + model: claude-opus-4-6 + max_tokens: 16000 + thinking: + type: enabled + budget_tokens: 10000 + messages: + - role: user + content: "..." + YAML + ``` + + ```python Python + client.messages.create( + model="claude-opus-4-6", + max_tokens=16000, + thinking={"type": "enabled", "budget_tokens": 10000}, + messages=[{"role": "user", "content": "..."}], + ) + ``` + + ```typescript TypeScript + await client.messages.create({ + model: "claude-opus-4-6", + max_tokens: 16000, + thinking: { type: "enabled", budget_tokens: 10000 }, + messages: [{ role: "user", content: "..." }] + }); + ``` + + ```csharp C# + using Anthropic; + using Anthropic.Models.Messages; + + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = "claude-opus-4-6", + MaxTokens = 16000, + Thinking = new ThinkingConfigEnabled(budgetTokens: 10000), + Messages = [new() { Role = Role.User, Content = "..." }] + }; + + var response = await client.Messages.Create(parameters); + Console.WriteLine(response); + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: "claude-opus-4-6", + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamOfEnabled(10000), + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("...")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(response) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model("claude-opus-4-6") + .maxTokens(16000L) + .enabledThinking(10000L) + .addUserMessage("...") + .build(); + + Message response = client.messages().create(params); + IO.println(response); + ``` + + ```php PHP + $client = new Client(); + + $message = $client->messages->create( + maxTokens: 16000, + messages: [['role' => 'user', 'content' => '...']], + model: 'claude-opus-4-6', + thinking: ['type' => 'enabled', 'budget_tokens' => 10000], + ); + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.create( + model: "claude-opus-4-6", + max_tokens: 16000, + thinking: { + type: "enabled", + budget_tokens: 10000 + }, + messages: [ + { role: "user", content: "..." } + ] + ) + ``` + -1. **Migrate to adaptive thinking (required on Opus 4.7):** `thinking: {type: "enabled", budget_tokens: N}` returns a 400 error on Claude Opus 4.7. Switch to `thinking: {type: "adaptive"}` and use the [effort parameter](/docs/en/build-with-claude/effort) to control thinking depth. See [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). + After (Claude Opus 4.8): - - ```python Before nocheck - response = client.beta.messages.create( - model="claude-opus-4-5", - max_tokens=16000, - thinking={"type": "enabled", "budget_tokens": 32000}, - betas=["interleaved-thinking-2025-05-14"], - messages=[...], - ) - ``` - - ```python After - response = client.messages.create( - model="claude-opus-4-7", - max_tokens=16000, - thinking={"type": "adaptive"}, - output_config={"effort": "high"}, - messages=[{"role": "user", "content": "Your prompt here"}], - ) - ``` - - ```bash CLI - ant messages create <<'YAML' - model: claude-opus-4-7 - max_tokens: 16000 - thinking: - type: adaptive - output_config: - effort: high - messages: - - role: user - content: Your prompt here - YAML - ``` - - ```typescript TypeScript hidelines={1..2} - import Anthropic from "@anthropic-ai/sdk"; - - const client = new Anthropic(); - - const response = await client.messages.create({ - model: "claude-opus-4-7", - max_tokens: 16000, - thinking: { type: "adaptive" }, - output_config: { effort: "high" }, - messages: [{ role: "user", content: "Your prompt here" }] - }); - ``` - - ```csharp C# - using Anthropic; - using Anthropic.Models.Messages; - - AnthropicClient client = new(); - - var parameters = new MessageCreateParams - { - Model = Model.ClaudeOpus4_7, - MaxTokens = 16000, - Thinking = new ThinkingConfigAdaptive(), - OutputConfig = new OutputConfig { Effort = Effort.High }, - Messages = [new() { Role = Role.User, Content = "Your prompt here" }] - }; - - var response = await client.Messages.Create(parameters); - Console.WriteLine(response); - ``` - - ```go Go hidelines={1..11,-1} - package main - - import ( - "context" - "fmt" - "log" - - "github.com/anthropics/anthropic-sdk-go" - ) - - func main() { - client := anthropic.NewClient() - - response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ - Model: anthropic.ModelClaudeOpus4_7, - MaxTokens: 16000, - Thinking: anthropic.ThinkingConfigParamUnion{ - OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{}, - }, - OutputConfig: anthropic.OutputConfigParam{ - Effort: anthropic.OutputConfigEffortHigh, - }, - Messages: []anthropic.MessageParam{ - anthropic.NewUserMessage(anthropic.NewTextBlock("Your prompt here")), - }, - }) - if err != nil { - log.Fatal(err) - } - fmt.Println(response) - } - ``` - - ```java Java hidelines={1..5,8..10,-2..} - import com.anthropic.client.AnthropicClient; - import com.anthropic.client.okhttp.AnthropicOkHttpClient; - import com.anthropic.models.messages.MessageCreateParams; - import com.anthropic.models.messages.Message; - import com.anthropic.models.messages.Model; - import com.anthropic.models.messages.OutputConfig; - import com.anthropic.models.messages.ThinkingConfigAdaptive; - - public class AdaptiveThinkingExample { - public static void main(String[] args) { - AnthropicClient client = AnthropicOkHttpClient.fromEnv(); - - MessageCreateParams params = MessageCreateParams.builder() - .model(Model.CLAUDE_OPUS_4_7) - .maxTokens(16000L) - .thinking(ThinkingConfigAdaptive.builder().build()) - .outputConfig(OutputConfig.builder() - .effort(OutputConfig.Effort.HIGH) - .build()) - .addUserMessage("Your prompt here") - .build(); - - Message response = client.messages().create(params); - System.out.println(response); - } - } - ``` - - ```php PHP hidelines={1..4} - messages->create( - maxTokens: 16000, - messages: [['role' => 'user', 'content' => 'Your prompt here']], - model: 'claude-opus-4-7', - thinking: ['type' => 'adaptive'], - outputConfig: ['effort' => 'high'], - ); - ``` - - ```ruby Ruby hidelines={1..2} - require "anthropic" - - client = Anthropic::Client.new - - response = client.messages.create( - model: "claude-opus-4-7", - max_tokens: 16000, - thinking: { type: "adaptive" }, - output_config: { effort: "high" }, - messages: [{ role: "user", content: "Your prompt here" }] - ) - ``` + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-opus-4-8", + "max_tokens": 16000, + "thinking": { + "type": "adaptive" + }, + "output_config": { + "effort": "high" + }, + "messages": [ + { + "role": "user", + "content": "..." + } + ] + }' + ``` + + ```bash CLI + ant messages create <<'YAML' + model: claude-opus-4-8 + max_tokens: 16000 + thinking: + type: adaptive + output_config: + effort: high + messages: + - role: user + content: "..." + YAML + ``` + + ```python Python + client.messages.create( + model="claude-opus-4-8", + max_tokens=16000, + thinking={"type": "adaptive"}, + output_config={"effort": "high"}, # or "max", "xhigh", "medium", "low" + messages=[{"role": "user", "content": "..."}], + ) + ``` + + ```typescript TypeScript + await client.messages.create({ + model: "claude-opus-4-8", + max_tokens: 16000, + thinking: { type: "adaptive" }, + output_config: { effort: "high" }, // or "max", "xhigh", "medium", "low" + messages: [{ role: "user", content: "..." }] + }); + ``` + + ```csharp C# + using Anthropic; + using Anthropic.Models.Messages; + + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = "claude-opus-4-8", + MaxTokens = 16000, + Thinking = new ThinkingConfigAdaptive(), + OutputConfig = new OutputConfig { Effort = Effort.High }, // or Max, Xhigh, Medium, Low + Messages = [new() { Role = Role.User, Content = "..." }] + }; + + var response = await client.Messages.Create(parameters); + Console.WriteLine(response); + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: "claude-opus-4-8", + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamUnion{ + OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{}, + }, + OutputConfig: anthropic.OutputConfigParam{ + Effort: anthropic.OutputConfigEffortHigh, // or Max, Xhigh, Medium, Low + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("...")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(response) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model("claude-opus-4-8") + .maxTokens(16000L) + .thinking(ThinkingConfigAdaptive.builder().build()) + .outputConfig(OutputConfig.builder() + .effort(OutputConfig.Effort.HIGH) // or MAX, XHIGH, MEDIUM, LOW + .build()) + .addUserMessage("...") + .build(); + + Message response = client.messages().create(params); + IO.println(response); + ``` + + ```php PHP + $client = new Client(); + + $message = $client->messages->create( + maxTokens: 16000, + messages: [['role' => 'user', 'content' => '...']], + model: 'claude-opus-4-8', + thinking: ['type' => 'adaptive'], + outputConfig: ['effort' => 'high'], // or 'max', 'xhigh', 'medium', 'low' + ); + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.create( + model: "claude-opus-4-8", + max_tokens: 16000, + thinking: { + type: "adaptive" + }, + output_config: { + effort: "high" # or "max", "xhigh", "medium", "low" + }, + messages: [ + { role: "user", content: "..." } + ] + ) + ``` - Note that the migration also moves from `client.beta.messages.create` to `client.messages.create`. Adaptive thinking and effort are GA features and do not require the beta SDK namespace or any beta headers. + Adaptive thinking is steerable through prompting. For guidance on tuning when the model over- or under-thinks, see [Calibrating effort and thinking depth](/docs/en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8#calibrating-effort-and-thinking-depth). -2. **Remove effort beta header:** The effort parameter is now GA. Remove `betas=["effort-2025-11-24"]` from your requests. +2. **Sampling parameters removed:** Setting `temperature`, `top_p`, or `top_k` to any non-default value on Claude Opus 4.7 returns a 400 error. The safest migration path is to omit these parameters entirely from request payloads. Prompting is the recommended way to guide model behavior on Claude Opus 4.7. If you were using `temperature = 0` for determinism, note that it never guaranteed identical outputs on prior models. -3. **Remove fine-grained tool streaming beta header:** Fine-grained tool streaming is now GA. Remove `betas=["fine-grained-tool-streaming-2025-05-14"]` from your requests. +3. **Thinking content omitted by default:** Thinking blocks still appear in the response stream on Claude Opus 4.7, but their `thinking` field is empty unless you explicitly opt in. This is a silent change from Claude Opus 4.6, where the default was to return summarized thinking text. To restore summarized thinking content on Claude Opus 4.7, set `thinking.display` to `"summarized"`: -4. **Remove interleaved thinking beta header:** Adaptive thinking automatically enables interleaved thinking on Claude Opus 4.7, Opus 4.6, and Sonnet 4.6. Remove `betas=["interleaved-thinking-2025-05-14"]` from your requests. The header is still functional on Sonnet 4.6 with manual extended thinking, but manual mode is deprecated. + + ```python Python + thinking = { + "type": "adaptive", + "display": "summarized", + } + ``` + + ```typescript TypeScript + const thinking = { + type: "adaptive", + display: "summarized" + }; + ``` + + ```csharp C# + var thinking = new ThinkingConfigAdaptive { Display = Display.Summarized }; + ``` + + ```go Go + thinking := anthropic.ThinkingConfigParamUnion{ + OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{ + Display: anthropic.ThinkingConfigAdaptiveDisplaySummarized, + }, + } + ``` + + ```java Java + ThinkingConfigAdaptive thinking = ThinkingConfigAdaptive.builder() + .display(ThinkingConfigAdaptive.Display.SUMMARIZED) + .build(); + ``` + + ```php PHP + $thinking = ['type' => 'adaptive', 'display' => 'summarized']; + ``` + + ```ruby Ruby + thinking = { + type: "adaptive", + display: "summarized" + } + ``` + -5. **Migrate to output_config.format:** If using structured outputs, update `output_format={...}` to `output_config={"format": {...}}`. The old parameter remains functional but is deprecated and will be removed in a future model release. + The default is `"omitted"` on Claude Opus 4.7. If your product streams reasoning to users, the new default appears as a long pause before output begins; set `display: "summarized"` to restore visible progress during thinking. See [Extended thinking](/docs/en/build-with-claude/extended-thinking#controlling-thinking-display) for details. -### Migrating from Claude 4.1 or earlier +4. **Updated token counting:** Claude Opus 4.7 uses a new tokenizer, contributing to its improved performance on a wide range of tasks. The new tokenizer may use roughly 1x to 1.35x as many tokens when processing text compared to previous models (up to \~35% more, varying by content). -If you're migrating from Opus 4.1, Sonnet 4 (deprecated), or earlier models directly to Claude Opus 4.7, apply the Claude Opus 4.7 changes at the top of this guide and the cumulative changes above plus the additional changes in this section. + [`/v1/messages/count_tokens`](/docs/en/build-with-claude/token-counting) returns a different number of tokens for Claude Opus 4.7 than it did for Claude Opus 4.6. Token efficiency can vary by workload shape. -```python -# From Opus 4.1 -model = "claude-opus-4-1-20250805" # Before -model = "claude-opus-4-7" # After + Prompting interventions, `task_budget`, and `effort` can help control costs and ensure appropriate token usage. These controls may trade off model intelligence. Update your `max_tokens` parameters to give additional headroom, including compaction triggers. Claude Opus 4.7 provides a 1M context window at standard API pricing with no long-context premium. -# From Sonnet 4 -model = "claude-sonnet-4-20250514" # Before -model = "claude-opus-4-7" # After +5. **Prefill removal (carried over from Opus 4.6):** Prefilling assistant messages returns a 400 error on Claude Opus 4.7. Use [structured outputs](/docs/en/build-with-claude/structured-outputs), system prompt instructions, or `output_config.format` instead. -# From Sonnet 3.7 -model = "claude-3-7-sonnet-20250219" # Before -model = "claude-opus-4-7" # After -``` +#### Choosing an effort level -#### Additional breaking changes +The [effort parameter](/docs/en/build-with-claude/effort) allows you to tune Claude's intelligence versus token spend, trading off capability for faster speed and lower costs. Start with the `xhigh` effort level for coding and agentic use cases, and use a minimum of `high` effort for most intelligence-sensitive use cases. Experiment with other effort levels to further tune token usage and intelligence: -1. **Remove sampling parameters** +* **`max`:** Max effort can deliver performance gains in some use cases, but may show diminishing returns from increased token usage. This setting can also sometimes be prone to overthinking. Test max effort for intelligence-demanding tasks. +* **`xhigh`:** Extra high effort is the best setting for most coding and agentic use cases. +* **`high`:** This setting balances token usage and intelligence. For most intelligence-sensitive use cases, use a minimum of `high` effort. +* **`medium`:** Good for cost-sensitive use cases that need to reduce token usage while trading off intelligence. +* **`low`:** Reserve for short, scoped tasks and latency-sensitive workloads that are not intelligence-sensitive. - - This is a breaking change when migrating from Claude 3.x models. - +Effort is more important for this model than for any prior Opus. Experiment with it actively when you upgrade. - Starting with Claude Opus 4.7, setting `temperature`, `top_p`, or `top_k` to any non-default value will return a 400 error. The safest migration path is to omit these parameters entirely from requests, and to use prompting to guide the model's behavior. If you were using `temperature = 0` for determinism, note that it never guaranteed identical outputs. +#### Behavior changes - - ```python Python nocheck - # Before - This will error in Claude 4+ models - response = client.messages.create( - model="claude-3-7-sonnet-20250219", - temperature=0.7, - top_p=0.9, # Non-default sampling params return 400 on Opus 4.7 - # ... - ) +Claude Opus 4.7 has several behavioral differences from Claude Opus 4.6 that are not API breaking changes but may require prompt updates or scaffolding removal. - # After - response = client.messages.create( - model="claude-opus-4-7", - # ... - ) - ``` +1. **Response length varies by use case:** Claude Opus 4.7 calibrates response length to how complex it judges the task to be, rather than defaulting to a fixed verbosity. This usually means shorter answers on simple lookups and much longer ones on open-ended analysis. -2. **Update tool versions** + If your product depends on a certain style or verbosity of output, you may need to tune your prompts. For example, to decrease verbosity, add: "Provide concise, focused responses. Skip non-essential context, and keep examples minimal." If you see specific kinds of over-explaining, add targeted instructions in your prompt to prevent them. - - This is a breaking change when migrating from Claude 3.x models. - + Positive examples showing how Claude can communicate with the appropriate level of concision tend to be more effective than negative examples or instructions that tell the model what not to do. - Update to the latest tool versions. Remove any code using the `undo_edit` command. +2. **More literal instruction following:** Claude Opus 4.7 interprets prompts more literally and explicitly than Claude Opus 4.6, particularly at lower effort levels. It does not silently generalize an instruction from one item to another, and it does not infer requests you didn't make. The upside of this literalism is precision and less thrash. It generally performs better for API use cases with carefully tuned prompts, structured extraction, and pipelines where you want predictable behavior. A prompt and harness review may be especially helpful for migration to Claude Opus 4.8. - ```python - # Before - tools = [{"type": "text_editor_20250124", "name": "str_replace_editor"}] +3. **More direct tone:** As with any new model, prose style on long-form writing may shift. Claude Opus 4.7 is more direct and opinionated, with less validation-forward phrasing and fewer emoji than Claude Opus 4.6's warmer style. If your product relies on a specific voice, re-evaluate style prompts against the new baseline. - # After - tools = [{"type": "text_editor_20250728", "name": "str_replace_based_edit_tool"}] - ``` +4. **Built-in progress updates in agentic traces:** Claude Opus 4.7 provides more regular, higher-quality updates to the user throughout long agentic traces. If you've added scaffolding to force interim status messages ("After every 3 tool calls, summarize progress"), try removing it. If you find that the length or contents of Claude Opus 4.7's user-facing updates are not well-calibrated to your use case, explicitly describe what these updates should look like in the prompt and provide examples. - - **Text editor:** Use `text_editor_20250728` and `str_replace_based_edit_tool`. See [Text editor tool documentation](/docs/en/agents-and-tools/tool-use/text-editor-tool) for details. - - **Code execution:** Upgrade to `code_execution_20250825`. See [Code execution tool documentation](/docs/en/agents-and-tools/tool-use/code-execution-tool#upgrade-to-latest-tool-version) for migration instructions. +5. **Fewer subagents spawned by default:** Claude Opus 4.7 tends to spawn fewer subagents by default. However, this behavior is steerable through prompting; give Claude Opus 4.7 explicit guidance around when subagents are desirable. -3. **Handle the `refusal` stop reason** +6. **Stricter effort calibration:** Meaningfully changing from Claude Opus 4.6, Claude Opus 4.7 respects [effort levels](/docs/en/build-with-claude/effort) strictly, especially at the low end. At `low` and `medium`, the model scopes its work to what was asked rather than doing more than requested. - Update your application to [handle `refusal` stop reasons](/docs/en/test-and-evaluate/strengthen-guardrails/handle-streaming-refusals): + This is good for latency and cost, but on moderately complex tasks running at `low` effort there is some risk of under-thinking. If you observe shallow reasoning on complex problems, raise effort to `high` or `xhigh` rather than prompting around it. - - ```python Python nocheck - response = client.messages.create(...) + If you need to keep effort at `low` for latency, add targeted guidance: "This task involves multistep reasoning. Think carefully through the problem before responding." See [Recommended effort levels for Claude Opus 4.7](/docs/en/build-with-claude/effort#recommended-effort-levels-for-claude-opus-4-7). - if response.stop_reason == "refusal": - # Handle refusal appropriately - pass - ``` +7. **Fewer tool calls by default:** Claude Opus 4.7 has a tendency to use tools less often than Claude Opus 4.6 and to use reasoning more. This produces better results in most cases. -4. **Handle the `model_context_window_exceeded` stop reason** + To increase tool usage, raise the effort setting. `high` or `xhigh` effort settings show substantially more tool usage in agentic search and coding. You can also adjust your prompt to explicitly instruct the model about when and how to properly use its tools. - Claude 4.5+ models return a `model_context_window_exceeded` stop reason when generation stops due to hitting the context window limit, rather than the requested `max_tokens` limit. Update your application to handle this new stop reason: +8. **Real-time cybersecurity safeguards:** Newly added in Claude Opus 4.7, requests that involve prohibited or high-risk topics may lead to refusals. For legitimate security work such as penetration testing, vulnerability research, or red-teaming, apply to the [Cyber Verification Program](https://claude.com/form/cyber-use-case) to request reduced restrictions. See [Safeguards, warnings, and appeals](https://support.claude.com/en/articles/8241253-safeguards-warnings-and-appeals) for background. - - ```python Python nocheck - response = client.messages.create(...) +9. **High-resolution image support:** Claude Opus 4.7 is the first Claude model with high-resolution image support. Maximum image resolution is 2,576 pixels on the long edge, up from 1,568 pixels on prior models. This unlocks gains on vision-heavy workloads and is particularly valuable for computer use, screenshot understanding, and document analysis. - if response.stop_reason == "model_context_window_exceeded": - # Handle context window limit appropriately - pass - ``` + High-resolution support is automatic and requires no beta header or client-side opt-in. Two things to plan for: -5. **Verify tool parameter handling (trailing newlines)** + * Full-resolution images can use up to approximately 3x more image tokens than on prior models (up to 4,784 tokens per image, compared to the previous cap of roughly 1,600 tokens per image). Re-budget `max_tokens` and cost expectations for image-heavy workloads, or downsample before sending if you do not need the additional fidelity. + * Pointing and bounding-box coordinates returned by the model are 1:1 with actual image pixels on Claude Opus 4.7, so no scale-factor conversion is required. - Claude 4.5+ models preserve trailing newlines in tool call string parameters that were previously stripped. If your tools rely on exact string matching against tool call parameters, verify your logic handles trailing newlines correctly. + See [High-resolution image support on Claude Opus 4.7](/docs/en/build-with-claude/vision#high-resolution-image-support-on-claude-opus-4-7) for details. -6. **Update your prompts for behavioral changes** +#### Recommended changes - Claude 4+ models have a more concise, direct communication style and require explicit direction. Review [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) for optimization guidance. +These are not required but will improve your experience: -#### Additional recommended changes - -- **Remove legacy beta headers:** Remove `token-efficient-tools-2025-02-19` and `output-128k-2025-02-19`. All Claude 4+ models have built-in token-efficient tool use and these headers have no effect. - -### Migration checklist (from Opus 4.5 or earlier) - -- [ ] Update model ID to `claude-opus-4-7` -- [ ] Apply all [Opus 4.7 breaking changes](#migrating-to-claude-opus-4-7) (extended thinking removed, sampling parameters removed, thinking display omitted by default, updated tokenization) -- [ ] **BREAKING:** Remove assistant message prefills (returns 400 error); use structured outputs or `output_config.format` instead -- [ ] **BREAKING on Opus 4.7:** Replace `thinking: {type: "enabled", budget_tokens: N}` with `thinking: {type: "adaptive"}` plus the [effort parameter](/docs/en/build-with-claude/effort) (returns 400 on Opus 4.7) -- [ ] Verify tool call JSON parsing uses a standard JSON parser -- [ ] Remove `effort-2025-11-24` beta header (effort is now GA) -- [ ] Remove `fine-grained-tool-streaming-2025-05-14` beta header -- [ ] Remove `interleaved-thinking-2025-05-14` beta header (adaptive thinking enables interleaved thinking automatically) -- [ ] Migrate `output_format` to `output_config.format` (if applicable) -- [ ] If migrating from Claude 4.1 or earlier: remove `temperature`, `top_p`, and `top_k` (non-default values return 400 on Opus 4.7) -- [ ] If migrating from Claude 4.1 or earlier: update tool versions (`text_editor_20250728`, `code_execution_20250825`) -- [ ] If migrating from Claude 4.1 or earlier: handle `refusal` stop reason -- [ ] If migrating from Claude 4.1 or earlier: handle `model_context_window_exceeded` stop reason -- [ ] If migrating from Claude 4.1 or earlier: verify tool string parameter handling for trailing newlines -- [ ] If migrating from Claude 4.1 or earlier: remove legacy beta headers (`token-efficient-tools-2025-02-19`, `output-128k-2025-02-19`) -- [ ] Review and update prompts following [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) -- [ ] Test in development environment before production deployment +1. **Re-evaluate `max_tokens`:** Because the same text produces a higher token count on Claude Opus 4.7 and later models, update your `max_tokens` parameters to give additional headroom, including compaction triggers. Prompting interventions, [`task_budget`](/docs/en/build-with-claude/task-budgets), and [`effort`](/docs/en/build-with-claude/effort) can help control costs and ensure appropriate token usage. ---- +2. **Audit token-count expectations:** Any code path that estimates tokens client-side or assumes a fixed token-to-character ratio should be re-tested against Claude Opus 4.8. Use the [Token counting endpoint](/docs/en/build-with-claude/token-counting) to verify. -## Migrating to Claude Sonnet 4.6 +3. **Adopt [task budgets](/docs/en/build-with-claude/task-budgets) (beta):** Claude Opus 4.7 introduces task budgets. These budgets let you inform Claude how many tokens it has for a full agentic loop, including thinking, tool calls, tool results, and final output. The model sees a running countdown and uses it to prioritize work and finish the task gracefully as the budget is consumed. To use, set the beta header `task-budgets-2026-03-13` and add the following to your output config: -Claude Sonnet 4.6 combines strong intelligence with fast performance, featuring improved agentic search capabilities and free code execution when used with web search or web fetch. It is ideal for everyday coding, analysis, and content tasks. + + ```python Python + output_config = { + "effort": "high", + "task_budget": {"type": "tokens", "total": 128000}, + } + ``` + + ```typescript TypeScript + const output_config = { + effort: "high", + task_budget: { type: "tokens", total: 128000 } + }; + ``` + + ```csharp C# + var outputConfig = new BetaOutputConfig + { + Effort = Effort.High, + TaskBudget = new BetaTokenTaskBudget + { + Total = 128000, + }, + }; + ``` + + ```go Go + outputConfig := anthropic.BetaOutputConfigParam{ + Effort: anthropic.BetaOutputConfigEffortHigh, + TaskBudget: anthropic.BetaTokenTaskBudgetParam{ + Total: 128000, + }, + } + ``` + + ```java Java + BetaOutputConfig outputConfig = BetaOutputConfig.builder() + .effort(BetaOutputConfig.Effort.HIGH) + .taskBudget(BetaTokenTaskBudget.builder() + .total(128000L) + .build()) + .build(); + ``` + + ```php PHP + $outputConfig = [ + 'effort' => 'high', + 'taskBudget' => [ + 'type' => 'tokens', + 'total' => 128000, + ], + ]; + ``` + + ```ruby Ruby + output_config = { + effort: :high, + task_budget: { + type: :tokens, + total: 128_000 + } + } + ``` + -For a complete overview of capabilities, see the [models overview](/docs/en/about-claude/models/overview). + You may need to experiment with different task budgets for your use case. If the model is given a task budget that is too restrictive, it may complete the task less thoroughly, referencing its budget as the constraint. - -Sonnet 4.6 pricing is $3 per million input tokens, $15 per million output tokens. See [Claude pricing](/docs/en/about-claude/pricing) for details. - + For open-ended agentic tasks where quality matters more than speed, do not set a task budget. Reserve task budgets for workloads where you need the model to scope its work to a token allowance. The minimum value for a task budget is 20k tokens. -**Update your model name:** + A task budget is not a hard cap; it's a suggestion that the model is aware of. It differs from `max_tokens`: -```python -# From Sonnet 4.5 -model = "claude-sonnet-4-5" # Before -model = "claude-sonnet-4-6" # After + * **`task_budget`:** an advisory cap across the full agentic loop. The model sees it and uses it to pace itself. + * **`max_tokens`:** a hard per-request ceiling on generated tokens. It is not passed to the model, so the model is not aware of it. -# From Sonnet 4 -model = "claude-sonnet-4-20250514" # Before -model = "claude-sonnet-4-6" # After -``` + Use `task_budget` when you want the model to self-moderate, and `max_tokens` as a hard ceiling to cap usage. -### Breaking changes +4. **Set a large `max_tokens` at `max` or `xhigh` effort:** If you are running Claude Opus 4.7 or a later model at `max` or `xhigh` effort, set a large max output token budget so the model has room to think and act across its subagents and tool calls. Start at 64k tokens and tune from there. -#### When migrating from Sonnet 4.5 +5. **Downsample images if high resolution is unnecessary:** Claude Opus 4.7 supports images up to 2576px / 3.75MP. High-res images use more tokens. If the additional image fidelity is unnecessary, downsample images before sending to Claude to avoid token-usage increases. See [Images and vision](/docs/en/build-with-claude/vision). -1. **Prefilling assistant messages is no longer supported** +#### Migration checklist - - This is a breaking change when migrating from Sonnet 4.5 or earlier. - +* Update model name from `claude-opus-4-6` to `claude-opus-4-8` (or update aliases). +* Remove `temperature`, `top_p`, and `top_k` from request payloads. +* Replace `thinking: {type: "enabled", budget_tokens: N}` with `thinking: {type: "adaptive"}` plus the [effort parameter](/docs/en/build-with-claude/effort). +* Remove any assistant-message prefills. +* If your UI displays thinking content, explicitly opt in to thinking summarization. +* Re-benchmark end-to-end cost and latency under the updated tokenization. +* Re-tune `max_tokens` to account for the updated tokenization. +* Re-test any client-side token-count estimations. +* If your application sends images, re-budget for [high-resolution image support](/docs/en/build-with-claude/vision#high-resolution-image-support-on-claude-opus-4-7) (up to approximately 3x more image tokens per full-resolution image). Downsample before sending if you do not need the additional fidelity. +* If you consume pointing or bounding-box coordinates from the model, remove any scale-factor conversion; coordinates are 1:1 with actual image pixels on Claude Opus 4.7. +* Review prompts for the behavior changes above (response length, literalism, tone, progress updates, subagents, effort calibration, tool triggering, cyber safeguards, high-resolution image handling). +* Re-baseline response length with existing length-control prompts removed, then tune explicitly. +* If using `xhigh` or `max` effort, raise `max_tokens` to at least 64k as a starting point. +* Consider adopting task budgets (beta) for agentic workflows. +* If your product does legitimate security work, apply to the [Cyber Verification Program](https://claude.com/form/cyber-use-case) for access to lower restrictions on cyber content. - Prefilling assistant messages returns a `400` error on Sonnet 4.6. Use [structured outputs](/docs/en/build-with-claude/structured-outputs), system prompt instructions, or `output_config.format` instead. +### Migrating to Claude Opus 4.8 from Claude Opus 4.5 or earlier - **Common prefill use cases and migrations:** +If you are migrating from Claude Opus 4.5, Opus 4.1 (deprecated), or an earlier model directly to Claude Opus 4.8, apply **all of the changes in [Migrating to Claude Opus 4.8 from Claude Opus 4.6](#migrating-from-claude-opus-46)** plus the cumulative changes in this section that took effect between Opus 4.5 and Opus 4.7. If you are migrating from Opus 4.6, you only need the [from Claude Opus 4.6 section](#migrating-from-claude-opus-46). - - **Controlling output formatting** (forcing JSON/YAML output): Use [structured outputs](/docs/en/build-with-claude/structured-outputs) or tools with enum fields for classification tasks. +#### Update your model name - - **Eliminating preambles** (removing "Here is..." phrases): Add direct instructions in the system prompt: "Respond directly without preamble. Do not start with phrases like 'Here is...', 'Based on...', etc." +```python +# Opus migration +model = "claude-opus-4-5" # Before +model = "claude-opus-4-8" # After +``` - - **Avoiding bad refusals:** Claude is much better at appropriate refusals now. Clear prompting in the user message without prefill should be sufficient. +#### Breaking changes - - **Continuations** (resuming interrupted responses): Move the continuation to the user message: "Your previous response was interrupted and ended with `[previous_response]`. Continue from where you left off." +1. **Prefill removal** is covered in the [breaking changes for migrating from Claude Opus 4.6](#breaking-changes). - - **Context hydration / role consistency** (refreshing context in long conversations): Inject what were previously prefilled-assistant reminders into the user turn instead. +2. **Tool parameter quoting:** Claude Opus 4.6 and later models may produce slightly different JSON string escaping in tool call arguments (for example, different handling of Unicode escapes or forward slash escaping). If you parse tool call `input` as a raw string rather than using a JSON parser, verify your parsing logic. Standard JSON parsers (such as `json.loads()` or `JSON.parse()`) handle these differences automatically. -2. **Tool parameter JSON escaping may differ** +#### Recommended changes - - This is a breaking change when migrating from Sonnet 4.5 or earlier. - +These changes improve your experience on Claude Opus 4.7 and later models. Items marked **(required on Opus 4.7)** were optional recommendations when Opus 4.6 launched but are now mandatory; the rest remain recommended. - JSON string escaping in tool parameters may differ from previous models. Standard JSON parsers handle this automatically, but custom string-based parsing may need updates. +1. **Migrate to adaptive thinking (required on Opus 4.7):** `thinking: {type: "enabled", budget_tokens: N}` returns a 400 error on Claude Opus 4.7. Switch to `thinking: {type: "adaptive"}` and use the [effort parameter](/docs/en/build-with-claude/effort) to control thinking depth. See [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). -#### When migrating from Claude 3.x + + ```bash cURL + curl -sS https://api.anthropic.com/v1/messages \ + -H "content-type: application/json" \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -d '{ + "model": "claude-opus-4-8", + "max_tokens": 16000, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "high"}, + "messages": [{"role": "user", "content": "Your prompt here"}] + }' + ``` + + ```python Before + response = client.beta.messages.create( + model="claude-opus-4-5", + max_tokens=16000, + thinking={"type": "enabled", "budget_tokens": 32000}, + betas=["interleaved-thinking-2025-05-14"], + messages=[{"role": "user", "content": "Your prompt here"}], + ) + ``` + + ```python After + response = client.messages.create( + model="claude-opus-4-8", + max_tokens=16000, + thinking={"type": "adaptive"}, + output_config={"effort": "high"}, + messages=[{"role": "user", "content": "Your prompt here"}], + ) + ``` + + ```bash CLI + ant messages create <<'YAML' + model: claude-opus-4-8 + max_tokens: 16000 + thinking: + type: adaptive + output_config: + effort: high + messages: + - role: user + content: Your prompt here + YAML + ``` + + ```typescript TypeScript + const client = new Anthropic(); + + const response = await client.messages.create({ + model: "claude-opus-4-8", + max_tokens: 16000, + thinking: { type: "adaptive" }, + output_config: { effort: "high" }, + messages: [{ role: "user", content: "Your prompt here" }] + }); + ``` + + ```csharp C# + using Anthropic; + using Anthropic.Models.Messages; + + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = Model.ClaudeOpus4_8, + MaxTokens = 16000, + Thinking = new ThinkingConfigAdaptive(), + OutputConfig = new OutputConfig { Effort = Effort.High }, + Messages = [new() { Role = Role.User, Content = "Your prompt here" }] + }; + + var response = await client.Messages.Create(parameters); + Console.WriteLine(response); + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamUnion{ + OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{}, + }, + OutputConfig: anthropic.OutputConfigParam{ + Effort: anthropic.OutputConfigEffortHigh, + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("Your prompt here")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(response) + ``` + + ```java Java + import com.anthropic.models.messages.OutputConfig; + import com.anthropic.models.messages.ThinkingConfigAdaptive; + // ... + public class AdaptiveThinkingExample { + public static void main(String[] args) { + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(16000L) + .thinking(ThinkingConfigAdaptive.builder().build()) + .outputConfig(OutputConfig.builder() + .effort(OutputConfig.Effort.HIGH) + .build()) + .addUserMessage("Your prompt here") + .build(); + + Message response = client.messages().create(params); + System.out.println(response); + } + } + ``` + + ```php PHP + $client = new Client(); + + $response = $client->messages->create( + maxTokens: 16000, + messages: [['role' => 'user', 'content' => 'Your prompt here']], + model: 'claude-opus-4-8', + thinking: ['type' => 'adaptive'], + outputConfig: ['effort' => 'high'], + ); + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + response = client.messages.create( + model: "claude-opus-4-8", + max_tokens: 16000, + thinking: { type: "adaptive" }, + output_config: { effort: "high" }, + messages: [{ role: "user", content: "Your prompt here" }] + ) + ``` + -3. **Update sampling parameters** + Note that the migration also moves from `client.beta.messages.create` to `client.messages.create`. Adaptive thinking and effort are GA features and do not require the beta SDK namespace or any beta headers. - - This is a breaking change when migrating from Claude 3.x models. - +2. **Remove effort beta header:** The effort parameter is now GA. Remove `betas=["effort-2025-11-24"]` from your requests. - Use only `temperature` OR `top_p`, not both. +3. **Remove fine-grained tool streaming beta header:** Fine-grained tool streaming is now GA. Remove `betas=["fine-grained-tool-streaming-2025-05-14"]` from your requests. -4. **Update tool versions** +4. **Remove interleaved thinking beta header:** Adaptive thinking automatically enables interleaved thinking on Claude Opus 4.7, Opus 4.6, and Sonnet 4.6. Remove `betas=["interleaved-thinking-2025-05-14"]` from your requests. The header is still functional on Sonnet 4.6 with manual extended thinking, but manual mode is deprecated. + +5. **Migrate to output\_config.format:** If using structured outputs, update `output_format={...}` to `output_config={"format": {...}}`. The old parameter remains functional but is deprecated and will be removed in a future model release. + +#### Migrating from Claude 4.1 or earlier + +If you're migrating from Opus 4.1 (deprecated) or earlier models directly to Claude Opus 4.8, apply the [Migrating to Claude Opus 4.8 from Claude Opus 4.6](#migrating-from-claude-opus-46) changes and the cumulative changes earlier in this section, plus the additional changes in this sub-section. + +```python +# From Opus 4.1 +model = "claude-opus-4-1-20250805" # Before +model = "claude-opus-4-8" # After + +# From Sonnet 3.7 +model = "claude-3-7-sonnet-20250219" # Before +model = "claude-opus-4-8" # After +``` + +##### Additional breaking changes + +1. **Remove sampling parameters** - This is a breaking change when migrating from Claude 3.x models. + This is a breaking change when migrating from Claude 3.x models. - Update to the latest tool versions (`text_editor_20250728`, `code_execution_20250825`). Remove any code using the `undo_edit` command. + Starting with Claude Opus 4.7, setting `temperature`, `top_p`, or `top_k` to any non-default value returns a 400 error. The safest migration path is to omit these parameters entirely from requests, and to use prompting to guide the model's behavior. If you were using `temperature = 0` for determinism, note that it never guaranteed identical outputs. + + + ```python Python + # Before - This will error in Claude 4+ models + response = client.messages.create( + model="claude-3-7-sonnet-20250219", + temperature=0.7, + top_p=0.9, # Non-default sampling params return 400 on Opus 4.7 + # ... + ) + + # After + response = client.messages.create( + model="claude-opus-4-8", + # ... + ) + ``` + + ```typescript TypeScript + // Before - This will error in Claude 4+ models + await client.messages.create({ + model: "claude-3-7-sonnet-20250219", + temperature: 0.7, + top_p: 0.9 // Non-default sampling params return 400 on Opus 4.7 + // ... + }); + + // After + await client.messages.create({ + model: "claude-opus-4-8" + // ... + }); + ``` + + ```csharp C# + // Before - This will error in Claude 4+ models + await client.Messages.Create(new MessageCreateParams + { + Model = "claude-3-7-sonnet-20250219", + Temperature = 0.7, + TopP = 0.9, // Non-default sampling params return 400 on Opus 4.7 + // ... + }); + + // After + await client.Messages.Create(new MessageCreateParams + { + Model = "claude-opus-4-8", + // ... + }); + ``` + + ```go Go + // Before - This will error in Claude 4+ models + client.Messages.New(ctx, anthropic.MessageNewParams{ + Model: "claude-3-7-sonnet-20250219", + Temperature: anthropic.Float(0.7), + TopP: anthropic.Float(0.9), // Non-default sampling params return 400 on Opus 4.7 + // ... + }) + + // After + client.Messages.New(ctx, anthropic.MessageNewParams{ + Model: "claude-opus-4-8", + // ... + }) + ``` + + ```java Java + // Before - This will error in Claude 4+ models + client.messages().create(MessageCreateParams.builder() + .model("claude-3-7-sonnet-20250219") + .temperature(0.7) + .topP(0.9) // Non-default sampling params return 400 on Opus 4.7 + // ... + .build()); + + // After + client.messages().create(MessageCreateParams.builder() + .model("claude-opus-4-8") + // ... + .build()); + ``` + + ```php PHP + // Before - This will error in Claude 4+ models + $client->messages->create( + model: 'claude-3-7-sonnet-20250219', + temperature: 0.7, + topP: 0.9, // Non-default sampling params return 400 on Opus 4.7 + // ... + ); + + // After + $client->messages->create( + model: 'claude-opus-4-8', + // ... + ); + ``` + + ```ruby Ruby + # Before - This will error in Claude 4+ models + client.messages.create( + model: "claude-3-7-sonnet-20250219", + temperature: 0.7, + top_p: 0.9, # Non-default sampling params return 400 on Opus 4.7 + # ... + ) -5. **Handle the `refusal` stop reason** + # After + client.messages.create( + model: "claude-opus-4-8", + # ... + ) + ``` + - Update your application to [handle `refusal` stop reasons](/docs/en/test-and-evaluate/strengthen-guardrails/handle-streaming-refusals). +2. **Update tool versions** -6. **Update your prompts for behavioral changes** + + This is a breaking change when migrating from Claude 3.x models. + - Claude 4 models have a more concise, direct communication style. Review [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) for optimization guidance. + Update to the latest tool versions. Remove any code using the `undo_edit` command. -### Recommended changes + + ```python Python + # Before + tools = [{"type": "text_editor_20250124", "name": "str_replace_editor"}] + + # After + tools = [{"type": "text_editor_20250728", "name": "str_replace_based_edit_tool"}] + ``` + + ```typescript TypeScript + // Before + const legacyTools = [{ type: "text_editor_20250124", name: "str_replace_editor" }]; + + // After + const tools = [{ type: "text_editor_20250728", name: "str_replace_based_edit_tool" }]; + ``` + + ```csharp C# + var parameters = new MessageCreateParams + { + // Before: {"type": "text_editor_20250124", "name": "str_replace_editor"} + // After: + Tools = [new ToolTextEditor20250728()], + // ... + }; + ``` + + ```go Go + params := anthropic.MessageNewParams{ + // Before: {"type": "text_editor_20250124", "name": "str_replace_editor"} + // After: + Tools: []anthropic.ToolUnionParam{ + {OfTextEditor20250728: &anthropic.ToolTextEditor20250728Param{}}, + }, + // ... + } + ``` + + ```java Java + MessageCreateParams params = MessageCreateParams.builder() + // Before: {"type": "text_editor_20250124", "name": "str_replace_editor"} + // After: + .addTool(ToolTextEditor20250728.builder().build()) + // ... + .build(); + ``` + + ```php PHP + $message = $client->messages->create( + // Before: ['type' => 'text_editor_20250124', 'name' => 'str_replace_editor'] + // After: + tools: [new ToolTextEditor20250728()], + // ... + ); + ``` + + ```ruby Ruby + # Before + legacy_tools = [{type: "text_editor_20250124", name: "str_replace_editor"}] + + # After + tools = [{type: "text_editor_20250728", name: "str_replace_based_edit_tool"}] + ``` + -1. **Remove `fine-grained-tool-streaming-2025-05-14` beta header:** Fine-grained tool streaming is now GA on Sonnet 4.6 and no longer requires a beta header. -2. **Migrate `output_format` to `output_config.format`:** The `output_format` parameter is deprecated. Use `output_config.format` instead. + * **Text editor:** Use `text_editor_20250728` and `str_replace_based_edit_tool`. See [Text editor tool](/docs/en/agents-and-tools/tool-use/text-editor-tool) documentation for details. + * **Code execution:** Upgrade to `code_execution_20260521`. See [Code execution tool](/docs/en/agents-and-tools/tool-use/code-execution-tool#upgrade-to-latest-tool-version) documentation for migration instructions. -### Migrating from Sonnet 4.5 +3. **Handle the `refusal` stop reason** -Consider migrating from Sonnet 4.5 to Sonnet 4.6, which delivers more intelligence at the same price point. + Update your application to [handle `refusal` stop reasons](/docs/en/test-and-evaluate/strengthen-guardrails/handle-streaming-refusals): - -Sonnet 4.6 defaults to an effort level of `high`, in contrast to Sonnet 4.5 which had no effort parameter. Consider adjusting the effort parameter as you migrate from Sonnet 4.5 to Sonnet 4.6. If not explicitly set, you may experience higher latency with the default effort level. - + + ```python Python + response = client.messages.create(...) -#### If you're not using extended thinking - -If you're not using extended thinking on Sonnet 4.5, you can continue without it on Sonnet 4.6. You should explicitly set effort to the level appropriate for your use case. At `low` effort with thinking disabled, you can expect similar or better performance relative to Sonnet 4.5 with no extended thinking. - - -```bash cURL -curl https://api.anthropic.com/v1/messages \ - --header "x-api-key: $ANTHROPIC_API_KEY" \ - --header "anthropic-version: 2023-06-01" \ - --header "content-type: application/json" \ - --data \ -'{ - "model": "claude-sonnet-4-6", - "max_tokens": 8192, - "output_config": { - "effort": "low" - }, - "messages": [ - { - "role": "user", - "content": "Your prompt here" - } - ] -}' -``` + if response.stop_reason == "refusal": + # Handle refusal appropriately + pass + ``` -```bash CLI -ant messages create <<'YAML' -model: claude-sonnet-4-6 -max_tokens: 8192 -output_config: - effort: low -messages: - - role: user - content: Your prompt here -YAML -``` + ```typescript TypeScript + const response = await client.messages.create(/* ... */); -```python Python -response = client.messages.create( - model="claude-sonnet-4-6", - max_tokens=8192, - output_config={"effort": "low"}, - messages=[{"role": "user", "content": "Your prompt here"}], -) -``` + if (response.stop_reason === "refusal") { + // Handle refusal appropriately + } + ``` -```typescript TypeScript -const response = await client.messages.create({ - model: "claude-sonnet-4-6", - max_tokens: 8192, - output_config: { effort: "low" }, - messages: [{ role: "user", content: "Your prompt here" }] -}); -``` + ```csharp C# + var response = await client.Messages.Create(...); -```csharp C# -using Anthropic; -using Anthropic.Models.Messages; - -AnthropicClient client = new(); - -var parameters = new MessageCreateParams -{ - Model = Model.ClaudeSonnet4_6, - MaxTokens = 8192, - OutputConfig = new OutputConfig - { - Effort = Effort.Low - }, - Messages = [new() { Role = Role.User, Content = "Your prompt here" }] -}; -var message = await client.Messages.Create(parameters); -Console.WriteLine(message); -``` + if (response.StopReason?.Value() == StopReason.Refusal) + { + // Handle refusal appropriately + } + ``` -```go Go hidelines={1..11,-1} -package main - -import ( - "context" - "fmt" - "log" - - "github.com/anthropics/anthropic-sdk-go" -) - -func main() { - client := anthropic.NewClient() - - response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ - Model: anthropic.Model("claude-sonnet-4-6"), - MaxTokens: 8192, - OutputConfig: anthropic.OutputConfigParam{ - Effort: anthropic.OutputConfigEffortLow, - }, - Messages: []anthropic.MessageParam{ - anthropic.NewUserMessage(anthropic.NewTextBlock("Your prompt here")), - }, - }) - if err != nil { - log.Fatal(err) - } - fmt.Println(response.Content[0].Text) -} -``` + ```go Go + response, _ := client.Messages.New(ctx, params) // your existing request -```java Java hidelines={1..5,7..9,-2..} -import com.anthropic.client.AnthropicClient; -import com.anthropic.client.okhttp.AnthropicOkHttpClient; -import com.anthropic.models.messages.MessageCreateParams; -import com.anthropic.models.messages.Message; -import com.anthropic.models.messages.Model; -import com.anthropic.models.messages.OutputConfig; - -public class Main { - public static void main(String[] args) { - AnthropicClient client = AnthropicOkHttpClient.fromEnv(); - - MessageCreateParams params = MessageCreateParams.builder() - .model(Model.CLAUDE_SONNET_4_6) - .maxTokens(8192L) - .outputConfig(OutputConfig.builder() - .effort(OutputConfig.Effort.LOW) - .build()) - .addUserMessage("Your prompt here") - .build(); - - Message response = client.messages().create(params); - response.content().stream() - .flatMap(block -> block.text().stream()) - .forEach(textBlock -> System.out.println(textBlock.text())); - } -} -``` + if response.StopReason == anthropic.StopReasonRefusal { + // Handle refusal appropriately + } + ``` -```php PHP hidelines={1..4} -messages->create(...); -$message = $client->messages->create( - maxTokens: 8192, - messages: [['role' => 'user', 'content' => 'Your prompt here']], - model: 'claude-sonnet-4-6', - outputConfig: ['effort' => 'low'], -); -echo $message->content[0]->text; -``` + if ($response->stopReason === 'refusal') { + // Handle refusal appropriately + } + ``` -```ruby Ruby hidelines={1..2} -require "anthropic" - -client = Anthropic::Client.new - -message = client.messages.create( - model: "claude-sonnet-4-6", - max_tokens: 8192, - output_config: { - effort: "low" - }, - messages: [ - { role: "user", content: "Your prompt here" } - ] -) -puts message.content.first.text -``` - - -#### If you're using extended thinking - -If you're using extended thinking with `budget_tokens` on Sonnet 4.5, it is still functional on Sonnet 4.6 but is deprecated. Migrate to [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) with the [effort parameter](/docs/en/build-with-claude/effort). - -##### Migrating to adaptive thinking - -[Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) is the recommended replacement for `budget_tokens` on Sonnet 4.6. It is particularly well suited to the following workload patterns: - -- **Autonomous multi-step agents:** coding agents that turn requirements into working software, data analysis pipelines, and bug finding where the model runs independently across many steps. Adaptive thinking lets the model calibrate its reasoning per step, staying on path over longer trajectories. For these workloads, start at `high` effort. If latency or token usage is a concern, scale down to `medium`. -- **Computer use agents:** Sonnet 4.6 achieved best-in-class accuracy on computer use evaluations using adaptive mode. -- **Bimodal workloads:** a mix of easy and hard tasks where adaptive skips thinking on simple queries and reasons deeply on complex ones. - -When using adaptive thinking, evaluate `medium` and `high` effort on your tasks. The right level depends on your workload's tradeoff between quality, latency, and token usage. - - -```bash cURL -curl https://api.anthropic.com/v1/messages \ - --header "x-api-key: $ANTHROPIC_API_KEY" \ - --header "anthropic-version: 2023-06-01" \ - --header "content-type: application/json" \ - --data \ -'{ - "model": "claude-sonnet-4-6", - "max_tokens": 64000, - "thinking": { - "type": "adaptive" - }, - "output_config": { - "effort": "medium" - }, - "messages": [ - { - "role": "user", - "content": "Your prompt here" - } - ] -}' -``` + ```ruby Ruby + response = client.messages.create(...) -```bash CLI nocheck -ant messages create <<'YAML' -model: claude-sonnet-4-6 -max_tokens: 64000 -thinking: - type: adaptive -output_config: - effort: medium -messages: - - role: user - content: Your prompt here -YAML -``` + if response.stop_reason == :refusal + # Handle refusal appropriately + end + ``` + -```python Python nocheck -response = client.messages.create( - model="claude-sonnet-4-6", - max_tokens=64000, - thinking={"type": "adaptive"}, - output_config={"effort": "medium"}, - messages=[{"role": "user", "content": "Your prompt here"}], -) -``` +4. **Handle the `model_context_window_exceeded` stop reason** -```typescript TypeScript nocheck -const response = await client.messages.create({ - model: "claude-sonnet-4-6", - max_tokens: 64000, - thinking: { type: "adaptive" }, - output_config: { effort: "medium" }, - messages: [{ role: "user", content: "Your prompt here" }] -}); -``` + Claude 4.5+ models return a `model_context_window_exceeded` stop reason when generation stops because of hitting the context window limit, rather than the requested `max_tokens` limit. Update your application to handle this new stop reason: -```csharp C# nocheck -using Anthropic; -using Anthropic.Models.Messages; + + ```python Python + response = client.messages.create(...) -AnthropicClient client = new(); + if response.stop_reason == "model_context_window_exceeded": + # Handle context window limit appropriately + pass + ``` -var parameters = new MessageCreateParams -{ - Model = Model.ClaudeSonnet4_6, - MaxTokens = 64000, - Thinking = new ThinkingConfigAdaptive(), - OutputConfig = new OutputConfig { Effort = Effort.Medium }, - Messages = [new() { Role = Role.User, Content = "Your prompt here" }] -}; + ```typescript TypeScript + const response = await client.messages.create(/* ... */); -var message = await client.Messages.Create(parameters); -Console.WriteLine(message); -``` + if (response.stop_reason === "model_context_window_exceeded") { + // Handle context window limit appropriately + } + ``` -```go Go nocheck hidelines={1..11,-1} -package main - -import ( - "context" - "fmt" - "log" - - "github.com/anthropics/anthropic-sdk-go" -) - -func main() { - client := anthropic.NewClient() - - response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ - Model: "claude-sonnet-4-6", - MaxTokens: 64000, - Thinking: anthropic.ThinkingConfigParamUnion{ - OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{}, - }, - OutputConfig: anthropic.OutputConfigParam{ - Effort: anthropic.OutputConfigEffortMedium, - }, - Messages: []anthropic.MessageParam{ - anthropic.NewUserMessage(anthropic.NewTextBlock("Your prompt here")), - }, - }) - if err != nil { - log.Fatal(err) - } - fmt.Println(response) -} -``` + ```csharp C# + var response = await client.Messages.Create(...); -```java Java nocheck hidelines={1..5,8..10,-2..} -import com.anthropic.client.AnthropicClient; -import com.anthropic.client.okhttp.AnthropicOkHttpClient; -import com.anthropic.models.messages.MessageCreateParams; -import com.anthropic.models.messages.Message; -import com.anthropic.models.messages.Model; -import com.anthropic.models.messages.OutputConfig; -import com.anthropic.models.messages.ThinkingConfigAdaptive; - -public class Main { - public static void main(String[] args) { - AnthropicClient client = AnthropicOkHttpClient.fromEnv(); - - MessageCreateParams params = MessageCreateParams.builder() - .model(Model.CLAUDE_SONNET_4_6) - .maxTokens(64000L) - .thinking(ThinkingConfigAdaptive.builder().build()) - .outputConfig(OutputConfig.builder() - .effort(OutputConfig.Effort.MEDIUM) - .build()) - .addUserMessage("Your prompt here") - .build(); - - Message response = client.messages().create(params); - System.out.println(response); - } -} -``` + if (response.StopReason?.Raw() == "model_context_window_exceeded") + { + // Handle context window limit appropriately + } + ``` -```php PHP hidelines={1..4} nocheck -messages->create( - maxTokens: 64000, - messages: [['role' => 'user', 'content' => 'Your prompt here']], - model: 'claude-sonnet-4-6', - thinking: ['type' => 'adaptive'], - outputConfig: ['effort' => 'medium'], -); + StopReason reason = response.stopReason().orElse(StopReason.END_TURN); + if (reason.equals(StopReason.of("model_context_window_exceeded"))) { + // Handle context window limit appropriately + } + ``` -echo array_find($message->content, fn($block) => $block->type === 'text')->text; -``` + ```php PHP + $response = $client->messages->create(...); -```ruby Ruby nocheck hidelines={1..2} -require "anthropic" - -client = Anthropic::Client.new - -message = client.messages.create( - model: "claude-sonnet-4-6", - max_tokens: 64000, - thinking: { - type: "adaptive" - }, - output_config: { - effort: "medium" - }, - messages: [ - { role: "user", content: "Your prompt here" } - ] -) -puts message.content.find { |block| block.type == :text }.text -``` - + if ($response->stopReason === 'model_context_window_exceeded') { + // Handle context window limit appropriately + } + ``` - -If you see inconsistent behavior or quality regressions with adaptive thinking, try lowering the [effort](/docs/en/build-with-claude/effort) setting or using `max_tokens` as a hard limit first. Extended thinking with `budget_tokens` is still functional on Sonnet 4.6 but is deprecated and no longer recommended. - + ```ruby Ruby + response = client.messages.create(...) -##### Keeping budget_tokens during migration - -If you need to keep `budget_tokens` temporarily while migrating, a budget around 16k tokens provides headroom for harder problems without risk of runaway token usage. This configuration is deprecated and will be removed in a future model release. - -###### Coding and agentic use cases - -For agentic coding, frontend design, tool-heavy workflows, and complex enterprise workflows, start with `medium` effort. If you find latency is too high, consider reducing effort to `low`. If you need higher intelligence, consider increasing effort to `high` or migrating to Opus 4.7. - - -```bash cURL -curl https://api.anthropic.com/v1/messages \ - --header "x-api-key: $ANTHROPIC_API_KEY" \ - --header "anthropic-version: 2023-06-01" \ - --header "anthropic-beta: interleaved-thinking-2025-05-14" \ - --header "content-type: application/json" \ - --data \ -'{ - "model": "claude-sonnet-4-6", - "max_tokens": 16384, - "thinking": { - "type": "enabled", - "budget_tokens": 16384 - }, - "output_config": { - "effort": "medium" - }, - "messages": [ - { - "role": "user", - "content": "Your prompt here" - } - ] -}' -``` + if response.stop_reason == :model_context_window_exceeded + # Handle context window limit appropriately + end + ``` + -```bash CLI -ant beta:messages create --beta interleaved-thinking-2025-05-14 <<'YAML' -model: claude-sonnet-4-6 -max_tokens: 16384 -thinking: - type: enabled - budget_tokens: 16384 -output_config: - effort: medium -messages: - - role: user - content: Your prompt here -YAML -``` +5. **Verify tool parameter handling (trailing newlines)** -```python Python -response = client.beta.messages.create( - model="claude-sonnet-4-6", - max_tokens=16384, - thinking={"type": "enabled", "budget_tokens": 16384}, - output_config={"effort": "medium"}, - betas=["interleaved-thinking-2025-05-14"], - messages=[{"role": "user", "content": "Your prompt here"}], -) -``` + Claude 4.5+ models preserve trailing newlines in tool call string parameters that were previously stripped. If your tools rely on exact string matching against tool call parameters, verify your logic handles trailing newlines correctly. -```typescript TypeScript -const response = await client.beta.messages.create({ - model: "claude-sonnet-4-6", - max_tokens: 16384, - thinking: { type: "enabled", budget_tokens: 16384 }, - output_config: { effort: "medium" }, - betas: ["interleaved-thinking-2025-05-14"], - messages: [{ role: "user", content: "Your prompt here" }] -}); -``` +6. **Update your prompts for behavioral changes** -```csharp C# -using Anthropic; -using Anthropic.Models.Beta; -using Anthropic.Models.Beta.Messages; - -AnthropicClient client = new(); - -var parameters = new MessageCreateParams -{ - Model = "claude-sonnet-4-6", - MaxTokens = 16384, - Thinking = new BetaThinkingConfigEnabled { BudgetTokens = 16384 }, - OutputConfig = new BetaOutputConfig - { - Effort = Effort.Medium - }, - Betas = [AnthropicBeta.InterleavedThinking2025_05_14], - Messages = [new() { Role = Role.User, Content = "Your prompt here" }] -}; - -var message = await client.Beta.Messages.Create(parameters); -Console.WriteLine(message); -``` + Claude 4+ models have a more concise, direct communication style and require explicit direction. Review [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) for optimization guidance. -```go Go hidelines={1..11,-1} -package main - -import ( - "context" - "fmt" - "log" - - "github.com/anthropics/anthropic-sdk-go" -) - -func main() { - client := anthropic.NewClient() - - response, err := client.Beta.Messages.New(context.TODO(), anthropic.BetaMessageNewParams{ - Model: "claude-sonnet-4-6", - MaxTokens: 16384, - Thinking: anthropic.BetaThinkingConfigParamOfEnabled(16384), - OutputConfig: anthropic.BetaOutputConfigParam{ - Effort: anthropic.BetaOutputConfigEffortMedium, - }, - Messages: []anthropic.BetaMessageParam{ - anthropic.NewBetaUserMessage(anthropic.NewBetaTextBlock("Your prompt here")), - }, - Betas: []anthropic.AnthropicBeta{anthropic.AnthropicBetaInterleavedThinking2025_05_14}, - }) - if err != nil { - log.Fatal(err) - } - fmt.Println(response) -} -``` +##### Additional recommended changes -```java Java hidelines={1..6,9..11,-2..} -import com.anthropic.client.AnthropicClient; -import com.anthropic.client.okhttp.AnthropicOkHttpClient; -import com.anthropic.models.beta.messages.MessageCreateParams; -import com.anthropic.models.beta.messages.BetaMessage; -import com.anthropic.models.messages.Model; -import com.anthropic.models.beta.AnthropicBeta; -import com.anthropic.models.beta.messages.BetaThinkingConfigEnabled; -import com.anthropic.models.beta.messages.BetaOutputConfig; - -public class Main { - public static void main(String[] args) { - AnthropicClient client = AnthropicOkHttpClient.fromEnv(); - - MessageCreateParams params = MessageCreateParams.builder() - .model(Model.CLAUDE_SONNET_4_6) - .maxTokens(16384L) - .thinking(BetaThinkingConfigEnabled.builder() - .budgetTokens(16384L) - .build()) - .outputConfig(BetaOutputConfig.builder() - .effort(BetaOutputConfig.Effort.MEDIUM) - .build()) - .addBeta(AnthropicBeta.INTERLEAVED_THINKING_2025_05_14) - .addUserMessage("Your prompt here") - .build(); - - BetaMessage response = client.beta().messages().create(params); - System.out.println(response); - } -} -``` +* **Remove legacy beta headers:** Remove `token-efficient-tools-2025-02-19` and `output-128k-2025-02-19`. All Claude 4+ models have built-in token-efficient tool use and these headers have no effect. -```php PHP hidelines={1..4} -beta->messages->create( - maxTokens: 16384, - messages: [['role' => 'user', 'content' => 'Your prompt here']], - model: 'claude-sonnet-4-6', - thinking: ['type' => 'enabled', 'budget_tokens' => 16384], - outputConfig: ['effort' => 'medium'], - betas: ['interleaved-thinking-2025-05-14'], -); +## Migrating to Claude Sonnet 5 -echo array_find($message->content, fn($block) => $block->type === 'text')->text; -``` +Claude Sonnet 5 offers the best combination of speed and intelligence in the Claude model family. It builds on Claude Sonnet 4.6. -```ruby Ruby hidelines={1..2} -require "anthropic" - -client = Anthropic::Client.new - -message = client.beta.messages.create( - model: "claude-sonnet-4-6", - max_tokens: 16384, - thinking: { - type: "enabled", - budget_tokens: 16384 - }, - output_config: { - effort: "medium" - }, - betas: ["interleaved-thinking-2025-05-14"], - messages: [ - { role: "user", content: "Your prompt here" } - ] -) -puts message.content.find { |block| block.type == :text }.text -``` - - -###### Chat and non-coding use cases - -For chat, content generation, search, classification, and other non-coding tasks, start with `low` effort with extended thinking. If you need more depth, increase effort to `medium`. - - -```bash cURL -curl https://api.anthropic.com/v1/messages \ - --header "x-api-key: $ANTHROPIC_API_KEY" \ - --header "anthropic-version: 2023-06-01" \ - --header "anthropic-beta: interleaved-thinking-2025-05-14" \ - --header "content-type: application/json" \ - --data \ -'{ - "model": "claude-sonnet-4-6", - "max_tokens": 8192, - "thinking": { - "type": "enabled", - "budget_tokens": 16384 - }, - "output_config": { - "effort": "low" - }, - "messages": [ - { - "role": "user", - "content": "Your prompt here" - } - ] -}' -``` +Claude Sonnet 5 is a drop-in upgrade for Claude Sonnet 4.6. Introductory pricing of $2/$10 USD per million input/output tokens is in effect through August 31, 2026, after which the standard pricing of $3/$15 USD per million input/output tokens will take effect; see [Pricing](/docs/en/about-claude/pricing#claude-sonnet-5-introductory-pricing) for details. There are two breaking API changes for code already running on Claude Sonnet 4.6: manual extended thinking (`thinking: {type: "enabled", budget_tokens: N}`) and sampling parameters (`temperature`, `top_p`, `top_k`) set to non-default values are no longer accepted and return a 400 error. Use [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) with the [effort parameter](/docs/en/build-with-claude/effort) instead. Claude Sonnet 5 supports the same set of features as Claude Sonnet 4.6, including the [1M token context window](/docs/en/build-with-claude/context-windows), [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking), [prompt caching](/docs/en/build-with-claude/prompt-caching), [batch processing](/docs/en/build-with-claude/batch-processing), the [Files API](/docs/en/build-with-claude/files), [PDF support](/docs/en/build-with-claude/pdf-support), [vision](/docs/en/build-with-claude/vision), and the full set of server-side and client-side [tools](/docs/en/agents-and-tools/tool-use/overview). [Priority Tier](/docs/en/api/service-tiers#supported-models) is not available on Claude Sonnet 5. Claude Sonnet 5 also uses a new tokenizer. -```bash CLI -ant beta:messages create --beta interleaved-thinking-2025-05-14 <<'YAML' -model: claude-sonnet-4-6 -max_tokens: 8192 -thinking: - type: enabled - budget_tokens: 16384 -output_config: - effort: low -messages: - - role: user - content: Your prompt here -YAML -``` +### Migrating to Claude Sonnet 5 from Claude Sonnet 4.6 -```python Python -response = client.beta.messages.create( - model="claude-sonnet-4-6", - max_tokens=8192, - thinking={"type": "enabled", "budget_tokens": 16384}, - output_config={"effort": "low"}, - betas=["interleaved-thinking-2025-05-14"], - messages=[{"role": "user", "content": "Your prompt here"}], -) -``` + + If your code is on Claude Sonnet 4.5 or earlier, also apply [Migrating to Claude Sonnet 5 from Claude Sonnet 4.5 or earlier](#migrating-from-sonnet-45). Those steps include breaking changes (assistant message prefilling rejected, tool parameter JSON escaping differences) that this section alone does not cover. + -```typescript TypeScript -const response = await client.beta.messages.create({ - model: "claude-sonnet-4-6", - max_tokens: 8192, - thinking: { type: "enabled", budget_tokens: 16384 }, - output_config: { effort: "low" }, - betas: ["interleaved-thinking-2025-05-14"], - messages: [{ role: "user", content: "Your prompt here" }] -}); -``` +#### Update your model name -```csharp C# -using Anthropic; -using Anthropic.Models.Beta; -using Anthropic.Models.Beta.Messages; - -AnthropicClient client = new(); - -var parameters = new MessageCreateParams -{ - Model = "claude-sonnet-4-6", - MaxTokens = 8192, - Thinking = new BetaThinkingConfigEnabled { BudgetTokens = 16384 }, - OutputConfig = new BetaOutputConfig - { - Effort = Effort.Low - }, - Betas = [AnthropicBeta.InterleavedThinking2025_05_14], - Messages = [new() { Role = Role.User, Content = "Your prompt here" }] -}; - -var message = await client.Beta.Messages.Create(parameters); -Console.WriteLine(message); +```python +# Sonnet migration +model = "claude-sonnet-4-6" # Before +model = "claude-sonnet-5" # After ``` -```go Go hidelines={1..11,-1} -package main - -import ( - "context" - "fmt" - "log" - - "github.com/anthropics/anthropic-sdk-go" -) - -func main() { - client := anthropic.NewClient() - - response, err := client.Beta.Messages.New(context.TODO(), anthropic.BetaMessageNewParams{ - Model: "claude-sonnet-4-6", - MaxTokens: 8192, - Thinking: anthropic.BetaThinkingConfigParamOfEnabled(16384), - OutputConfig: anthropic.BetaOutputConfigParam{ - Effort: anthropic.BetaOutputConfigEffortLow, - }, - Messages: []anthropic.BetaMessageParam{ - anthropic.NewBetaUserMessage(anthropic.NewBetaTextBlock("Your prompt here")), - }, - Betas: []anthropic.AnthropicBeta{anthropic.AnthropicBetaInterleavedThinking2025_05_14}, - }) - if err != nil { - log.Fatal(err) - } - fmt.Println(response) -} -``` +#### What changed + +Items 4 and 5 in the following list are breaking changes. `max_tokens` remains a hard limit on total output (thinking plus response text), so revisit it for workloads that ran without thinking on Claude Sonnet 4.6. + +1. **New tokenizer:** Claude Sonnet 5 uses a new tokenizer. The same input text produces approximately 30% more tokens than on Claude Sonnet 4.6. The exact increase depends on the content. Requests, responses, and streaming events keep the same shape, and no code changes are required, but anything you measure or budget in tokens shifts: `usage` fields and [token counting](/docs/en/build-with-claude/token-counting) results for the same text are higher, the 1M token context window holds less text, and a `max_tokens` limit tuned for Claude Sonnet 4.6 may truncate equivalent output. Per-token pricing is unchanged, so the cost of an equivalent request can differ. Re-run token counting against Claude Sonnet 5 rather than reusing counts measured against earlier models. + +2. **128k max output tokens (unchanged):** Claude Sonnet 5 supports up to 128k output tokens, the same as Claude Sonnet 4.6. Existing `max_tokens` values remain valid. Account for the new tokenizer when sizing them. + +3. **Assistant message prefilling (unchanged):** Prefilling the assistant message returns a `400` error on Claude Sonnet 5, the same as on Claude Sonnet 4.6. If you removed prefill when migrating to Claude Sonnet 4.6, no further changes are needed. Use [structured outputs](/docs/en/build-with-claude/structured-outputs), system prompt instructions, or `output_config.format` instead. + +4. **Adaptive thinking on by default:** On Claude Sonnet 4.6, requests without a `thinking` field run without thinking; on Claude Sonnet 5, the same requests run with [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). To turn thinking off, pass `thinking: {type: "disabled"}`. Manual extended thinking (`thinking: {type: "enabled", budget_tokens: N}`) is not supported and returns a 400 error. Use the [effort parameter](/docs/en/build-with-claude/effort) (default `high`) to control thinking depth. + + + + + Adaptive thinking is on by default for Claude Sonnet 5. The `thinking` field is shown explicitly here to set `display: "summarized"`; if you omit `thinking`, Claude Sonnet 5 omits thinking content from the response by default. For per-model defaults, see [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). + + + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-sonnet-5", + "max_tokens": 16000, + "thinking": { + "type": "adaptive", + "display": "summarized" + }, + "output_config": { + "effort": "high" + }, + "messages": [ + { + "role": "user", + "content": "Are there an infinite number of prime numbers such that n mod 4 == 3?" + } + ] + }' + ``` + + ```bash CLI + ant messages create \ + --transform content --format yaml <<'YAML' + model: claude-sonnet-5 + max_tokens: 16000 + thinking: + type: adaptive + display: summarized + output_config: + effort: high + messages: + - role: user + content: Are there an infinite number of prime numbers such that n mod 4 == 3? + YAML + ``` + + ```python Python + client = anthropic.Anthropic() + + response = client.messages.create( + model="claude-sonnet-5", + max_tokens=16000, + thinking={"type": "adaptive", "display": "summarized"}, + output_config={"effort": "high"}, + messages=[ + { + "role": "user", + "content": "Are there an infinite number of prime numbers such that n mod 4 == 3?", + } + ], + ) + + # The response contains summarized thinking blocks and text blocks + for block in response.content: + match block.type: + case "thinking": + print(f"\nThinking summary: {block.thinking}") + case "text": + print(f"\nResponse: {block.text}") + ``` + + ```typescript TypeScript + const client = new Anthropic(); + + const response = await client.messages.create({ + model: "claude-sonnet-5", + max_tokens: 16000, + thinking: { + type: "adaptive", + display: "summarized" + }, + output_config: { + effort: "high" + }, + messages: [ + { + role: "user", + content: "Are there an infinite number of prime numbers such that n mod 4 == 3?" + } + ] + }); + + // The response contains summarized thinking blocks and text blocks + for (const block of response.content) { + if (block.type === "thinking") { + console.log(`\nThinking summary: ${block.thinking}`); + } else if (block.type === "text") { + console.log(`\nResponse: ${block.text}`); + } + } + ``` + + ```csharp C# + AnthropicClient client = new(); + + var response = await client.Messages.Create(new() + { + Model = Model.ClaudeSonnet5, + MaxTokens = 16000, + Thinking = new ThinkingConfigAdaptive { Display = Display.Summarized }, + OutputConfig = new OutputConfig { Effort = Effort.High }, + Messages = + [ + new() + { + Role = Role.User, + Content = "Are there an infinite number of prime numbers such that n mod 4 == 3?", + }, + ], + }); + + // The response contains summarized thinking blocks and text blocks + foreach (var block in response.Content) + { + if (block.TryPickThinking(out var thinking)) + { + Console.WriteLine($"\nThinking summary: {thinking.Thinking}"); + } + else if (block.TryPickText(out var text)) + { + Console.WriteLine($"\nResponse: {text.Text}"); + } + } + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.Background(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeSonnet5, + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamUnion{ + OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{ + Display: anthropic.ThinkingConfigAdaptiveDisplaySummarized, + }, + }, + OutputConfig: anthropic.OutputConfigParam{ + Effort: anthropic.OutputConfigEffortHigh, + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("Are there an infinite number of prime numbers such that n mod 4 == 3?")), + }, + }) + if err != nil { + log.Fatal(err) + } + + // The response contains summarized thinking blocks and text blocks + for _, block := range response.Content { + switch block := block.AsAny().(type) { + case anthropic.ThinkingBlock: + fmt.Printf("\nThinking summary: %s", block.Thinking) + case anthropic.TextBlock: + fmt.Printf("\nResponse: %s", block.Text) + } + } + ``` + + ```java Java + import com.anthropic.client.okhttp.AnthropicOkHttpClient; + import com.anthropic.models.messages.MessageCreateParams; + import com.anthropic.models.messages.Model; + import com.anthropic.models.messages.OutputConfig; + import com.anthropic.models.messages.ThinkingConfigAdaptive; + + void main() { + var client = AnthropicOkHttpClient.fromEnv(); + + var params = MessageCreateParams.builder() + .model(Model.CLAUDE_SONNET_5) + .maxTokens(16_000) + .thinking(ThinkingConfigAdaptive.builder() + .display(ThinkingConfigAdaptive.Display.SUMMARIZED) + .build()) + .outputConfig(OutputConfig.builder() + .effort(OutputConfig.Effort.HIGH) + .build()) + .addUserMessage("Are there an infinite number of prime numbers such that n mod 4 == 3?") + .build(); + + var response = client.messages().create(params); + + // The response contains summarized thinking blocks and text blocks + for (var block : response.content()) { + block.thinking().ifPresent(thinkingBlock -> + IO.println("\nThinking summary: " + thinkingBlock.thinking()) + ); + block.text().ifPresent(textBlock -> + IO.println("\nResponse: " + textBlock.text()) + ); + } + } + ``` + + ```php PHP + $client = new Client(); + + $response = $client->messages->create( + model: 'claude-sonnet-5', + maxTokens: 16000, + thinking: ['type' => 'adaptive', 'display' => 'summarized'], + outputConfig: ['effort' => 'high'], + messages: [ + [ + 'role' => 'user', + 'content' => 'Are there an infinite number of prime numbers such that n mod 4 == 3?', + ], + ], + ); + + // The response contains summarized thinking blocks and text blocks + foreach ($response->content as $block) { + echo match ($block->type) { + 'thinking' => "\nThinking summary: {$block->thinking}", + 'text' => "\nResponse: {$block->text}", + default => '', + }; + } + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + response = client.messages.create( + model: "claude-sonnet-5", + max_tokens: 16_000, + thinking: {type: :adaptive, display: :summarized}, + output_config: {effort: :high}, + messages: [ + { + role: :user, + content: "Are there an infinite number of prime numbers such that n mod 4 == 3?" + } + ] + ) + + # The response contains summarized thinking blocks and text blocks + response.content.each do |block| + case block + in {type: :thinking, thinking:} + puts "\nThinking summary: #{thinking}" + in {type: :text, text:} + puts "\nResponse: #{text}" + else + end + end + ``` + + + + + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-sonnet-4-6", + "max_tokens": 16000, + "thinking": { + "type": "enabled", + "budget_tokens": 10000 + }, + "messages": [ + { + "role": "user", + "content": "Are there an infinite number of prime numbers such that n mod 4 == 3?" + } + ] + }' + ``` + + ```bash CLI + ant messages create \ + --transform content --format yaml <<'YAML' + model: claude-sonnet-4-6 + max_tokens: 16000 + thinking: + type: enabled + budget_tokens: 10000 + messages: + - role: user + content: Are there an infinite number of prime numbers such that n mod 4 == 3? + YAML + ``` + + ```python Python + client = anthropic.Anthropic() + + response = client.messages.create( + model="claude-sonnet-4-6", + max_tokens=16000, + thinking={"type": "enabled", "budget_tokens": 10000}, + messages=[ + { + "role": "user", + "content": "Are there an infinite number of prime numbers such that n mod 4 == 3?", + } + ], + ) + + # The response contains summarized thinking blocks and text blocks + for block in response.content: + match block.type: + case "thinking": + print(f"\nThinking summary: {block.thinking}") + case "text": + print(f"\nResponse: {block.text}") + ``` + + ```typescript TypeScript + const client = new Anthropic(); + + const response = await client.messages.create({ + model: "claude-sonnet-4-6", + max_tokens: 16000, + thinking: { + type: "enabled", + budget_tokens: 10000, + }, + messages: [ + { + role: "user", + content: "Are there an infinite number of prime numbers such that n mod 4 == 3?", + }, + ], + }); + + // The response contains summarized thinking blocks and text blocks + for (const block of response.content) { + if (block.type === "thinking") { + console.log(`\nThinking summary: ${block.thinking}`); + } else if (block.type === "text") { + console.log(`\nResponse: ${block.text}`); + } + } + ``` + + ```csharp C# + AnthropicClient client = new(); + + var response = await client.Messages.Create(new() + { + Model = Model.ClaudeSonnet4_6, + MaxTokens = 16000, + Thinking = new ThinkingConfigEnabled(budgetTokens: 10000), + Messages = + [ + new() + { + Role = Role.User, + Content = "Are there an infinite number of prime numbers such that n mod 4 == 3?", + }, + ], + }); + + // The response contains summarized thinking blocks and text blocks + foreach (var block in response.Content) + { + if (block.TryPickThinking(out var thinking)) + { + Console.WriteLine($"\nThinking summary: {thinking.Thinking}"); + } + else if (block.TryPickText(out var text)) + { + Console.WriteLine($"\nResponse: {text.Text}"); + } + } + ``` + + ```go Go + client := anthropic.NewClient() + + response, err := client.Messages.New(context.Background(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeSonnet4_6, + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamOfEnabled(10000), + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("Are there an infinite number of prime numbers such that n mod 4 == 3?")), + }, + }) + if err != nil { + log.Fatal(err) + } + + // The response contains summarized thinking blocks and text blocks + for _, block := range response.Content { + switch block := block.AsAny().(type) { + case anthropic.ThinkingBlock: + fmt.Printf("\nThinking summary: %s", block.Thinking) + case anthropic.TextBlock: + fmt.Printf("\nResponse: %s", block.Text) + } + } + ``` + + ```java Java + import com.anthropic.client.okhttp.AnthropicOkHttpClient; + import com.anthropic.models.messages.MessageCreateParams; + import com.anthropic.models.messages.Model; + + void main() { + var client = AnthropicOkHttpClient.fromEnv(); + + var params = MessageCreateParams.builder() + .model(Model.CLAUDE_SONNET_4_6) + .maxTokens(16_000) + .enabledThinking(10_000) + .addUserMessage("Are there an infinite number of prime numbers such that n mod 4 == 3?") + .build(); + + var response = client.messages().create(params); + + // The response contains summarized thinking blocks and text blocks + for (var block : response.content()) { + block.thinking().ifPresent(thinkingBlock -> + IO.println("\nThinking summary: " + thinkingBlock.thinking()) + ); + block.text().ifPresent(textBlock -> + IO.println("\nResponse: " + textBlock.text()) + ); + } + } + ``` + + ```php PHP + $client = new Client(); + + $response = $client->messages->create( + model: 'claude-sonnet-4-6', + maxTokens: 16000, + thinking: ['type' => 'enabled', 'budget_tokens' => 10000], + messages: [ + [ + 'role' => 'user', + 'content' => 'Are there an infinite number of prime numbers such that n mod 4 == 3?', + ], + ], + ); + + // The response contains summarized thinking blocks and text blocks + foreach ($response->content as $block) { + echo match ($block->type) { + 'thinking' => "\nThinking summary: {$block->thinking}", + 'text' => "\nResponse: {$block->text}", + default => '', + }; + } + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + response = client.messages.create( + model: "claude-sonnet-4-6", + max_tokens: 16_000, + thinking: { + type: :enabled, + budget_tokens: 10_000 + }, + messages: [ + { + role: :user, + content: "Are there an infinite number of prime numbers such that n mod 4 == 3?" + } + ] + ) + + # The response contains summarized thinking blocks and text blocks + response.content.each do |block| + case block + in {type: :thinking, thinking:} + puts "\nThinking summary: #{thinking}" + in {type: :text, text:} + puts "\nResponse: #{text}" + else + end + end + ``` + + + + +5. **Sampling parameters removed:** Sampling parameters (`temperature`, `top_p`, `top_k`) set to a non-default value are not accepted and return a 400 error. + +6. **Cybersecurity safeguards:** Claude Sonnet 5 is the first Sonnet-tier model with real-time cybersecurity safeguards. Requests that involve prohibited or high-risk cybersecurity topics may be refused. Refusals return as a successful HTTP 200 response with `stop_reason: "refusal"`, not an error. See [Safeguards, warnings, and appeals](https://support.claude.com/en/articles/8241253-safeguards-warnings-and-appeals) for background. + +#### Migration checklist + +* Update model name from `claude-sonnet-4-6` to `claude-sonnet-5`. +* Re-run [token counting](/docs/en/build-with-claude/token-counting) against Claude Sonnet 5. The new tokenizer produces approximately 30% more tokens for the same text, which can change per-request cost even though per-token pricing is unchanged. The exact increase depends on the content and workload shape. +* Revisit `max_tokens` limits sized close to your expected output length, and raise them up to the 128k maximum (unchanged from Claude Sonnet 4.6) where useful. +* Remove `thinking: {type: "enabled", budget_tokens: N}` configuration (returns a 400 error). Adaptive thinking is on by default; pass `{type: "disabled"}` to turn it off, or use the [effort parameter](/docs/en/build-with-claude/effort) to control depth. +* Remove `temperature`, `top_p`, and `top_k` parameters set to non-default values (they return a 400 error on Claude Sonnet 5). +* Add handling for `stop_reason: "refusal"` if your workload may touch cybersecurity topics. +* Re-baseline cost on your typical workload before production deployment. +* Review `max_tokens` for workloads that previously ran without thinking. + +### Migrating to Claude Sonnet 5 from Claude Sonnet 4.5 or earlier + +If you are migrating from Claude Sonnet 4.5 or an earlier Sonnet model directly to Claude Sonnet 5, apply the [Migrating to Claude Sonnet 5 from Claude Sonnet 4.6](#migrating-from-claude-sonnet-4-6-to-claude-sonnet-5) changes plus the changes in this section. -```java Java hidelines={1..6,9..11,-2..} -import com.anthropic.client.AnthropicClient; -import com.anthropic.client.okhttp.AnthropicOkHttpClient; -import com.anthropic.models.beta.messages.MessageCreateParams; -import com.anthropic.models.beta.messages.BetaMessage; -import com.anthropic.models.messages.Model; -import com.anthropic.models.beta.AnthropicBeta; -import com.anthropic.models.beta.messages.BetaThinkingConfigEnabled; -import com.anthropic.models.beta.messages.BetaOutputConfig; - -public class Main { - public static void main(String[] args) { - AnthropicClient client = AnthropicOkHttpClient.fromEnv(); - - MessageCreateParams params = MessageCreateParams.builder() - .model(Model.CLAUDE_SONNET_4_6) - .maxTokens(8192L) - .thinking(BetaThinkingConfigEnabled.builder() - .budgetTokens(16384L) - .build()) - .outputConfig(BetaOutputConfig.builder() - .effort(BetaOutputConfig.Effort.LOW) - .build()) - .addBeta(AnthropicBeta.INTERLEAVED_THINKING_2025_05_14) - .addUserMessage("Your prompt here") - .build(); - - BetaMessage response = client.beta().messages().create(params); - System.out.println(response); - } -} -``` + + Claude Sonnet 5 defaults to an effort level of `high`, in contrast to Sonnet 4.5 which had no effort parameter. Consider adjusting the [effort parameter](/docs/en/build-with-claude/effort) as you migrate. If not explicitly set, you may experience higher latency with the default effort level. + -```php PHP hidelines={1..4} -beta->messages->create( - maxTokens: 8192, - messages: [['role' => 'user', 'content' => 'Your prompt here']], - model: 'claude-sonnet-4-6', - thinking: ['type' => 'enabled', 'budget_tokens' => 16384], - outputConfig: ['effort' => 'low'], - betas: ['interleaved-thinking-2025-05-14'], -); + + This is a breaking change when migrating from Sonnet 4.5 or earlier. + -echo array_find($message->content, fn($block) => $block->type === 'text')->text; -``` + Prefilling assistant messages returns a `400` error on Claude Sonnet 4.6 and later models, including Claude Sonnet 5. Use [structured outputs](/docs/en/build-with-claude/structured-outputs), system prompt instructions, or `output_config.format` instead. -```ruby Ruby hidelines={1..2} -require "anthropic" - -client = Anthropic::Client.new - -message = client.beta.messages.create( - model: "claude-sonnet-4-6", - max_tokens: 8192, - thinking: { - type: "enabled", - budget_tokens: 16384 - }, - output_config: { - effort: "low" - }, - betas: ["interleaved-thinking-2025-05-14"], - messages: [ - { role: "user", content: "Your prompt here" } - ] -) -puts message.content.find { |block| block.type == :text }.text -``` - - -### Sonnet 4.6 migration checklist - -- [ ] Update model ID to `claude-sonnet-4-6` -- [ ] **BREAKING:** Remove assistant message prefilling; use structured outputs or `output_config.format` instead -- [ ] **BREAKING:** Verify tool parameter JSON parsing handles escaping differences -- [ ] **BREAKING:** Update tool versions to latest (`text_editor_20250728`, `code_execution_20250825`); legacy versions are not supported (if migrating from 3.x) -- [ ] **BREAKING:** Remove any code using the `undo_edit` command (if applicable) -- [ ] **BREAKING:** Update sampling parameters to use only `temperature` OR `top_p`, not both (if migrating from 3.x) -- [ ] Handle new `refusal` stop reason in your application -- [ ] Remove `fine-grained-tool-streaming-2025-05-14` beta header (now GA) -- [ ] Migrate `output_format` to `output_config.format` -- [ ] Review and update prompts following [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) -- [ ] **Recommended:** Migrate from `thinking: {type: "enabled", budget_tokens: N}` to `thinking: {type: "adaptive"}` with the [effort parameter](/docs/en/build-with-claude/effort) (`budget_tokens` is deprecated and will be removed in a future release) -- [ ] Test in development environment before production deployment + **Common prefill use cases and migrations:** ---- + * **Controlling output formatting** (forcing JSON/YAML output): Use [structured outputs](/docs/en/build-with-claude/structured-outputs) or tools with enum fields for classification tasks. -## Migrating to Claude Sonnet 4.5 + * **Eliminating preambles** (removing "Here is..." phrases): Add direct instructions in the system prompt: "Respond directly without preamble. Do not start with phrases like 'Here is...', 'Based on...', etc." -Claude Sonnet 4.5 combines strong intelligence with fast performance, making it ideal for everyday coding, analysis, and content tasks. + * **Avoiding bad refusals:** Claude is much better at appropriate refusals now. Clear prompting in the user message without prefill should be sufficient. -For a complete overview of capabilities, see the [models overview](/docs/en/about-claude/models/overview). + * **Continuations** (resuming interrupted responses): Move the continuation to the user message: "Your previous response was interrupted and ended with `[previous_response]`. Continue from where you left off." - -Sonnet 4.5 pricing is $3 per million input tokens, $15 per million output tokens. See [Claude pricing](/docs/en/about-claude/pricing) for details. - + * **Context hydration / role consistency** (refreshing context in long conversations): Inject what were previously prefilled-assistant reminders into the user turn instead. -**Update your model name:** +2. **Tool parameter JSON escaping may differ** -```python -# From Sonnet 4 -model = "claude-sonnet-4-20250514" # Before -model = "claude-sonnet-4-5-20250929" # After + + This is a breaking change when migrating from Sonnet 4.5 or earlier. + -# From Sonnet 3.7 -model = "claude-3-7-sonnet-20250219" # Before -model = "claude-sonnet-4-5-20250929" # After -``` + JSON string escaping in tool parameters may differ from previous models. Standard JSON parsers handle this automatically, but custom string-based parsing may need updates. -### Breaking changes +**Extended thinking changes:** `budget_tokens` configurations from Claude Sonnet 4.5 (`thinking: {type: "enabled", budget_tokens: N}`) are not supported on Claude Sonnet 5 and return a 400 error. Adaptive thinking is on by default, so most workloads need no `thinking` configuration at all; use the [effort parameter](/docs/en/build-with-claude/effort) to control thinking depth. If you ran Claude Sonnet 4.5 without extended thinking, pass `thinking: {type: "disabled"}` to preserve that behavior. -These breaking changes apply when migrating from Claude 3.x Sonnet models. +##### When migrating from Claude 3.x -1. **Update sampling parameters** +3. **Remove sampling parameters** - This is a breaking change when migrating from Claude 3.x models. + This is a breaking change when migrating from Claude 3.x models. - Use only `temperature` OR `top_p`, not both. + Sampling parameters (`temperature`, `top_p`, `top_k`) set to a non-default value return a 400 error on Claude Sonnet 5. Remove them from requests, and use prompting to guide the model's behavior instead. -2. **Update tool versions** +4. **Update tool versions** - This is a breaking change when migrating from Claude 3.x models. + This is a breaking change when migrating from Claude 3.x models. - Update to the latest tool versions (`text_editor_20250728`, `code_execution_20250825`). Remove any code using the `undo_edit` command. + Update to the latest tool versions (`text_editor_20250728`, `code_execution_20260521`). Remove any code using the `undo_edit` command. -3. **Handle the `refusal` stop reason** +5. **Handle the `refusal` stop reason** Update your application to [handle `refusal` stop reasons](/docs/en/test-and-evaluate/strengthen-guardrails/handle-streaming-refusals). -4. **Update your prompts for behavioral changes** +6. **Update your prompts for behavioral changes** Claude 4 models have a more concise, direct communication style. Review [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) for optimization guidance. -### Sonnet 4.5 migration checklist - -- [ ] Update model ID to `claude-sonnet-4-5-20250929` -- [ ] **BREAKING:** Update tool versions to latest (`text_editor_20250728`, `code_execution_20250825`); legacy versions are not supported (if migrating from 3.x) -- [ ] **BREAKING:** Remove any code using the `undo_edit` command (if applicable) -- [ ] **BREAKING:** Update sampling parameters to use only `temperature` OR `top_p`, not both (if migrating from 3.x) -- [ ] Handle new `refusal` stop reason in your application -- [ ] Review and update prompts following [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) -- [ ] Consider enabling extended thinking for complex reasoning tasks -- [ ] Test in development environment before production deployment - ---- +*** ## Migrating to Claude Haiku 4.5 @@ -1470,9 +2427,21 @@ Claude Haiku 4.5 is the fastest and most intelligent Haiku model with near-front For a complete overview of capabilities, see the [models overview](/docs/en/about-claude/models/overview). -Haiku 4.5 pricing is $1 per million input tokens, $5 per million output tokens. See [Claude pricing](/docs/en/about-claude/pricing) for details. + For Claude Haiku 4.5 pricing, see [Claude pricing](/docs/en/about-claude/pricing). + + For significant performance improvements on coding and reasoning tasks, consider enabling extended thinking with `thinking: {type: "enabled", budget_tokens: N}`. + + + + Extended thinking impacts [prompt caching](/docs/en/build-with-claude/prompt-caching#caching-with-thinking-blocks) efficiency. + + Extended thinking is deprecated in Claude 4.6 models and removed in Claude Opus 4.7. If using newer models, use [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) instead. + + +### Migrating to Claude Haiku 4.5 from Claude Haiku 3.5 or earlier + **Update your model name:** ```python @@ -1481,36 +2450,26 @@ model = "claude-3-5-haiku-20241022" # Before model = "claude-haiku-4-5-20251001" # After ``` -**Review new rate limits:** Haiku 4.5 has separate rate limits from Haiku 3.5. See [Rate limits documentation](/docs/en/api/rate-limits) for details. - - -For significant performance improvements on coding and reasoning tasks, consider enabling extended thinking with `thinking: {type: "enabled", budget_tokens: N}`. - - - -Extended thinking impacts [prompt caching](/docs/en/build-with-claude/prompt-caching#caching-with-thinking-blocks) efficiency. - -Extended thinking is deprecated in Claude 4.6 models and removed in Claude Opus 4.7. If using newer models, use [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) instead. - +**Review new rate limits:** Haiku 4.5 has separate rate limits from Haiku 3.5. See [Rate limits](/docs/en/api/rate-limits) documentation for details. **Explore new capabilities:** See the [models overview](/docs/en/about-claude/models/overview) for details on context awareness, increased output capacity (64k tokens), higher intelligence, and improved speed. -### Breaking changes +#### Breaking changes These breaking changes apply when migrating from Claude 3.x Haiku models. 1. **Update sampling parameters** - This is a breaking change when migrating from Claude 3.x models. + This is a breaking change when migrating from Claude 3.x models. - Use only `temperature` OR `top_p`, not both. + Use only `temperature` OR `top_p`, not both. Setting both returns a 400 error on Claude Haiku 4.5. 2. **Update tool versions** - This is a breaking change when migrating from Claude 3.x models. + This is a breaking change when migrating from Claude 3.x models. Update to the latest tool versions (`text_editor_20250728`, `code_execution_20250825`). Remove any code using the `undo_edit` command. @@ -1523,23 +2482,23 @@ These breaking changes apply when migrating from Claude 3.x Haiku models. Claude 4 models have a more concise, direct communication style. Review [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) for optimization guidance. -### Haiku 4.5 migration checklist +#### Haiku 4.5 migration checklist -- [ ] Update model ID to `claude-haiku-4-5-20251001` -- [ ] **BREAKING:** Update tool versions to latest (`text_editor_20250728`, `code_execution_20250825`); legacy versions are not supported -- [ ] **BREAKING:** Remove any code using the `undo_edit` command (if applicable) -- [ ] **BREAKING:** Update sampling parameters to use only `temperature` OR `top_p`, not both -- [ ] Handle new `refusal` stop reason in your application -- [ ] Review and adjust for new rate limits (separate from Haiku 3.5) -- [ ] Review and update prompts following [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) -- [ ] Consider enabling extended thinking for complex reasoning tasks -- [ ] Test in development environment before production deployment +* Update model ID to `claude-haiku-4-5-20251001` +* **BREAKING:** Update tool versions to latest (`text_editor_20250728`, `code_execution_20250825`); legacy versions are not supported +* **BREAKING:** Remove any code using the `undo_edit` command (if applicable) +* **BREAKING:** Update sampling parameters to use only `temperature` OR `top_p`, not both (setting both returns a 400 error) +* Handle new `refusal` stop reason in your application +* Review and adjust for new rate limits (separate from Haiku 3.5) +* Review and update prompts following [prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices) +* Consider enabling extended thinking for complex reasoning tasks +* Test in development environment before production deployment ---- +*** ## Get help -- Check the [API documentation](/docs/en/api/overview) for detailed specifications -- Review [model capabilities](/docs/en/about-claude/models/overview) for performance comparisons -- Review [API release notes](/docs/en/release-notes/api) for API updates -- Contact support if you encounter any issues during migration \ No newline at end of file +* Check the [API documentation](/docs/en/api/overview) for detailed specifications +* Review [model capabilities](/docs/en/about-claude/models/overview) for performance comparisons +* Review [API release notes](/docs/en/release-notes/api) for API updates +* Contact support if you encounter any issues during migration diff --git a/content/en/about-claude/models/model-ids-and-versions.md b/content/en/about-claude/models/model-ids-and-versions.md index d56b63640..74cf24bee 100644 --- a/content/en/about-claude/models/model-ids-and-versions.md +++ b/content/en/about-claude/models/model-ids-and-versions.md @@ -14,29 +14,31 @@ Claude model IDs follow a versioned naming scheme. Starting with the Claude 4.6 generation, model IDs use a dateless format: -```text -claude-{name}-{major}-{minor} +```text wrap +claude-{name}-{major}[-{minor}] ``` -For example: `claude-sonnet-4-6`, `claude-opus-4-6`, and `claude-opus-4-7` +Major-version releases such as Claude Sonnet 5 omit the minor segment. + +For example: `claude-sonnet-4-6`, `claude-sonnet-5`, `claude-opus-4-6`, `claude-opus-4-7`, and `claude-opus-4-8` On Amazon Bedrock, the corresponding format is: -```text -anthropic.claude-{name}-{major}-{minor} +```text wrap +anthropic.claude-{name}-{major}[-{minor}] ``` -For example: `anthropic.claude-sonnet-4-6`, `anthropic.claude-opus-4-7` +For example: `anthropic.claude-sonnet-4-6`, `anthropic.claude-sonnet-5`, `anthropic.claude-opus-4-7`, `anthropic.claude-opus-4-8` Claude Opus 4.6 is the last Bedrock model ID to include the `-v1` suffix (`anthropic.claude-opus-4-6-v1`). Anthropic dropped the suffix starting with Claude Sonnet 4.6. -On Google Cloud Vertex AI, the format matches the Claude API. +On Google Cloud, the format matches the Claude API. ### Before the 4.6 generation Models before the 4.6 generation include a snapshot date in the ID: -```text +```text wrap claude-{name}-{major}-{minor}-{YYYYMMDD} ``` @@ -44,15 +46,15 @@ For example: `claude-sonnet-4-5-20250929`, `claude-haiku-4-5-20251001` On Amazon Bedrock, these use the format: -```text +```text wrap anthropic.claude-{name}-{major}-{minor}-{YYYYMMDD}-v1:0 ``` For example: `anthropic.claude-sonnet-4-5-20250929-v1:0` -On Google Cloud Vertex AI, the date is separated with `@`: +On Google Cloud, the date is separated with `@`: -```text +```text wrap claude-{name}-{major}-{minor}@{YYYYMMDD} ``` @@ -78,4 +80,4 @@ Occasionally, infrastructure updates produce minor differences in observable beh ## Current model IDs -For the full list of current model IDs and their Amazon Bedrock and Google Cloud Vertex AI equivalents, see [Models overview](/docs/en/about-claude/models/overview). \ No newline at end of file +For the full list of current model IDs and their Amazon Bedrock and Google Cloud equivalents, see [Models overview](/docs/en/about-claude/models/overview). diff --git a/content/en/about-claude/models/overview.md b/content/en/about-claude/models/overview.md index 2ffcf5b45..7d74e6ef7 100644 --- a/content/en/about-claude/models/overview.md +++ b/content/en/about-claude/models/overview.md @@ -6,114 +6,188 @@ Claude is a family of state-of-the-art large language models developed by Anthro ## Choosing a model -If you're unsure which model to use, consider starting with **Claude Opus 4.7** for the most complex tasks. It is our most capable generally available model, with a step-change improvement in agentic coding over Claude Opus 4.6. +If you're unsure which model to use, start with **Claude Opus 4.8** for complex agentic coding and enterprise work. For workloads that need the highest available capability, use [Claude Fable 5](#claude-fable-5-and-claude-mythos-5). -All current Claude models support text and image input, text output, multilingual capabilities, and vision. Models are available through the Claude API, [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws), [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock), [Vertex AI](/docs/en/build-with-claude/claude-on-vertex-ai), and [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry). +All current Claude models support text and image input, text output, multilingual capabilities, and vision. Models are available through the Claude API, [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock), [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws), [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai), and [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry). Once you've picked a model, [learn how to make your first API call](/docs/en/get-started). +### Claude Fable 5 and Claude Mythos 5 + +Claude Fable 5 (`claude-fable-5`) is Anthropic's most capable widely released model. Claude Mythos 5 (`claude-mythos-5`) shares Claude Fable 5's specs and pricing and joins the invitation-only Claude Mythos Preview (`claude-mythos-preview`) within [Project Glasswing](https://anthropic.com/glasswing). See [Introducing Claude Fable 5 and Claude Mythos 5](/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5) for launch details and API changes. + +Claude Fable 5 is generally available on the Claude API, Amazon Bedrock, Claude Platform on AWS, Google Cloud, and Microsoft Foundry beginning June 9, 2026. Claude Mythos 5 is not generally available: it is offered in limited availability to approved customers in [Project Glasswing](https://anthropic.com/glasswing), beginning the same day. For access, contact your Anthropic, AWS, or Google Cloud account team. + ### Latest models comparison -| Feature | Claude Opus 4.7 | Claude Sonnet 4.6 | Claude Haiku 4.5 | -|:--------|:--------------|:------------------|:-----------------| -| **Description** | Our most capable generally available model for complex reasoning and agentic coding | The best combination of speed and intelligence | The fastest model with near-frontier intelligence | -| **Claude API ID** | claude-opus-4-7 | claude-sonnet-4-6 | claude-haiku-4-5-20251001 | -| **Claude API alias** | claude-opus-4-7 | claude-sonnet-4-6 | claude-haiku-4-5 | -| **AWS Bedrock ID** | anthropic.claude-opus-4-73 | anthropic.claude-sonnet-4-6 | anthropic.claude-haiku-4-5-20251001-v1:0 | -| **Vertex AI ID** | claude-opus-4-7 | claude-sonnet-4-6 | claude-haiku-4-5@20251001 | -| **Pricing**1 | \$5 / input MTok
\$25 / output MTok | \$3 / input MTok
\$15 / output MTok | \$1 / input MTok
\$5 / output MTok | -| **[Extended thinking](/docs/en/build-with-claude/extended-thinking)** | No | Yes | Yes | -| **[Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking)** | Yes | Yes | No | -| **[Priority Tier](/docs/en/api/service-tiers)** | Yes | Yes | Yes | -| **Comparative latency** | Moderate | Fast | Fastest | -| **Context window** | 1M tokens | 1M tokens | 200k tokens | -| **Max output** | 128k tokens | 64k tokens | 64k tokens | -| **Reliable knowledge cutoff** | Jan 20262 | Aug 20252 | Feb 2025 | -| **Training data cutoff** | Jan 2026 | Jan 2026 | Jul 2025 | - -_1 - See [Pricing](/docs/en/about-claude/pricing) for complete pricing information including Batch API discounts and prompt caching rates._ - -_2 - **Reliable knowledge cutoff** indicates the date through which a model's knowledge is most extensive and reliable. **Training data cutoff** is the broader date range of training data used. For more information, see [Anthropic's Transparency Hub](https://www.anthropic.com/transparency)._ - -_3 - Claude Opus 4.7 is available on Bedrock through [Claude in Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) (the Messages-API Bedrock endpoint)._ +| Feature | Claude Fable 5 | Claude Opus 4.8 | Claude Sonnet 5 | Claude Haiku 4.5 | +| --------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------------- | +| **Description** | Next-generation intelligence for long-running agents | For complex agentic coding and enterprise work | The best combination of speed and intelligence | The fastest model with near-frontier intelligence | +| **Claude API ID** | claude-fable-5 | claude-opus-4-8 | claude-sonnet-5 | claude-haiku-4-5-20251001 | +| **Claude API alias** | claude-fable-5 | claude-opus-4-8 | claude-sonnet-5 | claude-haiku-4-5 | +| **AWS Bedrock ID** | anthropic.claude-fable-53 | anthropic.claude-opus-4-83 | anthropic.claude-sonnet-53 | anthropic.claude-haiku-4-5-20251001-v1:0 | +| **Google Cloud ID** | claude-fable-5 | claude-opus-4-8 | claude-sonnet-5 | claude-haiku-4-5\@20251001 | +| **Pricing**1 | $10 / input MTok $50 / output MTok | $5 / input MTok $25 / output MTok | $3 / input MTok $15 / output MTok4 | $1 / input MTok $5 / output MTok | +| **[Extended thinking](/docs/en/build-with-claude/extended-thinking)** | No | No | No | Yes | +| **[Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking)** | Yes (always on) | Yes | Yes | No | +| **Comparative latency** | Slower | Moderate | Fast | Fastest | +| **Context window** | 1M tokens | 1M tokens | 1M tokens | 200k tokens | +| **Max output** | 128k tokens | 128k tokens | 128k tokens | 64k tokens | +| **Reliable knowledge cutoff** | Jan 20262 | Jan 20262 | Jan 20262 | Feb 2025 | +| **Training data cutoff** | Jan 2026 | Jan 2026 | Jan 2026 | Jul 2025 | + +*1 - See [Pricing](/docs/en/about-claude/pricing) for complete pricing information including Batch API discounts and prompt caching rates.* + +*2 - **Reliable knowledge cutoff** indicates the date through which a model's knowledge is most extensive and reliable. **Training data cutoff** is the broader date range of training data used. For more information, see [Anthropic's Transparency Hub](https://www.anthropic.com/transparency).* + +*3 - Claude Fable 5, Claude Opus 4.8, and Claude Sonnet 5 are available on Bedrock through [Claude in Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) (the Messages-API Bedrock endpoint).* + +*4 - Introductory pricing of $2 / $10 per MTok applies to Claude Sonnet 5 through August 31, 2026. See [Pricing](/docs/en/about-claude/pricing#claude-sonnet-5-introductory-pricing).* -[Claude Mythos Preview](https://anthropic.com/glasswing) is offered separately as a research preview model for defensive cybersecurity workflows as part of [Project Glasswing](https://anthropic.com/glasswing). Access is invitation-only and there is no self-serve sign-up. + Claude Mythos 5 and Claude Mythos Preview are offered separately for defensive cybersecurity workflows as part of [Project Glasswing](https://anthropic.com/glasswing). Access is invitation-only and there is no self-serve sign-up. -Every Claude model ID is a pinned snapshot. Models with a date in the ID (for example, `20250929`) are fixed to that specific release. Starting with the Claude 4.6 generation, model IDs use a dateless format that is also a pinned snapshot, not an evergreen pointer. For models before the 4.6 generation, entries in the Claude API alias column are convenience pointers that resolve to a dated model ID. For details on the naming convention and how versioning works, see [Model IDs and versioning](/docs/en/about-claude/models/model-ids-and-versions). + + Every Claude model ID is a pinned snapshot. Models with a date in the ID (for example, -Starting with **Claude Sonnet 4.5 and all subsequent models** (including Claude Sonnet 4.6), Bedrock offers two endpoint types: **global endpoints** (dynamic routing for maximum availability) and **regional endpoints** (guaranteed data routing through specific geographic regions). Vertex AI offers three endpoint types: global endpoints, **multi-region endpoints** (dynamic routing within a geographic area), and regional endpoints. For more information, see [Cloud platform pricing](/docs/en/about-claude/pricing#cloud-platform-pricing). + `20250929` -**Claude Platform on AWS** uses the same model IDs as the Claude API (for example, `claude-opus-4-6`), not Bedrock-style IDs. Model lifecycle on Claude Platform on AWS follows Anthropic's first-party [Model deprecations](/docs/en/about-claude/model-deprecations), not Bedrock's. See [Available models](/docs/en/build-with-claude/claude-platform-on-aws#available-models) for the model list. + ) are fixed to that specific release. Starting with the Claude 4.6 generation, model IDs use a dateless format that is also a pinned snapshot, not an evergreen pointer. For models before the 4.6 generation, entries in the Claude API alias column are convenience pointers that resolve to a dated model ID. For details on the naming convention and how versioning works, see - -You can query model capabilities and token limits programmatically with the [Models API](/docs/en/api/models/list). The response includes `max_input_tokens`, `max_tokens`, and a `capabilities` object for every available model. - + [Model IDs and versioning](/docs/en/about-claude/models/model-ids-and-versions) + + . + -The Max output values above apply to the synchronous Messages API. On the [Message Batches API](/docs/en/build-with-claude/batch-processing#extended-output-beta), Opus 4.7, Opus 4.6, and Sonnet 4.6 support up to 300k output tokens by using the `output-300k-2026-03-24` beta header. + Starting with + + **Claude Sonnet 4.5 and all subsequent models** + + (including Claude Sonnet 4.6), Bedrock offers two endpoint types: + + **global endpoints** + + (dynamic routing for maximum availability) and + + **regional endpoints** + + (guaranteed data routing through specific geographic regions). Google Cloud offers three endpoint types: global endpoints, + + **multi-region endpoints** + + (dynamic routing within a geographic area), and regional endpoints. For more information, see + + [Cloud platform pricing](/docs/en/about-claude/pricing#cloud-platform-pricing) + + . -
+ + **Claude Platform on AWS** -The following models are still available. Consider migrating to current models for improved performance: + uses the same model IDs as the Claude API (for example, -| Feature | Claude Opus 4.6 | Claude Sonnet 4.5 | Claude Opus 4.5 | Claude Opus 4.1 | Claude Sonnet 4 (deprecated) | Claude Opus 4 (deprecated) | -|:--------|:----------------|:------------------|:----------------|:----------------|:----------------|:--------------| -| **Claude API ID** | claude-opus-4-6 | claude-sonnet-4-5-20250929 | claude-opus-4-5-20251101 | claude-opus-4-1-20250805 | claude-sonnet-4-20250514 | claude-opus-4-20250514 | -| **Claude API alias** | claude-opus-4-6 | claude-sonnet-4-5 | claude-opus-4-5 | claude-opus-4-1 | claude-sonnet-4-0 | claude-opus-4-0 | -| **AWS Bedrock ID** | anthropic.claude-opus-4-6-v1 | anthropic.claude-sonnet-4-5-20250929-v1:0 | anthropic.claude-opus-4-5-20251101-v1:0 | anthropic.claude-opus-4-1-20250805-v1:0 | anthropic.claude-sonnet-4-20250514-v1:0 | anthropic.claude-opus-4-20250514-v1:0 | -| **Vertex AI ID** | claude-opus-4-6 | claude-sonnet-4-5@20250929 | claude-opus-4-5@20251101 | claude-opus-4-1@20250805 | claude-sonnet-4@20250514 | claude-opus-4@20250514 | -| **Pricing** | \$5 / input MTok
\$25 / output MTok | \$3 / input MTok
\$15 / output MTok | \$5 / input MTok
\$25 / output MTok | \$15 / input MTok
\$75 / output MTok | \$3 / input MTok
\$15 / output MTok | \$15 / input MTok
\$75 / output MTok | -| **[Extended thinking](/docs/en/build-with-claude/extended-thinking)** | Yes | Yes | Yes | Yes | Yes | Yes | -| **[Priority Tier](/docs/en/api/service-tiers)** | Yes | Yes | Yes | Yes | Yes | Yes | -| **Comparative latency** | Moderate | Fast | Moderate | Moderate | Fast | Moderate | -| **Context window** | 1M tokens | 200k tokens | 200k tokens | 200k tokens | 200k tokens | 200k tokens | -| **Max output** | 128k tokens | 64k tokens | 64k tokens | 32k tokens | 64k tokens | 32k tokens | -| **Reliable knowledge cutoff** | May 20251 | Jan 20251 | May 20251 | Jan 20251 | Jan 20251 | Jan 20251 | -| **Training data cutoff** | Aug 2025 | Jul 2025 | Aug 2025 | Mar 2025 | Mar 2025 | Mar 2025 | + `claude-opus-4-6` - -Claude Sonnet 4 (`claude-sonnet-4-20250514`) and Claude Opus 4 (`claude-opus-4-20250514`) are deprecated and will be retired on June 15, 2026. Migrate to [Claude Sonnet 4.6](/docs/en/about-claude/models/overview#latest-models-comparison) and [Claude Opus 4.7](/docs/en/about-claude/models/overview#latest-models-comparison) respectively before the retirement date. + ), not Bedrock-style IDs. Model lifecycle on Claude Platform on AWS follows Anthropic's first-party -See [model deprecations](/docs/en/about-claude/model-deprecations) for details. - + [Model deprecations](/docs/en/about-claude/model-deprecations) -_1 - **Reliable knowledge cutoff** indicates the date through which a model's knowledge is most extensive and reliable. **Training data cutoff** is the broader date range of training data used._ + , not Bedrock's. See -
+ [Available models](/docs/en/build-with-claude/claude-platform-on-aws#available-models) + + for the model list. + + + + You can query model capabilities and token limits programmatically with the [Models API](/docs/en/api/models/list). The response includes `max_input_tokens`, `max_tokens`, and a `capabilities` object for every available model. + + + + On Claude Opus 4.8, the `effort` parameter defaults to `high` on all surfaces, including the Claude API, Claude Code, and claude.ai. On Claude Sonnet 5, it defaults to `high` on the Claude API and Claude Code. Set `effort` explicitly to use a different level. See [Effort](/docs/en/build-with-claude/effort) for guidance on choosing a level. + + + + The Max output values above apply to the synchronous Messages API. On the [Message Batches API](/docs/en/build-with-claude/batch-processing#extended-output-beta), Claude Opus 4.8, Opus 4.7, Opus 4.6, Sonnet 5, and Sonnet 4.6 support up to 300k output tokens by using the `output-300k-2026-03-24` beta header. + + + + + The following models are still available. Consider migrating to current models for improved performance: + + | Feature | Claude Opus 4.7 | Claude Opus 4.6 | Claude Sonnet 4.6 | Claude Sonnet 4.5 | Claude Opus 4.5 | Claude Opus 4.1 (deprecated) | + | --------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------- | + | **Claude API ID** | claude-opus-4-7 | claude-opus-4-6 | claude-sonnet-4-6 | claude-sonnet-4-5-20250929 | claude-opus-4-5-20251101 | claude-opus-4-1-20250805 | + | **Claude API alias** | claude-opus-4-7 | claude-opus-4-6 | claude-sonnet-4-6 | claude-sonnet-4-5 | claude-opus-4-5 | claude-opus-4-1 | + | **AWS Bedrock ID** | anthropic.claude-opus-4-76 | anthropic.claude-opus-4-6-v1 | anthropic.claude-sonnet-4-6 | anthropic.claude-sonnet-4-5-20250929-v1:0 | anthropic.claude-opus-4-5-20251101-v1:0 | anthropic.claude-opus-4-1-20250805-v1:0 | + | **Google Cloud ID** | claude-opus-4-7 | claude-opus-4-6 | claude-sonnet-4-6 | claude-sonnet-4-5\@20250929 | claude-opus-4-5\@20251101 | claude-opus-4-1\@20250805 | + | **Pricing** | $5 / input MTok $25 / output MTok | $5 / input MTok $25 / output MTok | $3 / input MTok $15 / output MTok | $3 / input MTok $15 / output MTok | $5 / input MTok $25 / output MTok | $15 / input MTok $75 / output MTok | + | **[Extended thinking](/docs/en/build-with-claude/extended-thinking)** | No | Yes | Yes | Yes | Yes | Yes | + | **[Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking)** | Yes | Yes | Yes | No | No | No | + | **Comparative latency** | Moderate | Moderate | Fast | Fast | Moderate | Moderate | + | **Context window** | 1M tokens | 1M tokens | 1M tokens | 200k tokens | 200k tokens | 200k tokens | + | **Max output** | 128k tokens | 128k tokens | 128k tokens | 64k tokens | 64k tokens | 32k tokens | + | **Reliable knowledge cutoff** | Jan 20265 | May 20255 | Aug 20255 | Jan 20255 | May 20255 | Jan 20255 | + | **Training data cutoff** | Jan 2026 | Aug 2025 | Jan 2026 | Jul 2025 | Aug 2025 | Mar 2025 | + + + Claude Opus 4.1 (`claude-opus-4-1-20250805`) is deprecated and will be retired on August 5, 2026. Migrate to [Claude Opus 4.8](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-47) before the retirement date. + + See [model deprecations](/docs/en/about-claude/model-deprecations) for details. + + + *5 - **Reliable knowledge cutoff** indicates the date through which a model's knowledge is most extensive and reliable. **Training data cutoff** is the broader date range of training data used.* + + *6 - Claude Opus 4.7 is available on Bedrock through [Claude in Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) (the Messages-API Bedrock endpoint).* + + ## Prompt and output performance -Claude 4 models excel in: -- **Performance**: Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing. See the [Claude 4 blog post](http://www.anthropic.com/news/claude-4) for more information. -- **Engaging responses**: Claude models are ideal for applications that require rich, human-like interactions. +Current Claude models excel in: + +* **Performance:** Top-tier results in reasoning, coding, multilingual tasks, long-context handling, honesty, and image processing. See [Prompting Claude Sonnet 5](/docs/en/build-with-claude/prompt-engineering/prompting-claude-sonnet-5) and [Prompting Claude Opus 4.8](/docs/en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8) for model-specific prompting guidance. + +* **Engaging responses:** Claude models are ideal for applications that require rich, human-like interactions. - - If you prefer more concise responses, you can adjust your prompts to guide the model toward the desired output length. Refer to the [prompt engineering guides](/docs/en/build-with-claude/prompt-engineering) for details. - - For prompting best practices, see [Prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices). -- **Output quality**: When migrating from previous model generations to Claude 4, you may notice larger improvements in overall performance. + * If you prefer more concise responses, you can adjust your prompts to guide the model toward the desired output length. Refer to the [prompt engineering guides](/docs/en/build-with-claude/prompt-engineering) for details. + * For prompting best practices, see [Prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices). -## Migrating to Claude Opus 4.7 +* **Output quality:** When migrating from a previous model generation, you may notice larger improvements in overall performance. -If you're currently using Claude Opus 4.6 or older Claude models, consider migrating to Claude Opus 4.7 to take advantage of improved intelligence and a step-change jump in agentic coding. For detailed migration instructions, see [Migrating to Claude Opus 4.7](/docs/en/about-claude/models/migration-guide). +## Migrating to Claude Opus 4.8 + +If you're currently using Claude Opus 4.7 or earlier Claude models, see [Migrating to Claude Opus 4.8](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-47). + +If you're currently using Claude Opus 4.6 or older Claude models, see [Migrating to Claude Opus 4.8 from Claude Opus 4.6](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-46). ## Get started with Claude If you're ready to start exploring what Claude can do for you, dive in! Whether you're a developer looking to integrate Claude into your applications or a user wanting to experience the power of AI firsthand, the following resources can help. -Looking to chat with Claude? Visit [claude.ai](http://www.claude.ai)! + + Looking to chat with Claude? Visit + + [claude.ai](https://claude.ai) + + ! + Explore Claude's capabilities and development flow. + Learn how to make your first API call in minutes. + Craft and test powerful prompts directly in your browser. -If you have any questions or need assistance, don't hesitate to reach out to the [support team](https://support.claude.com/) or consult the [Discord community](https://www.anthropic.com/discord). \ No newline at end of file +If you have any questions or need assistance, don't hesitate to reach out to the [support team](https://support.claude.com/) or consult the [Discord community](https://www.anthropic.com/discord). diff --git a/content/en/about-claude/models/whats-new-claude-4-8.md b/content/en/about-claude/models/whats-new-claude-4-8.md new file mode 100644 index 000000000..486b4f227 --- /dev/null +++ b/content/en/about-claude/models/whats-new-claude-4-8.md @@ -0,0 +1,320 @@ +# What's new in Claude Opus 4.8 + +Overview of new features and behavior changes in Claude Opus 4.8. + +--- + +Claude Opus 4.8 is built for complex agentic coding and enterprise work. It builds on Claude Opus 4.7. This page summarizes everything new at launch, including fast mode (research preview on the Claude API) and a lower 1,024-token minimum cacheable prompt length. + +## New model + +| Model | API model ID | Description | +| --------------- | --------------- | ---------------------------------------------- | +| Claude Opus 4.8 | claude-opus-4-8 | For complex agentic coding and enterprise work | + +Claude Opus 4.8 supports the [1M token context window](/docs/en/build-with-claude/context-windows) by default on the Claude API, Amazon Bedrock, Google Cloud, and Microsoft Foundry, 128k max output tokens, [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking), and the same set of tools and platform features as Claude Opus 4.7. + +For complete pricing and specs, see the [models overview](/docs/en/about-claude/models/overview). + +## New features + +### Mid-conversation system messages + +Claude Opus 4.8 accepts `role: "system"` messages immediately after a user turn in the `messages` array (subject to [placement rules](/docs/en/build-with-claude/mid-conversation-system-messages#limitations)). This lets you append updated instructions later in a long-running conversation without restating the full system prompt. Updating instructions this way preserves [prompt cache](/docs/en/build-with-claude/prompt-caching) hits on the earlier turns and reduces input cost on agentic loops. No beta header is required. See [Mid-conversation system messages](/docs/en/build-with-claude/mid-conversation-system-messages) for usage details. + +### Refusal stop details + +The `stop_details` object on refusal responses (available since Claude Opus 4.7) is now publicly documented. When Claude declines to complete a request, this object describes the category of refusal, in addition to the existing `refusal` stop reason. Your application can use it to tell apart different classes of declined request and route the user to the right next step. No beta header is required. See [Refusals and fallback](/docs/en/build-with-claude/refusals-and-fallback#refusal-response) for the category list and [Stop reasons and fallback](/docs/en/build-with-claude/handling-stop-reasons) for handling guidance. + +### Effort defaults + +The [effort parameter](/docs/en/build-with-claude/effort) default on Claude Opus 4.8 is `high` on all surfaces, including the Claude API and Claude Code. If you set effort explicitly today, your setting is unchanged. See [Effort](/docs/en/build-with-claude/effort) for per-level guidance. + +### Fast mode + +[Fast mode](/docs/en/build-with-claude/fast-mode) is now available for Claude Opus 4.8 as a research preview on the Claude API. Set `speed: "fast"` with the `fast-mode-2026-02-01` beta header to get up to 2.5x higher output tokens per second from the same model at premium pricing. See [Fast mode](/docs/en/build-with-claude/fast-mode) for access, supported models, and pricing. + +### Lower prompt cache minimum + +The minimum cacheable prompt length on Claude Opus 4.8 is 1,024 tokens, down from 2,048 tokens on Claude Opus 4.7. Prompts that were too short to cache on Claude Opus 4.7 can now create cache entries with no code changes. See [Prompt caching](/docs/en/build-with-claude/prompt-caching#cache-limitations) for per-model minimums. + +## API constraints inherited from Claude Opus 4.7 + + + These constraints are unchanged from Claude Opus 4.7, so code that already runs on Claude Opus 4.7 needs no changes. They apply to the Messages API only. Claude Managed Agents are unaffected. + + +### Sampling parameters not supported + +Setting `temperature`, `top_p`, or `top_k` to a non-default value returns a 400 error on Claude Opus 4.8, same as on Claude Opus 4.7. Omit these parameters and use prompting to guide the model's behavior. + +### Adaptive thinking is the only thinking mode + +Like Claude Opus 4.7, Claude Opus 4.8 does not support extended thinking budgets. Setting `thinking: {type: "enabled", budget_tokens: N}` returns a 400 error. + +The following diff updates a request written for Claude Opus 4.6 or earlier to run on Claude Opus 4.8. The removed lines (`-`) set the old model ID and the manual thinking budget that Claude Opus 4.8 rejects. The added lines (`+`) set the new model ID, switch to [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking), and control thinking depth with the [effort parameter](/docs/en/build-with-claude/effort), passed in the top-level `output_config` field. The model determines when and how much to think on each turn. If you remove the `thinking` field entirely, requests run without thinking: + + + ```diff cURL + curl https://api.anthropic.com/v1/messages \ + --header "x-api-key: $ANTHROPIC_API_KEY" \ + --header "anthropic-version: 2023-06-01" \ + --header "content-type: application/json" \ + --data \ + '{ + - "model": "claude-opus-4-6", + + "model": "claude-opus-4-8", + "max_tokens": 16000, + "thinking": { + - "type": "enabled", + - "budget_tokens": 10000 + + "type": "adaptive" + }, + + "output_config": { + + "effort": "high" + + }, + "messages": [ + { + "role": "user", + "content": "Explain why the sum of two even numbers is always even." + } + ] + }' + ``` + + ```diff CLI + ant messages create <<'YAML' + -model: claude-opus-4-6 + +model: claude-opus-4-8 + max_tokens: 16000 + thinking: + - type: enabled + - budget_tokens: 10000 + + type: adaptive + +output_config: + + effort: high + messages: + - role: user + content: Explain why the sum of two even numbers is always even. + YAML + ``` + + ```diff Python + import anthropic + + client = anthropic.Anthropic() + + response = client.messages.create( + - model="claude-opus-4-6", + + model="claude-opus-4-8", + max_tokens=16000, + - thinking={"type": "enabled", "budget_tokens": 10000}, + + thinking={"type": "adaptive"}, + + output_config={"effort": "high"}, + messages=[ + { + "role": "user", + "content": "Explain why the sum of two even numbers is always even.", + } + ], + ) + ``` + + ```diff TypeScript + import Anthropic from "@anthropic-ai/sdk"; + + const client = new Anthropic(); + + const response = await client.messages.create({ + - model: "claude-opus-4-6", + + model: "claude-opus-4-8", + max_tokens: 16000, + - thinking: { type: "enabled", budget_tokens: 10000 }, + + thinking: { type: "adaptive" }, + + output_config: { effort: "high" }, + messages: [ + { + role: "user", + content: "Explain why the sum of two even numbers is always even." + } + ] + }); + ``` + + ```diff C# + using Anthropic; + using Anthropic.Models.Messages; + + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + - Model = "claude-opus-4-6", + + Model = Model.ClaudeOpus4_8, + MaxTokens = 16000, + - Thinking = new ThinkingConfigEnabled(budgetTokens: 10000), + + Thinking = new ThinkingConfigAdaptive(), + + OutputConfig = new OutputConfig { Effort = Effort.High }, + Messages = [new() { Role = Role.User, Content = "Explain why the sum of two even numbers is always even." }] + }; + + var response = await client.Messages.Create(parameters); + Console.WriteLine(response); + ``` + + ```diff Go + package main + + import ( + "context" + "fmt" + "log" + + "github.com/anthropics/anthropic-sdk-go" + ) + + func main() { + client := anthropic.NewClient() + + response, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + - Model: "claude-opus-4-6", + + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 16000, + - Thinking: anthropic.ThinkingConfigParamOfEnabled(10000), + + Thinking: anthropic.ThinkingConfigParamUnion{ + + OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{}, + + }, + + OutputConfig: anthropic.OutputConfigParam{ + + Effort: anthropic.OutputConfigEffortHigh, + + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("Explain why the sum of two even numbers is always even.")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(response) + } + ``` + + ```diff Java + import com.anthropic.client.AnthropicClient; + import com.anthropic.client.okhttp.AnthropicOkHttpClient; + import com.anthropic.models.messages.Message; + import com.anthropic.models.messages.MessageCreateParams; + +import com.anthropic.models.messages.Model; + +import com.anthropic.models.messages.OutputConfig; + +import com.anthropic.models.messages.ThinkingConfigAdaptive; + + void main() { + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + - .model("claude-opus-4-6") + + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(16000L) + - .enabledThinking(10000L) + + .thinking(ThinkingConfigAdaptive.builder().build()) + + .outputConfig(OutputConfig.builder() + + .effort(OutputConfig.Effort.HIGH) + + .build()) + .addUserMessage("Explain why the sum of two even numbers is always even.") + .build(); + + Message response = client.messages().create(params); + IO.println(response); + } + ``` + + ```diff PHP + messages->create( + maxTokens: 16000, + messages: [['role' => 'user', 'content' => 'Explain why the sum of two even numbers is always even.']], + - model: 'claude-opus-4-6', + - thinking: ['type' => 'enabled', 'budget_tokens' => 10000], + + model: 'claude-opus-4-8', + + thinking: ['type' => 'adaptive'], + + outputConfig: ['effort' => 'high'], + ); + ``` + + ```diff Ruby + require "anthropic" + + client = Anthropic::Client.new + + response = client.messages.create( + - model: "claude-opus-4-6", + + model: "claude-opus-4-8", + max_tokens: 16000, + - thinking: { type: "enabled", budget_tokens: 10000 }, + + thinking: { type: "adaptive" }, + + output_config: { effort: "high" }, + messages: [ + { role: "user", content: "Explain why the sum of two even numbers is always even." } + ] + ) + ``` + + +## Capability improvements + +### Improvement areas + +Compared with Claude Opus 4.7, Claude Opus 4.8 targets behavioral improvements in: + +* **Long-horizon agentic coding**, including better long-context handling, fewer compactions, and better [compaction](/docs/en/build-with-claude/compaction) recovery. +* **Reasoning effort calibration**, with more reliable behavior at each effort level across a range of domains. +* **Tool triggering**, with fewer cases of skipping a tool call that the task required. + +### Adaptive thinking + +With [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) enabled, Claude Opus 4.8 triggers reasoning only when it determines the turn needs it. On simple lookups and short agentic steps it responds directly. On complex multistep problems it reasons before answering. This reduces wasted thinking tokens on bimodal workloads compared to Claude Opus 4.7 at the same effort level. As on Claude Opus 4.7, thinking is off unless you explicitly set `thinking: {type: "adaptive"}` in your request. + +## Behavior changes + +These are not API breaking changes but might require prompt updates. See [Migrating to Claude Opus 4.8](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-47) for full guidance. + +* **Fewer wasted thinking tokens** at the same effort level when adaptive thinking is enabled, because the model determines per turn whether to think. +* **Better tool triggering.** The model is less likely to skip a tool call the task required, an issue some users reported on Claude Opus 4.7. +* **Better compaction handling and long-context quality.** Long agentic traces stay on task with fewer derailments after compaction. +* **Effort levels recalibrated.** The token allocation behind each effort level changes compared to Claude Opus 4.7: `medium` allows somewhat more thinking, `high` somewhat less, and `xhigh` substantially more. If you tuned an effort level against Claude Opus 4.7, re-baseline cost and latency at that level before adjusting it. + +## Migration guide + +For step-by-step migration instructions and the full migration checklist, see [Migrating to Claude Opus 4.8](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-47). If you are upgrading from Claude Opus 4.6 or earlier, also apply [Migrating to Claude Opus 4.8 from Claude Opus 4.6](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-46). Those steps cover breaking changes that the upgrade from Claude Opus 4.7 alone does not. If you use Claude Code or the Agent SDK, the [Claude API skill](/docs/en/agents-and-tools/agent-skills/claude-api-skill) can apply these migration steps to your code base automatically. + +## Next steps + + + + Guide for migrating to the latest Claude models from previous Claude versions. + + + + Control how many tokens Claude uses when responding with the effort parameter, trading off between response thoroughness and token efficiency. + + + + Let Claude dynamically determine when and how much to use extended thinking with adaptive thinking mode. + + + + How mid-conversation system messages preserve cache hits. + + + + Learn what each stop\_reason value means and how to handle truncation, tool use, paused turns, and refusals in your application. + + + + Get up to 2.5x higher output tokens per second from Claude Opus models. + + diff --git a/content/en/about-claude/models/whats-new-sonnet-5.md b/content/en/about-claude/models/whats-new-sonnet-5.md new file mode 100644 index 000000000..61a52ede8 --- /dev/null +++ b/content/en/about-claude/models/whats-new-sonnet-5.md @@ -0,0 +1,130 @@ +# What's new in Claude Sonnet 5 + +Overview of new features and behavior changes in Claude Sonnet 5. + +--- + +Claude Sonnet 5 is the next generation of Anthropic's Sonnet model family. It is a drop-in upgrade for Claude Sonnet 4.6 with three behavior changes: [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) is on by default, manual extended thinking now returns a 400 error (it was deprecated on Claude Sonnet 4.6), and setting sampling parameters (`temperature`, `top_p`, `top_k`) to non-default values returns a 400 error. This page summarizes everything new at launch, including a new tokenizer. + +## New model + +| Model | API model ID | Description | +| --------------- | ----------------- | ---------------------------------------------- | +| Claude Sonnet 5 | `claude-sonnet-5` | The best combination of speed and intelligence | + +Claude Sonnet 5 supports the [1M token context window](/docs/en/build-with-claude/context-windows) by default (1M tokens is both the default and the maximum; there is no smaller context variant), 128k max output tokens, [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking), and the same set of tools and platform features as Claude Sonnet 4.6, except [Priority Tier](/docs/en/api/service-tiers#supported-models), which is not available on Claude Sonnet 5. + +For complete pricing and specs, see the [models overview](/docs/en/about-claude/models/overview). + +## Behavior changes + +### Adaptive thinking on by default + +On Claude Sonnet 4.6, requests without a `thinking` field run without thinking. On Claude Sonnet 5, the same requests run with [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). To turn thinking off, pass `thinking: {type: "disabled"}`. Because `max_tokens` is a hard limit on total output (thinking plus response text), revisit it for workloads that ran without thinking on Claude Sonnet 4.6. + +### Sampling parameters not accepted + +Setting `temperature`, `top_p`, or `top_k` to a non-default value returns a 400 error. Remove these parameters when migrating; the default value (or omitting the parameter) is accepted. Use system-prompt instructions to guide model behavior. This is new for Sonnet-class models; the same constraint was previously introduced on Claude Opus 4.7. + +### Manual extended thinking removed + +Manual extended thinking (`thinking: {type: "enabled", budget_tokens: N}`) was deprecated on Claude Sonnet 4.6; on Claude Sonnet 5 it is removed and returns a 400 error, the same as on Claude Opus 4.8 and Claude Opus 4.7. Use adaptive thinking with the [effort parameter](/docs/en/build-with-claude/effort) instead. + +```python Python +# Not supported on Claude Sonnet 5 (returns 400) +thinking = {"type": "enabled", "budget_tokens": 32000} + +# Use this instead +thinking = {"type": "adaptive"} +``` + +## New tokenizer + +Claude Sonnet 5 uses a new tokenizer. The same input text produces approximately 30% more tokens than on Claude Sonnet 4.6. The exact increase depends on the content. This is not an API change: requests, responses, and streaming events keep the same shape, and no code changes are required. + +The change affects anything you measure or budget in tokens: + +* **Token counts:** `usage` fields and [token counting](/docs/en/build-with-claude/token-counting) results for the same text are higher than on Claude Sonnet 4.6. Don't reuse counts measured against earlier models; recount against Claude Sonnet 5. +* **Context window capacity in text terms:** the context window is 1M tokens, but each token covers less text on average, so the same window holds less text than on Claude Sonnet 4.6. +* **`max_tokens` budgets:** an output limit tuned for Claude Sonnet 4.6 may truncate equivalent output on Claude Sonnet 5. Revisit limits sized close to your expected output length. +* **Per-request cost:** per-token pricing is unchanged (see [Pricing](#pricing)), but because the same text produces more tokens, the cost of an equivalent request can differ from Claude Sonnet 4.6. + +## API constraints inherited from Claude Sonnet 4.6 + + + This constraint is unchanged from Claude Sonnet 4.6. Aside from the three [behavior changes](#behavior-changes) (see [Migration guide](#migration-guide)), code that already runs on Claude Sonnet 4.6 needs no other changes. + + +### Assistant message prefilling not supported + +Prefilling the assistant message returns a `400` error, unchanged from Claude Sonnet 4.6. Use [structured outputs](/docs/en/build-with-claude/structured-outputs), system prompt instructions, or `output_config.format` instead. + +## Capability improvements + +Claude Sonnet 5 is a capability upgrade over Claude Sonnet 4.6 at the same price. It is also an option for workloads that need more capability than Claude Sonnet 4.6 provides without moving to an Opus-class model. + +The largest gains over Claude Sonnet 4.6 are in coding and agentic tasks. For benchmark results, see [Anthropic's Transparency Hub](https://www.anthropic.com/transparency). + +## Cybersecurity safeguards + +Claude Sonnet 5 is the first Sonnet-tier model with real-time cybersecurity safeguards. Requests that involve prohibited or high-risk cybersecurity topics may be refused. Refusals return as a successful HTTP 200 response with `stop_reason: "refusal"`, not an error. See [Safeguards, warnings, and appeals](https://support.claude.com/en/articles/8241253-safeguards-warnings-and-appeals) for background. + +## Pricing + +Claude Sonnet 5 is priced at $3 per million input tokens and $15 per million output tokens, unchanged from Claude Sonnet 4.6. Because the [new tokenizer](#new-tokenizer) produces approximately 30% more tokens for the same text, the cost of an equivalent request can differ from Claude Sonnet 4.6 even though per-token pricing is unchanged. The exact increase depends on the content and workload shape. + +Introductory pricing of $2/$10 per million input/output tokens is in effect through August 31, 2026, after which the standard pricing of $3/$15 per million input/output tokens will take effect. + +See [Pricing](/docs/en/about-claude/pricing) for complete pricing, including batch processing and prompt caching rates. + +## Availability + +At launch, Claude Sonnet 5 is available on: + +* **Claude API:** available to all customers. +* **AWS:** available through [Claude in Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) and [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws). On Amazon Bedrock, Claude Sonnet 5 is also reachable through the `InvokeModel` API, served by the same infrastructure as Claude in Amazon Bedrock. The legacy [Claude on Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy) integration does not include Claude Sonnet 5. +* **Google Cloud:** available through [Claude on Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai). +* **Microsoft Foundry:** available through [Claude in Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry). + +Claude Sonnet 5 supports [zero data retention](/docs/en/manage-claude/api-and-data-retention) for organizations with ZDR agreements. + +## Migration guide + +Claude Sonnet 5 is a drop-in replacement for Claude Sonnet 4.6. Update your model ID: + +```python +model = "claude-sonnet-4-6" # Before +model = "claude-sonnet-5" # After +``` + +Then review the following: + +1. **Token budgets and counts:** the [new tokenizer](#new-tokenizer) produces approximately 30% more tokens for the same text. The exact increase depends on the content and workload shape. Recount prompts with [token counting](/docs/en/build-with-claude/token-counting), and revisit `max_tokens` limits sized close to your expected output length. +2. **Extended thinking:** if you still set `budget_tokens`, migrate to [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). Manual extended thinking (`thinking: {type: "enabled"}`) is not supported and returns a 400 error. +3. **Sampling parameters:** requests that set sampling parameters (`temperature`, `top_p`, `top_k`) to a non-default value return a 400 error; remove them when migrating. Tool definitions and response shapes are unchanged, and assistant message prefilling was already unsupported on Claude Sonnet 4.6. + +See the [Claude Sonnet 5 section of the migration guide](/docs/en/about-claude/models/migration-guide#migrating-from-claude-sonnet-4-6-to-claude-sonnet-5) for details. + +## Next steps + + + + Complete specs and pricing for all current Claude models. + + + + Measure your prompts under the new tokenizer before you migrate. + + + + The recommended thinking-on mode on Claude Sonnet 5. + + + + How the 1M token context window works. + + + + Complete pricing, including batch processing and prompt caching rates. + + diff --git a/content/en/about-claude/pricing.md b/content/en/about-claude/pricing.md index 784b13196..13fd79577 100644 --- a/content/en/about-claude/pricing.md +++ b/content/en/about-claude/pricing.md @@ -12,73 +12,82 @@ For the most current pricing information, visit [claude.com/pricing](https://cla The following table shows pricing for all Claude models: -| Model | Base Input Tokens | 5m Cache Writes | 1h Cache Writes | Cache Hits & Refreshes | Output Tokens | -|-------------------|-------------------|-----------------|-----------------|----------------------|---------------| -| Claude Opus 4.7 | $5 / MTok | $6.25 / MTok | $10 / MTok | $0.50 / MTok | $25 / MTok | -| Claude Opus 4.6 | $5 / MTok | $6.25 / MTok | $10 / MTok | $0.50 / MTok | $25 / MTok | -| Claude Opus 4.5 | $5 / MTok | $6.25 / MTok | $10 / MTok | $0.50 / MTok | $25 / MTok | -| Claude Opus 4.1 | $15 / MTok | $18.75 / MTok | $30 / MTok | $1.50 / MTok | $75 / MTok | -| Claude Opus 4 | $15 / MTok | $18.75 / MTok | $30 / MTok | $1.50 / MTok | $75 / MTok | -| Claude Sonnet 4.6 | $3 / MTok | $3.75 / MTok | $6 / MTok | $0.30 / MTok | $15 / MTok | -| Claude Sonnet 4.5 | $3 / MTok | $3.75 / MTok | $6 / MTok | $0.30 / MTok | $15 / MTok | -| Claude Sonnet 4 | $3 / MTok | $3.75 / MTok | $6 / MTok | $0.30 / MTok | $15 / MTok | -| Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations)) | $3 / MTok | $3.75 / MTok | $6 / MTok | $0.30 / MTok | $15 / MTok | -| Claude Haiku 4.5 | $1 / MTok | $1.25 / MTok | $2 / MTok | $0.10 / MTok | $5 / MTok | -| Claude Haiku 3.5 | $0.80 / MTok | $1 / MTok | $1.6 / MTok | $0.08 / MTok | $4 / MTok | -| Claude Opus 3 ([deprecated](/docs/en/about-claude/model-deprecations)) | $15 / MTok | $18.75 / MTok | $30 / MTok | $1.50 / MTok | $75 / MTok | -| Claude Haiku 3 | $0.25 / MTok | $0.30 / MTok | $0.50 / MTok | $0.03 / MTok | $1.25 / MTok | +| Model | Base Input Tokens | 5m Cache Writes | 1h Cache Writes | Cache Hits & Refreshes | Output Tokens | +| ------------------------------------------------------------------------------------------------------------- | ----------------- | --------------- | --------------- | ---------------------- | ------------- | +| Claude Fable 5 | $10 / MTok | $12.50 / MTok | $20 / MTok | $1 / MTok | $50 / MTok | +| Claude Mythos 5 ([limited availability](https://anthropic.com/glasswing)) | $10 / MTok | $12.50 / MTok | $20 / MTok | $1 / MTok | $50 / MTok | +| Claude Opus 4.8 | $5 / MTok | $6.25 / MTok | $10 / MTok | $0.50 / MTok | $25 / MTok | +| Claude Opus 4.7 | $5 / MTok | $6.25 / MTok | $10 / MTok | $0.50 / MTok | $25 / MTok | +| Claude Opus 4.6 | $5 / MTok | $6.25 / MTok | $10 / MTok | $0.50 / MTok | $25 / MTok | +| Claude Opus 4.5 | $5 / MTok | $6.25 / MTok | $10 / MTok | $0.50 / MTok | $25 / MTok | +| Claude Opus 4.1 ([deprecated](/docs/en/about-claude/model-deprecations)) | $15 / MTok | $18.75 / MTok | $30 / MTok | $1.50 / MTok | $75 / MTok | +| Claude Opus 4 ([retired, except on Google Cloud](/docs/en/about-claude/model-deprecations)) | $15 / MTok | $18.75 / MTok | $30 / MTok | $1.50 / MTok | $75 / MTok | +| Claude Sonnet 5 [through August 31, 2026](/docs/en/about-claude/pricing#claude-sonnet-5-introductory-pricing) | $2 / MTok | $2.50 / MTok | $4 / MTok | $0.20 / MTok | $10 / MTok | +| Claude Sonnet 5 starting September 1, 2026 | $3 / MTok | $3.75 / MTok | $6 / MTok | $0.30 / MTok | $15 / MTok | +| Claude Sonnet 4.6 | $3 / MTok | $3.75 / MTok | $6 / MTok | $0.30 / MTok | $15 / MTok | +| Claude Sonnet 4.5 | $3 / MTok | $3.75 / MTok | $6 / MTok | $0.30 / MTok | $15 / MTok | +| Claude Sonnet 4 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | $3 / MTok | $3.75 / MTok | $6 / MTok | $0.30 / MTok | $15 / MTok | +| Claude Haiku 4.5 | $1 / MTok | $1.25 / MTok | $2 / MTok | $0.10 / MTok | $5 / MTok | +| Claude Haiku 3.5 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | $0.80 / MTok | $1 / MTok | $1.60 / MTok | $0.08 / MTok | $4 / MTok | + + + Introductory pricing of $2/$10 per million input/output tokens is in effect through August 31, 2026, after which the standard pricing of $3/$15 per million input/output tokens will take effect. + -MTok = Million tokens. The "Base Input Tokens" column shows standard input pricing, the "5m Cache Writes", "1h Cache Writes", and "Cache Hits & Refreshes" columns are specific to [prompt caching](#prompt-caching), and "Output Tokens" shows output pricing. See [prompt caching pricing](#prompt-caching) for an explanation of the cache columns and pricing multipliers. + MTok = Million tokens. The "Base Input Tokens" column shows standard input pricing, the "5m Cache Writes", "1h Cache Writes", and "Cache Hits & Refreshes" columns are specific to [prompt caching](#prompt-caching), and "Output Tokens" shows output pricing. See [prompt caching pricing](#prompt-caching) for an explanation of the cache columns and pricing multipliers. -Opus 4.7 uses a new tokenizer compared to previous models, contributing to its improved performance on a wide range of tasks. This new tokenizer may use up to 35% more tokens for the same fixed text. + Claude Opus 4.7 and later Opus models, Claude Fable 5, Claude Mythos 5, Claude Mythos Preview, and Claude Sonnet 5 use a newer tokenizer that contributes to their improved performance on a wide range of tasks. This tokenizer produces approximately 30% more tokens for the same text. The exact increase depends on the content and workload shape. Claude Sonnet 4.6 and earlier models use the previous tokenizer. For Claude Platform on AWS pricing, see [Claude Platform on AWS pricing](#claude-platform-on-aws-pricing). ## Cloud platform pricing -This section covers partner-operated cloud platforms, where the cloud provider invoices you. For Anthropic-operated cloud platforms billed through a marketplace, see [Claude Platform on AWS pricing](#claude-platform-on-aws-pricing) and [Claude in Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry). +This section covers partner-operated cloud platforms, where the cloud provider invoices you. For Anthropic-operated cloud platforms billed through a marketplace, see [Claude Platform on AWS pricing](#claude-platform-on-aws-pricing) and [Claude in Microsoft Foundry pricing](#claude-in-microsoft-foundry-pricing). + +Claude models are available on [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) and [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai). For official pricing, visit: -Claude models are available on [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) and [Vertex AI](/docs/en/build-with-claude/claude-on-vertex-ai). For official pricing, visit: -- [Amazon Bedrock pricing](https://aws.amazon.com/bedrock/pricing/) -- [Vertex AI pricing](https://cloud.google.com/vertex-ai/generative-ai/pricing#claude-models) +* [Amazon Bedrock pricing](https://aws.amazon.com/bedrock/pricing/) +* [Google Cloud pricing](https://cloud.google.com/vertex-ai/generative-ai/pricing#claude-models) -**Regional and multi-region endpoint pricing for Claude 4.5 models and beyond** + **Regional and multi-region endpoint pricing for Claude 4.5 models and beyond** + + Starting with Claude Sonnet 4.5, Haiku 4.5, and Opus 4.5: -Starting with Claude Sonnet 4.5, Haiku 4.5, and Opus 4.5: -- **Bedrock** offers two endpoint types: global endpoints (dynamic routing for maximum availability) and regional endpoints (guaranteed data routing through specific geographic regions). -- **Vertex AI** offers three endpoint types: global endpoints, multi-region endpoints (dynamic routing within a geographic area), and regional endpoints. + * **Bedrock** offers two endpoint types: global endpoints (dynamic routing for maximum availability) and regional endpoints (guaranteed data routing through specific geographic regions). + * **Google Cloud** offers three endpoint types: global endpoints, multi-region endpoints (dynamic routing within a geographic area), and regional endpoints. -Regional and multi-region endpoints include a 10% premium over global endpoints. The Claude API (first-party) is global by default; for first-party data residency options and pricing, see [Data residency pricing](#data-residency-pricing). + Regional and multi-region endpoints include a 10% premium over global endpoints. The Claude API (first-party) is global by default; for first-party data residency options and pricing, see [Data residency pricing](#data-residency-pricing). -**Scope:** This pricing structure applies to Claude Sonnet 4.5, Haiku 4.5, Opus 4.5, and all future models. Earlier models (Claude Sonnet 4 (deprecated), Opus 4 (deprecated), and prior releases) retain their existing pricing. + **Scope:** This pricing structure applies to Claude Sonnet 4.5, Haiku 4.5, Opus 4.5, and all future models. Earlier models (Claude Opus 4.1 (deprecated) and prior releases) retain their existing pricing. -For implementation details and code examples: -- [Amazon Bedrock global vs regional endpoints](/docs/en/build-with-claude/claude-in-amazon-bedrock#regions) for Opus 4.7, Haiku 4.5, and later models, or [the legacy integration](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy#global-vs-regional-endpoints) for all other models on Bedrock -- [Vertex AI global, multi-region, and regional endpoints](/docs/en/build-with-claude/claude-on-vertex-ai#global-multi-region-and-regional-endpoints) + For implementation details and code examples: + + * [Amazon Bedrock global vs regional endpoints](/docs/en/build-with-claude/claude-in-amazon-bedrock#regions) for Opus 4.7, Haiku 4.5, and later models, or [the legacy integration](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy#global-vs-regional-endpoints) for all other models on Bedrock + * [Google Cloud global, multi-region, and regional endpoints](/docs/en/build-with-claude/claude-on-vertex-ai#global-multi-region-and-regional-endpoints) ## Claude Platform on AWS pricing [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) bills through AWS Marketplace using Claude Consumption Units (CCUs). Anthropic rates your token usage in USD at standard per-model, per-feature rates, applies any negotiated discount, converts the result to CCUs at $0.01 per CCU, and reports the CCU quantity to AWS Marketplace hourly. Your AWS bill shows a single CCU line item. -| Concept | Details | -| :--- | :--- | -| **Billing unit** | Claude Consumption Unit (CCU) | -| **CCU price** | $0.01 per CCU (fixed; discounts apply at token-to-CCU conversion, not to the CCU price) | -| **Conversion** | Token usage rated in USD at standard per-model, per-feature rates (same as [Claude API pricing](#model-pricing)), then converted to CCUs at $0.01 per CCU | -| **Billing cadence** | Hourly metering to AWS Marketplace; monthly invoices | -| **Payment model** | Arrears only (postpaid); no prepaid credits | -| **Discounts** | Applied as fewer CCUs metered | -| **Tax** | Pre-tax metering; AWS Marketplace handles tax | -| **Cost visibility** | Real-time breakdown in the Claude Console (access through the AWS Console); AWS Cost Explorer shows aggregated CCU | +| Concept | Details | +| ------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Billing unit** | Claude Consumption Unit (CCU) | +| **CCU price** | $0.01 per CCU (fixed; discounts apply at token-to-CCU conversion, not to the CCU price) | +| **Conversion** | Token usage rated in USD at standard per-model, per-feature rates (same as [Claude API pricing](#model-pricing)), then converted to CCUs at $0.01 per CCU | +| **Billing cadence** | Hourly metering to AWS Marketplace; monthly invoices | +| **Payment model** | Arrears only (postpaid); no prepaid credits | +| **Discounts** | Applied as fewer CCUs metered | +| **Tax** | Pre-tax metering; AWS Marketplace handles tax | +| **Cost visibility** | Real-time breakdown in the Claude Console (access through the AWS Console); AWS Cost Explorer shows aggregated CCU | -**Claude Consumption Units.** If Customer accesses the Services through certain Marketplace Platforms (e.g., Claude Platform on AWS), usage will be invoiced in Claude Consumption Units ("CCU") rather than per MTok. A CCU is a unit of measure used solely for Marketplace Platform invoicing. One hundred (100) CCU represents $1.00 USD of fees owed for the Services, calculated at the applicable prices on [claude.com/pricing#api](https://claude.com/pricing#api), after application of any discounts. + **Claude Consumption Units.** If Customer accesses the Services through certain Marketplace Platforms (e.g., Claude Platform on AWS), usage will be invoiced in Claude Consumption Units ("CCU") rather than per MTok. A CCU is a unit of measure used solely for Marketplace Platform invoicing. One hundred (100) CCU represents $1.00 USD of fees owed for the Services, calculated at the applicable prices on [claude.com/pricing#api](https://claude.com/pricing#api), after application of any discounts. ### Inference geography @@ -90,9 +99,32 @@ For Claude Opus 4.6, Claude Sonnet 4.6, and later models, using `inference_geo: When you sign up on the AWS Console **Claude Platform on AWS** service page, the AWS Console looks up any private offer associated with your account and prompts you to accept it in AWS Marketplace. Contact your Anthropic account representative for private offer terms. -If you have an existing Amazon Bedrock private offer, contact your Anthropic or AWS account representative before getting started with Claude Platform on AWS to ensure your discounts are applied correctly. Discounts cannot be applied retroactively to usage incurred before your private offer is accepted. + If you have an existing Amazon Bedrock private offer, contact your Anthropic or AWS account representative before getting started with Claude Platform on AWS to ensure your discounts are applied correctly. Discounts cannot be applied retroactively to usage incurred before your private offer is accepted. +## Claude in Microsoft Foundry pricing + +[Claude in Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) bills through the Azure Marketplace using Claude Consumption Units (CCUs). Anthropic rates your token usage in USD at standard per-model, per-feature rates, applies any negotiated discount, converts the result to CCUs at $0.01 per CCU, and reports the CCU quantity to the Azure Marketplace hourly. Your Azure bill shows a single CCU line item. + +| Concept | Details | +| ------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Billing unit** | Claude Consumption Unit (CCU) | +| **CCU price** | $0.01 per CCU (fixed; discounts apply at token-to-CCU conversion, not to the CCU price) | +| **Conversion** | Token usage rated in USD at standard per-model, per-feature rates (same as [Claude API pricing](#model-pricing)), then converted to CCUs at $0.01 per CCU | +| **Billing cadence** | Hourly metering to the Azure Marketplace; monthly invoices | +| **Payment model** | Arrears only (postpaid); no prepaid credits | +| **Discounts** | Applied as fewer CCUs metered | +| **Tax** | Pre-tax metering; Azure Marketplace handles tax | +| **Cost visibility** | Azure Cost Management shows aggregated CCU | + + + **Claude Consumption Units.** If Customer accesses the Services through certain Marketplace Platforms (e.g., Claude Platform on AWS, Claude in Microsoft Foundry), usage will be invoiced in Claude Consumption Units ("CCU") rather than per MTok. A CCU is a unit of measure used solely for Marketplace Platform invoicing. One hundred (100) CCU represents $1.00 USD of fees owed for the Services, calculated at the applicable prices on [claude.com/pricing#api](https://claude.com/pricing#api), after application of any discounts. + + +### Inference geography + +Deployments hosted on Azure can use the US Data Zone Standard deployment type, which keeps inference within the United States. This is equivalent to `inference_geo: "us"` on the Claude API and applies the same 1.1x pricing multiplier. See [Data residency](/docs/en/manage-claude/data-residency) for details. + ## Feature-specific pricing ### Prompt caching @@ -101,16 +133,16 @@ Prompt caching reduces costs and latency by reusing previously processed portion There are two ways to enable prompt caching: -- **Automatic caching:** Add a single `cache_control` field at the top level of your request. The system automatically manages cache breakpoints as conversations grow. This is the recommended starting point for most use cases. -- **Explicit cache breakpoints:** Place `cache_control` directly on individual content blocks for fine-grained control over exactly what gets cached. +* **Automatic caching:** Add a single `cache_control` field at the top level of your request. The system automatically manages cache breakpoints as conversations grow. This is the recommended starting point for most use cases. +* **Explicit cache breakpoints:** Place `cache_control` directly on individual content blocks for fine-grained control over exactly what gets cached. Prompt caching uses the following pricing multipliers relative to base input token rates: -| Cache operation | Multiplier | Duration | -|:----------------|:-----------|:---------| -| 5-minute cache write | 1.25x base input price | Cache valid for 5 minutes | -| 1-hour cache write | 2x base input price | Cache valid for 1 hour | -| Cache read (hit) | 0.1x base input price | Same duration as the preceding write | +| Cache operation | Multiplier | Duration | +| -------------------- | ---------------------- | ------------------------------------ | +| 5-minute cache write | 1.25x base input price | Cache valid for 5 minutes | +| 1-hour cache write | 2x base input price | Cache valid for 1 hour | +| Cache read (hit) | 0.1x base input price | Same duration as the preceding write | Cache write tokens are charged when content is first stored. Cache read tokens are charged when a subsequent request retrieves the cached content. A cache hit costs 10% of the standard input price, which means caching pays off after just one cache read for the 5-minute duration (1.25x write), or after two cache reads for the 1-hour duration (2x write). @@ -122,21 +154,25 @@ For implementation details, supported models, and code examples, see [Prompt cac For Claude Opus 4.6, Claude Sonnet 4.6, and later models, specifying US-only inference through the `inference_geo` parameter incurs a 1.1x multiplier on all token pricing categories, including input tokens, output tokens, cache writes, and cache reads. Global routing (the default) uses standard pricing. -This applies to the Claude API (first-party) and Claude Platform on AWS. Partner-operated platforms (Bedrock and Vertex AI) have independent regional pricing. See [Bedrock](https://aws.amazon.com/bedrock/pricing/) and [Vertex AI](https://cloud.google.com/vertex-ai/generative-ai/pricing#claude-models) for details. Earlier models do not support the `inference_geo` parameter and always use standard pricing; requests that include the parameter on these models return a 400 error. +This applies to the Claude API (first-party) and Claude Platform on AWS. On Claude in Microsoft Foundry, the same 1.1x multiplier applies to deployments that use the US Data Zone Standard deployment type (see [Inference geography](#foundry-inference-geography)). Partner-operated platforms (Bedrock and Google Cloud) have independent regional pricing. See [Bedrock](https://aws.amazon.com/bedrock/pricing/) and [Google Cloud](https://cloud.google.com/vertex-ai/generative-ai/pricing#claude-models) for details. Earlier models do not support the `inference_geo` parameter and always use standard pricing; requests that include the parameter on these models return a 400 error. For more information, see [Data residency](/docs/en/manage-claude/data-residency). ### Fast mode pricing -[Fast mode](/docs/en/build-with-claude/fast-mode), in beta (research preview), provides significantly faster output for Claude Opus 4.6 at premium pricing (6x standard rates). Fast mode pricing applies across the full context window, including requests over 200k input tokens. Fast mode is not available on Claude Platform on AWS. Currently supported on Opus 4.6: +[Fast mode](/docs/en/build-with-claude/fast-mode), in research preview, provides significantly faster output for Claude Opus 4.8 and Claude Opus 4.7 at premium pricing. Fast mode pricing applies across the full context window, including requests over 200k input tokens. Fast mode is not available on Claude Platform on AWS. + +| Model | Input | Output | +| --------------- | ---------- | ----------- | +| Claude Opus 4.8 | $10 / MTok | $50 / MTok | +| Claude Opus 4.7 | $30 / MTok | $150 / MTok | -| Input | Output | -|:------|:-------| -| $30 / MTok | $150 / MTok | +Fast mode for Claude Opus 4.7 is deprecated and will be removed on July 24, 2026. As of June 29, 2026, fast mode is not available on Claude Opus 4.6: requests to `claude-opus-4-6` with `speed: "fast"` run at standard speed and are billed at standard rates. See [Fast mode](/docs/en/build-with-claude/fast-mode#supported-models). Fast mode pricing stacks with other pricing modifiers: -- [Prompt caching multipliers](#prompt-caching) apply on top of fast mode pricing -- [Data residency](/docs/en/manage-claude/data-residency) multipliers apply on top of fast mode pricing + +* [Prompt caching multipliers](#prompt-caching) apply on top of fast mode pricing +* [Data residency](/docs/en/manage-claude/data-residency) multipliers apply on top of fast mode pricing Fast mode is not available with the [Batch API](#batch-processing). @@ -146,31 +182,34 @@ For more information, see [Fast mode](/docs/en/build-with-claude/fast-mode). The Batch API allows asynchronous processing of large volumes of requests with a 50% discount on both input and output tokens. -| Model | Batch input | Batch output | -|-------------------|------------------|-----------------| -| Claude Opus 4.7 | $2.50 / MTok | $12.50 / MTok | -| Claude Opus 4.6 | $2.50 / MTok | $12.50 / MTok | -| Claude Opus 4.5 | $2.50 / MTok | $12.50 / MTok | -| Claude Opus 4.1 | $7.50 / MTok | $37.50 / MTok | -| Claude Opus 4 | $7.50 / MTok | $37.50 / MTok | -| Claude Sonnet 4.6 | $1.50 / MTok | $7.50 / MTok | -| Claude Sonnet 4.5 | $1.50 / MTok | $7.50 / MTok | -| Claude Sonnet 4 | $1.50 / MTok | $7.50 / MTok | -| Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations)) | $1.50 / MTok | $7.50 / MTok | -| Claude Haiku 4.5 | $0.50 / MTok | $2.50 / MTok | -| Claude Haiku 3.5 | $0.40 / MTok | $2 / MTok | -| Claude Opus 3 ([deprecated](/docs/en/about-claude/model-deprecations)) | $7.50 / MTok | $37.50 / MTok | -| Claude Haiku 3 | $0.125 / MTok | $0.625 / MTok | +| Model | Batch input | Batch output | +| ------------------------------------------------------------------------------------------------------------- | ------------ | ------------- | +| Claude Fable 5 | $5 / MTok | $25 / MTok | +| Claude Mythos 5 ([limited availability](https://anthropic.com/glasswing)) | $5 / MTok | $25 / MTok | +| Claude Opus 4.8 | $2.50 / MTok | $12.50 / MTok | +| Claude Opus 4.7 | $2.50 / MTok | $12.50 / MTok | +| Claude Opus 4.6 | $2.50 / MTok | $12.50 / MTok | +| Claude Opus 4.5 | $2.50 / MTok | $12.50 / MTok | +| Claude Opus 4.1 ([deprecated](/docs/en/about-claude/model-deprecations)) | $7.50 / MTok | $37.50 / MTok | +| Claude Opus 4 ([retired, except on Google Cloud](/docs/en/about-claude/model-deprecations)) | $7.50 / MTok | $37.50 / MTok | +| Claude Sonnet 5 [through August 31, 2026](/docs/en/about-claude/pricing#claude-sonnet-5-introductory-pricing) | $1 / MTok | $5 / MTok | +| Claude Sonnet 5 starting September 1, 2026 | $1.50 / MTok | $7.50 / MTok | +| Claude Sonnet 4.6 | $1.50 / MTok | $7.50 / MTok | +| Claude Sonnet 4.5 | $1.50 / MTok | $7.50 / MTok | +| Claude Sonnet 4 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | $1.50 / MTok | $7.50 / MTok | +| Claude Haiku 4.5 | $0.50 / MTok | $2.50 / MTok | +| Claude Haiku 3.5 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | $0.40 / MTok | $2 / MTok | For more information about batch processing, see [Batch processing](/docs/en/build-with-claude/batch-processing). ### Long context pricing -[Claude Mythos Preview](https://anthropic.com/glasswing), Opus 4.7, Opus 4.6, and Sonnet 4.6 include the full [1M token context window](/docs/en/build-with-claude/context-windows) at standard pricing. (A 900k-token request is billed at the same per-token rate as a 9k-token request.) Prompt caching and batch processing discounts apply at standard rates across the full context window. +Claude Fable 5, [Claude Mythos 5](https://anthropic.com/glasswing), [Claude Mythos Preview](https://anthropic.com/glasswing), Claude Opus 4.8, Opus 4.7, Opus 4.6, Sonnet 5, and Sonnet 4.6 include the full [1M token context window](/docs/en/build-with-claude/context-windows) at standard pricing. (A 900k-token request is billed at the same per-token rate as a 9k-token request.) Prompt caching and batch processing discounts apply at standard rates across the full context window. ### Tool use pricing Tool use requests are priced based on: + 1. The total number of input tokens sent to the model (including in the `tools` parameter) 2. The number of output tokens generated 3. For server-side tools, additional usage-based pricing (e.g., web search charges per search performed) @@ -179,28 +218,26 @@ Client-side tools are priced the same as any other Claude API request, while ser The additional tokens from tool use come from: -- The `tools` parameter in API requests (tool names, descriptions, and schemas) -- `tool_use` content blocks in API requests and responses -- `tool_result` content blocks in API requests - -When you use `tools`, we also automatically include a special system prompt for the model which enables tool use. The number of tool use tokens required for each model are listed below (excluding the additional tokens listed above). Note that the table assumes at least 1 tool is provided. If no `tools` are provided, then a tool choice of `none` uses 0 additional system prompt tokens. - -| Model | Tool choice | Tool use system prompt token count | -|--------------------------|------------------------------------------------------|---------------------------------------------| -| Claude Opus 4.7 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Opus 4.6 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Opus 4.5 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Opus 4.1 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Opus 4 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Sonnet 4.6 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Sonnet 4.5 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Sonnet 4 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations)) | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Haiku 4.5 | `auto`, `none`
`any`, `tool` | 346 tokens
313 tokens | -| Claude Haiku 3.5 | `auto`, `none`
`any`, `tool` | 264 tokens
340 tokens | -| Claude Opus 3 ([deprecated](/docs/en/about-claude/model-deprecations)) | `auto`, `none`
`any`, `tool` | 530 tokens
281 tokens | -| Claude Sonnet 3 | `auto`, `none`
`any`, `tool` | 159 tokens
235 tokens | -| Claude Haiku 3 | `auto`, `none`
`any`, `tool` | 264 tokens
340 tokens | +* The `tools` parameter in API requests (tool names, descriptions, and schemas) +* `tool_use` content blocks in API requests and responses +* `tool_result` content blocks in API requests + +When you use `tools`, the API also automatically includes a special system prompt for the model which enables tool use. The number of tool use tokens required for each model are listed below (excluding the additional tokens listed above). Note that the table assumes at least 1 tool is provided. If no `tools` are provided, then a tool choice of `none` uses 0 additional system prompt tokens. + +| Model | Tool choice | Tool use system prompt token count | +| ---------------------------------------------------------------------------------------------------------- | ------------------------------ | ---------------------------------- | +| Claude Opus 4.8 | `auto`, `none`***`any`, `tool` | 290 tokens***410 tokens | +| Claude Opus 4.7 | `auto`, `none`***`any`, `tool` | 675 tokens***804 tokens | +| Claude Opus 4.6 | `auto`, `none`***`any`, `tool` | 497 tokens***589 tokens | +| Claude Opus 4.5 | `auto`, `none`***`any`, `tool` | 496 tokens***588 tokens | +| Claude Opus 4.1 ([deprecated](/docs/en/about-claude/model-deprecations)) | `auto`, `none`***`any`, `tool` | 313 tokens***315 tokens | +| Claude Opus 4 ([retired, except on Google Cloud](/docs/en/about-claude/model-deprecations)) | `auto`, `none`***`any`, `tool` | 313 tokens***315 tokens | +| Claude Sonnet 5 | `auto`, `none`***`any`, `tool` | 354 tokens***474 tokens | +| Claude Sonnet 4.6 | `auto`, `none`***`any`, `tool` | 497 tokens***589 tokens | +| Claude Sonnet 4.5 | `auto`, `none`***`any`, `tool` | 496 tokens***588 tokens | +| Claude Sonnet 4 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | `auto`, `none`***`any`, `tool` | 313 tokens***315 tokens | +| Claude Haiku 4.5 | `auto`, `none`***`any`, `tool` | 496 tokens***588 tokens | +| Claude Haiku 3.5 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | `auto`, `none`***`any`, `tool` | 264 tokens***355 tokens | These token counts are added to your normal input and output tokens to calculate the total cost of a request. @@ -212,34 +249,42 @@ For more information about tool use implementation and best practices, see [Tool #### Bash tool -The bash tool adds **245 input tokens** to your API calls. +The bash tool definition adds the following input tokens to your request. This is in addition to the per-model [tool use system prompt](/docs/en/agents-and-tools/tool-use/overview#pricing) that applies whenever any tool is present. + +| Model | Additional input tokens | +| ----------------------------------------------- | ----------------------- | +| Claude Opus 4.7 and Claude Opus 4.8 | 325 tokens | +| Claude Opus 4.6, Claude Sonnet 4.6, and earlier | 244 tokens | Additional tokens are consumed by: -- Command outputs (stdout/stderr) -- Error messages -- Large file contents + +* Command outputs (stdout/stderr) +* Error messages +* Large file contents See [tool use pricing](#tool-use-pricing) for complete pricing details. #### Code execution tool -**Code execution is free when used with web search or web fetch.** When `web_search_20260209` or `web_fetch_20260209` is included in your API request, there are no additional charges for code execution tool calls beyond the standard input and output token costs. +**Code execution is free when used with web search or web fetch.** When `web_search_20260209` (or later) or `web_fetch_20260209` (or later) is included in your API request, there are no additional charges for code execution tool calls beyond the standard input and output token costs. When used without these tools, code execution is billed by execution time, tracked separately from token usage: -- Execution time has a minimum of 5 minutes -- Each organization receives **1,550 free hours** of usage per month -- Additional usage beyond 1,550 hours is billed at **$0.05 per hour, per container** -- If files are included in the request, execution time is billed even if the tool is not invoked, due to files being preloaded onto the container +* Execution time has a minimum of 5 minutes +* Each organization receives **1,550 free hours** of usage per month +* Additional usage beyond 1,550 hours is billed at **$0.05 USD per hour, per container** +* If files are included in the request, execution time is billed even if the tool is not called, because files are preloaded onto the container Code execution usage is tracked in the response: ```json -"usage": { - "input_tokens": 105, - "output_tokens": 239, - "server_tool_use": { - "code_execution_requests": 1 +{ + "usage": { + "input_tokens": 105, + "output_tokens": 239, + "server_tool_use": { + "code_execution_requests": 1 + } } } ``` @@ -250,10 +295,9 @@ The text editor tool uses the same pricing structure as other tools used with Cl In addition to the base tokens, the following additional input tokens are needed for the text editor tool: -| Tool | Additional input tokens | -| ----------------------------------------- | --------------------------------------- | -| `text_editor_20250429` (Claude 4.x) | 700 tokens | -| `text_editor_20250124` (Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations))) | 700 tokens | +| Tool | Additional input tokens | +| ----------------------------------- | ----------------------- | +| `text_editor_20250429` (Claude 4.x) | 700 tokens | See [tool use pricing](#tool-use-pricing) for complete pricing details. @@ -262,13 +306,15 @@ See [tool use pricing](#tool-use-pricing) for complete pricing details. Web search usage is charged in addition to token usage: ```json -"usage": { - "input_tokens": 105, - "output_tokens": 6039, - "cache_read_input_tokens": 7123, - "cache_creation_input_tokens": 7345, - "server_tool_use": { - "web_search_requests": 1 +{ + "usage": { + "input_tokens": 105, + "output_tokens": 6039, + "cache_read_input_tokens": 7123, + "cache_creation_input_tokens": 7345, + "server_tool_use": { + "web_search_requests": 1 + } } } ``` @@ -282,13 +328,15 @@ Each web search counts as one use, regardless of the number of results returned. Web fetch usage has **no additional charges** beyond standard token costs: ```json -"usage": { - "input_tokens": 25039, - "output_tokens": 931, - "cache_read_input_tokens": 0, - "cache_creation_input_tokens": 0, - "server_tool_use": { - "web_fetch_requests": 1 +{ + "usage": { + "input_tokens": 25039, + "output_tokens": 931, + "cache_read_input_tokens": 0, + "cache_creation_input_tokens": 0, + "server_tool_use": { + "web_fetch_requests": 1 + } } } ``` @@ -298,28 +346,30 @@ The web fetch tool is available on the Claude API at **no additional cost**. You To protect against inadvertently fetching large content that would consume excessive tokens, use the `max_content_tokens` parameter to set appropriate limits based on your use case and budget considerations. Example token usage for typical content: -- Average web page (10 kB): ~2,500 tokens -- Large documentation page (100 kB): ~25,000 tokens -- Research paper PDF (500 kB): ~125,000 tokens + +* Average web page (10 kB): \~2,500 tokens +* Large documentation page (100 kB): \~25,000 tokens +* Research paper PDF (500 kB): \~125,000 tokens #### Computer use tool Computer use follows the standard [tool use pricing](/docs/en/agents-and-tools/tool-use/overview#pricing). When using the computer use tool: -**System prompt overhead**: The computer use beta adds 466-499 tokens to the system prompt +**System prompt overhead:** The computer use beta adds 466–499 tokens to the system prompt + +**Computer use tool token usage:** -**Computer use tool token usage**: -| Model | Input tokens per tool definition | -| ----- | -------------------------------- | -| Claude 4.x models | 735 tokens | -| Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations)) | 735 tokens | +| Model | Input tokens per tool definition | +| ----------------- | -------------------------------- | +| Claude 4.x models | 735 tokens | -**Additional token consumption**: -- Screenshot images (see [Vision pricing](/docs/en/build-with-claude/vision)) -- Tool execution results returned to Claude +**Additional token consumption:** + +* Screenshot images (see [Vision pricing](/docs/en/build-with-claude/vision)) +* Tool execution results returned to Claude -If you're also using bash or text editor tools alongside computer use, those tools have their own token costs as documented in their respective pages. + If you're also using bash or text editor tools alongside computer use, those tools have their own token costs as documented in their respective pages. ## Claude Managed Agents pricing @@ -332,51 +382,52 @@ All tokens consumed by a Claude Managed Agents session are billed at the rates s The following Messages API modifiers do **not** apply to Claude Managed Agents sessions: -| Modifier | Why it doesn't apply | -| --- | --- | -| [Batch API discount](#batch-processing) | Sessions are stateful and interactive. There is no batch mode. | -| [Fast mode premium](#fast-mode-pricing) | Inference speed is managed by the runtime. | -| [Data residency multiplier](#data-residency-pricing) | `inference_geo` is a Messages API request field. | -| [Cloud platform pricing](#cloud-platform-pricing) | Not available on partner-operated cloud platforms. | +| Modifier | Why it doesn't apply | +| ---------------------------------------------------- | -------------------------------------------------------------- | +| [Batch API discount](#batch-processing) | Sessions are stateful and interactive. There is no batch mode. | +| [Fast mode premium](#fast-mode-pricing) | Inference speed is managed by the runtime. | +| [Data residency multiplier](#data-residency-pricing) | `inference_geo` is a Messages API request field. | +| [Cloud platform pricing](#cloud-platform-pricing) | Not available on partner-operated cloud platforms. | ### Session runtime -| SKU | Rate | Metering | -| --- | --- | --- | +| SKU | Rate | Metering | +| --------------- | ---------------------- | ------------------------- | | Session runtime | $0.08 per session-hour | `running` status duration | Runtime is measured to the millisecond and accrues only while the session's status is `running`. Time spent `idle` (waiting for your next message or a tool confirmation), `rescheduling`, or `terminated` does not count toward runtime. -Session runtime replaces the [Code Execution](#code-execution-tool) container-hour billing model when using Claude Managed Agents. You are not separately billed for container hours on top of session runtime. + Session runtime replaces the [Code Execution](#code-execution-tool) container-hour billing model when using Claude Managed Agents. You are not separately billed for container hours on top of session runtime. ### Worked example -A one-hour coding session using Claude Opus 4.7 that consumes 50,000 input tokens and 15,000 output tokens: +A one-hour coding session using Claude Opus 4.8 that consumes 50,000 input tokens and 15,000 output tokens: -| Line item | Calculation | Cost | -| --- | --- | --- | -| Input tokens | 50,000 × $5 / 1,000,000 | $0.25 | -| Output tokens | 15,000 × $25 / 1,000,000 | $0.375 | -| Session runtime | 1.0 hour × $0.08 | $0.08 | -| **Total** | | **$0.705** | +| Line item | Calculation | Cost | +| --------------- | ------------------------ | ---------- | +| Input tokens | 50,000 × $5 / 1,000,000 | $0.25 | +| Output tokens | 15,000 × $25 / 1,000,000 | $0.375 | +| Session runtime | 1.0 hour × $0.08 | $0.08 | +| **Total** | | **$0.705** | If prompt caching is active and 40,000 of the input tokens are cache reads: -| Line item | Calculation | Cost | -| --- | --- | --- | -| Uncached input tokens | 10,000 × $5 / 1,000,000 | $0.05 | -| Cache read tokens | 40,000 × $5 × 0.1 / 1,000,000 | $0.02 | -| Output tokens | 15,000 × $25 / 1,000,000 | $0.375 | -| Session runtime | 1.0 hour × $0.08 | $0.08 | -| **Total** | | **$0.525** | +| Line item | Calculation | Cost | +| --------------------- | ----------------------------- | ---------- | +| Uncached input tokens | 10,000 × $5 / 1,000,000 | $0.05 | +| Cache read tokens | 40,000 × $5 × 0.1 / 1,000,000 | $0.02 | +| Output tokens | 15,000 × $25 / 1,000,000 | $0.375 | +| Session runtime | 1.0 hour × $0.08 | $0.08 | +| **Total** | | **$0.525** | Example calculation for processing 10,000 support tickets: - - Average ~3,700 tokens per conversation - - Using Claude Haiku 4.5 at $1/MTok input, $5/MTok output - - Total cost: ~$37.00 per 10,000 tickets + + * Average \~3,700 tokens per conversation + * Using Claude Haiku 4.5 at $1/MTok input, $5/MTok output + * Total cost: \~$37.00 per 10,000 tickets For a detailed walkthrough of this calculation, see the [customer support agent guide](/docs/en/about-claude/use-case-guides/customer-support-chat). @@ -400,41 +451,39 @@ When building agents with Claude: Rate limits vary by usage tier and affect how many requests you can make: -- **Tier 1:** Entry-level usage with basic limits -- **Tier 2:** Increased limits for growing applications -- **Tier 3:** Higher limits for established applications -- **Tier 4:** Maximum standard limits -- **Enterprise:** Custom limits available +* **Start tier:** Entry-level limits for getting started +* **Build tier:** Increased limits for growing applications +* **Scale tier:** Highest standard limits for production workloads For detailed rate limit information, see [Rate limits](/docs/en/api/rate-limits). -For higher rate limits or custom pricing arrangements, [contact the sales team](https://claude.com/contact-sales). +For limits beyond the Scale tier or custom pricing arrangements, [contact the sales team](https://claude.com/contact-sales). ### Volume discounts Volume discounts may be available for high-volume users. These are negotiated on a case-by-case basis. -- Standard tiers use the pricing shown in [Model pricing](#model-pricing) -- Enterprise customers can [contact sales](mailto:sales@anthropic.com) for custom pricing -- Academic and research discounts may be available +* Standard usage tiers use the pricing shown in [Model pricing](#model-pricing) +* Enterprise customers can [contact sales](mailto:sales@anthropic.com) for custom pricing +* Academic and research discounts may be available ### Enterprise pricing For enterprise customers with specific needs: -- Custom rate limits -- Volume discounts -- Dedicated support -- Custom terms +* Custom rate limits +* Volume discounts +* Dedicated support +* Custom terms Contact the sales team at [sales@anthropic.com](mailto:sales@anthropic.com) or through the [Claude Console](/settings/limits) to discuss enterprise pricing options. ## Billing and payment -- Billing is based on actual monthly usage -- All payments are in USD -- Credit card and invoicing options available -- Usage tracking available in the [Claude Console](/) +* Billing is based on actual monthly usage +* All payments are in USD +* Credit card and invoicing options available +* Usage tracking available in the [Claude Console](/) ## Frequently asked questions @@ -454,4 +503,4 @@ Batch API and prompt caching discounts can be combined. For example, using both Major credit cards are accepted for standard accounts. Enterprise customers can arrange invoicing and other payment methods. -For additional questions about pricing, contact [support@anthropic.com](mailto:support@anthropic.com). \ No newline at end of file +For additional questions about pricing, contact [support@anthropic.com](mailto:support@anthropic.com). diff --git a/content/en/about-claude/use-case-guides/classification.md b/content/en/about-claude/use-case-guides/classification.md new file mode 100644 index 000000000..edf12abc6 --- /dev/null +++ b/content/en/about-claude/use-case-guides/classification.md @@ -0,0 +1,121 @@ +# Classification + +Claude excels at processing, understanding, and recognizing patterns in text, images, and data. These capabilities make Claude especially powerful for classification tasks. + +--- + +This guide walks through the process of determining the best approach for building a classifier with Claude and the essentials of end-to-end deployment for a Claude classifier, from use case exploration to back-end integration. + Visit the + + [classification cookbooks](https://platform.claude.com/cookbook/capabilities-classification-guide) + + to see example classification implementations using Claude. + + +## When to use Claude for classification + +When should you consider using an LLM instead of a traditional ML approach for your classification tasks? Here are some key indicators: + +1. **Rule-based classes**: Use Claude when classes are defined by conditions rather than examples, as it can understand underlying rules. +2. **Evolving classes**: Claude adapts well to new or changing domains with emerging classes and shifting boundaries. +3. **Unstructured inputs**: Claude can handle large volumes of unstructured text inputs of varying lengths. +4. **Limited labeled examples**: With few-shot learning capabilities, Claude learns accurately from limited labeled training data. +5. **Reasoning Requirements**: Claude excels at classification tasks requiring semantic understanding, context, and higher-level reasoning. + +*** + +## Establish your classification use case + +Below is a non-exhaustive list of common classification use cases where Claude excels by industry. + + + + * **Content moderation**: automatically identify and flag inappropriate, offensive, or harmful content in user-generated text, images, or videos. + * **Bug prioritization**: classify software bug reports based on their severity, impact, or complexity to prioritize development efforts and allocate resources effectively. + + + + * **Intent analysis**: determine what the user wants to achieve or what action they want the system to perform based on their text inputs. + * **Support ticket routing**: analyze customer interactions, such as call center transcripts or support tickets, to route issues to the appropriate teams, prioritize critical cases, and identify recurring problems for proactive resolution. + + + + * **Patient triaging**: classify customer intake conversations and data according to the urgency, topic, or required expertise for efficient triaging. + * **Clinical trial screening**: analyze patient data and medical records to identify and categorize eligible participants based on specified inclusion and exclusion criteria. + + + + * **Fraud detection**: identify suspicious patterns or anomalies in financial transactions, insurance claims, or user behavior to prevent and mitigate fraudulent activities. + * **Credit risk assessment**: classify loan applicants based on their creditworthiness into risk categories to automate credit decisions and optimize lending processes. + + + + * **Legal document categorization**: classify legal documents, such as pleadings, motions, briefs, or memoranda, based on their document type, purpose, or relevance to specific cases or clients. + + + +*** + +## Implement Claude for classification + +The three key model decision factors are: intelligence, latency, and price. + +For classification, a smaller model like Claude Haiku 4.5 is typically ideal due to its speed and efficiency. Though, for classification tasks where specialized knowledge or complex reasoning is required, Sonnet or Opus may be a better choice. Learn more about how Opus, Sonnet, and Haiku compare in the [models overview](/docs/en/about-claude/models). + +Use evaluations to gauge whether a Claude model is performing well enough to launch into production. + +### 1. Build a strong input prompt + +While Claude offers high-level baseline performance out of the box, a strong input prompt helps get the best results. + +For a generic classifier that you can adapt to your specific use case, copy the starter prompt below: + + + ```text wrap + You will be building a text classifier that can automatically categorize text into a set of predefined categories. + Here are the categories the classifier will use: + + + {{CATEGORIES}} + + + To help you understand how to classify text into these categories, here are some example texts that have already been labeled with their correct category: + + + {{EXAMPLES}} + + + Please carefully study these examples to identify the key features and characteristics that define each category. Write out your analysis of each category inside tags, explaining the main topics, themes, writing styles, etc. that seem to be associated with each one. + + Once you feel you have a good grasp of the categories, your task is to build a classifier that can take in new, unlabeled texts and output a prediction of which category it most likely belongs to. + + Before giving your final classification, show your step-by-step process and reasoning inside tags. Weigh the evidence for each potential category. + + Then output your final for which category you think the example text belongs to. + + The goal is to build a classifier that can accurately categorize new texts into the most appropriate category, as defined by the examples. + ``` + + +### 2. Develop your test cases + +To run your classification evaluation, you will need test cases to run it on. Take a look at the guide to [developing test cases](/docs/en/test-and-evaluate/develop-tests). + +### 3. Run your eval + +#### Evaluation metrics + +Some success metrics to consider evaluating Claude’s performance on a classification task include: + +| Criteria | Description | +| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Accuracy** | The model's output exactly matches the golden answer or correctly classifies the input according to the task's requirements. This is typically calculated as (Number of correct predictions) / (Overall number of predictions). | +| **F1 Score** | The model's output optimally balances precision and recall. | +| **Consistency** | The model's output is consistent with its predictions for similar inputs or follows a logical pattern. | +| **Structure** | The model's output follows the expected format or structure, making it easy to parse and interpret. For example, many classifiers are expected to output JSON format. | +| **Speed** | The model provides a response within the acceptable time limit or latency threshold for the task. | +| **Bias and Fairness** | If classifying data about people, is it important that the model does not demonstrate any biases based on gender, ethnicity, or other characteristics that would lead to its misclassification. | + +## Deploy your classifier + +To see code examples of how to use Claude for classification, check out the [Classification Guide](https://platform.claude.com/cookbook/capabilities-classification-guide) in the Claude Cookbook. diff --git a/content/en/about-claude/use-case-guides/content-moderation.md b/content/en/about-claude/use-case-guides/content-moderation.md index 32ef9aeeb..2e001ccc9 100644 --- a/content/en/about-claude/use-case-guides/content-moderation.md +++ b/content/en/about-claude/use-case-guides/content-moderation.md @@ -6,7 +6,13 @@ Content moderation is a critical aspect of maintaining a safe, respectful, and p > Visit the [content moderation cookbook](https://platform.claude.com/cookbook/misc-building-moderation-filter) to see an example content moderation implementation using Claude. -This guide is focused on moderating user-generated content within your application. If you're looking for guidance on moderating interactions with Claude, refer to the [guardrails guide](/docs/en/test-and-evaluate/strengthen-guardrails/reduce-hallucinations). + + This guide is focused on moderating user-generated content within your application. If you're looking for guidance on moderating interactions with Claude, refer to the + + [guardrails guide](/docs/en/test-and-evaluate/strengthen-guardrails/reduce-hallucinations) + + . + ## Before building with Claude @@ -14,33 +20,47 @@ Content moderation is a critical aspect of maintaining a safe, respectful, and p Here are some key indicators that you should use an LLM like Claude instead of a traditional ML or rules-based approach for content moderation: -
-Traditional ML methods require significant engineering resources, ML expertise, and infrastructure costs. Human moderation systems incur even higher costs. With Claude, you can have a sophisticated moderation system up and running in a fraction of the time for a fraction of the price. -
-
-Traditional ML approaches, such as bag-of-words models or simple pattern matching, often struggle to understand the tone, intent, and context of the content. While human moderation systems excel at understanding semantic meaning, they require time for content to be reviewed. Claude bridges the gap by combining semantic understanding with the ability to deliver moderation decisions quickly. -
-
-By leveraging its advanced reasoning capabilities, Claude can interpret and apply complex moderation guidelines uniformly. This consistency helps ensure fair treatment of all content, reducing the risk of inconsistent or biased moderation decisions that can undermine user trust. -
-
-Once a traditional ML approach has been established, changing it is a laborious and data-intensive undertaking. On the other hand, as your product or customer needs evolve, Claude can easily adapt to changes or additions to moderation policies without extensive relabeling of training data. -
-
-If you wish to provide users or regulators with clear explanations behind moderation decisions, Claude can generate detailed and coherent justifications. This transparency is important for building trust and ensuring accountability in content moderation practices. -
-
-Traditional ML approaches typically require separate models or extensive translation processes for each supported language. Human moderation requires hiring a workforce fluent in each supported language. Claude’s multilingual capabilities allow it to classify tickets in various languages without the need for separate models or extensive translation processes, streamlining moderation for global customer bases. -
-
-Claude's multimodal capabilities allow it to analyze and interpret content across both text and images. This makes it a versatile tool for comprehensive content moderation in environments where different media types need to be evaluated together. -
- -Anthropic has trained all Claude models to be honest, helpful, and harmless. This may result in Claude moderating content deemed particularly dangerous (in line with the [Acceptable Use Policy](https://www.anthropic.com/legal/aup)), regardless of the prompt used. For example, an adult website that wants to allow users to post explicit sexual content may find that Claude still flags explicit content as requiring moderation, even if they specify in their prompt not to moderate explicit sexual content. Consider reviewing the AUP in advance of building a moderation solution. + + + Traditional ML methods require significant engineering resources, ML expertise, and infrastructure costs. Human moderation systems incur even higher costs. With Claude, you can have a sophisticated moderation system operational in significantly less time and at a much lower cost. + + + + Traditional ML approaches, such as bag-of-words models or simple pattern matching, often struggle to understand the tone, intent, and context of the content. While human moderation systems excel at understanding semantic meaning, they require time for content to be reviewed. Claude addresses both needs by combining semantic understanding with the ability to deliver moderation decisions quickly. + + + + By leveraging its advanced reasoning capabilities, Claude can interpret and apply complex moderation guidelines uniformly. This consistency helps ensure fair treatment of all content, reducing the risk of inconsistent or biased moderation decisions that can undermine user trust. + + + + Once a traditional ML approach has been established, changing it is a laborious and data-intensive undertaking. On the other hand, as your product or customer needs evolve, Claude can easily adapt to changes or additions to moderation policies without extensive relabeling of training data. + + + + If you want to provide users or regulators with clear explanations behind moderation decisions, Claude can generate detailed and coherent justifications. This transparency is important for building trust and ensuring accountability in content moderation practices. + + + + Traditional ML approaches typically require separate models or extensive translation processes for each supported language. Human moderation requires hiring a workforce fluent in each supported language. Claude’s multilingual capabilities allow it to classify tickets in various languages without the need for separate models or extensive translation processes, streamlining moderation for global customer bases. + + + + Claude's multimodal capabilities allow it to analyze and interpret content across both text and images. This makes it a versatile tool for comprehensive content moderation in environments where different media types need to be evaluated together. + + + + + Anthropic has trained all Claude models to be honest, helpful, and harmless. This may result in Claude moderating content deemed particularly dangerous (in line with the + + [Acceptable Use Policy](https://www.anthropic.com/legal/aup) + + ), regardless of the prompt used. For example, an adult website that wants to allow users to post explicit sexual content may find that Claude still flags explicit content as requiring moderation, even if they specify in their prompt not to moderate explicit sexual content. Consider reviewing the AUP in advance of building a moderation solution. + ### Generate examples of content to moderate -Before developing a content moderation solution, first create examples of content that should be flagged and content that should not be flagged. Ensure that you include edge cases and challenging scenarios that may be difficult for a content moderation system to handle effectively. Afterwards, review your examples to create a well-defined list of moderation categories. -For instance, the examples generated by a social media platform might include the following: + +Before developing a content moderation solution, first create examples of content that should be flagged and content that should not be flagged. Ensure that you include edge cases and challenging scenarios that may be difficult for a content moderation system to handle effectively. Afterward, review your examples to create a well-defined list of moderation categories. For instance, the examples generated by a social media platform might include the following: ```python allowed_user_comments = [ @@ -77,44 +97,58 @@ unsafe_categories = [ Effectively moderating these examples requires a nuanced understanding of language. In the comment, `This movie was great, I really enjoyed it. The main actor really killed it!`, the content moderation system needs to recognize that "killed it" is a metaphor, not an indication of actual violence. Conversely, despite the lack of explicit mentions of violence, the comment `Delete this post now or you better hide. I am coming after you and your family.` should be flagged by the content moderation system. -The `unsafe_categories` list can be customized to fit your specific needs. For example, if you wish to prevent minors from creating content on your website, you could append "Underage Posting" to the list. +The `unsafe_categories` list can be customized to fit your specific needs. For example, if you want to prevent minors from creating content on your website, you could append "Underage Posting" to the list. -___ +*** ## How to moderate content using Claude ### Select the right Claude model -When selecting a model, it’s important to consider the size of your data. If costs are a concern, a smaller model like Claude Haiku 4.5 is an excellent choice due to its cost-effectiveness. Below is an estimate of the cost to moderate text for a social media platform that receives one billion posts per month: + +When selecting a model, it’s important to consider the size of your data. If costs are a concern, a smaller model such as Claude Haiku 4.5 is an excellent choice because of its cost-effectiveness. The following is an estimate of the cost to moderate text for a social media platform that receives one billion posts per month: * **Content size** - * Posts per month: 1bn - * Characters per post: 100 - * Total characters: 100bn + + * Posts per month: 1B + * Characters per post: 100 + * Total characters: 100B * **Estimated tokens** - * Input tokens: 28.6bn (assuming 1 token per 3.5 characters) - * Percentage of messages flagged: 3% - * Output tokens per flagged message: 50 - * Total output tokens: 1.5bn + + * Input tokens: 28.6B (assuming 1 token per 3.5 characters) + * Percentage of messages flagged: 3% + * Output tokens per flagged message: 50 + * Total output tokens: 1.5B * **Claude Haiku 4.5 estimated cost** - * Input token cost: 28,600 MTok * \$1.00/MTok = \$28,600 USD - * Output token cost: 1,500 MTok * \$5.00/MTok = \$7,500 USD - * Monthly cost: \$28,600 + \$7,500 = \$36,100 USD -* **Claude Opus 4.7 estimated cost** - * Input token cost: 28,600 MTok * \$5.00/MTok = \$143,000 USD - * Output token cost: 1,500 MTok * \$25.00/MTok = \$37,500 USD - * Monthly cost: \$143,000 + \$37,500 = \$180,500 USD + * Input token cost: 28,600 MTok \* $1.00/MTok = $28,600 USD + * Output token cost: 1,500 MTok \* $5.00/MTok = $7,500 USD + * Monthly cost: $28,600 + $7,500 = $36,100 USD + +* **Claude Opus 4.8 estimated cost** + + * Input token cost: 28,600 MTok \* $5.00/MTok = $143,000 USD + * Output token cost: 1,500 MTok \* $25.00/MTok = $37,500 USD + * Monthly cost: $143,000 + $37,500 = $180,500 USD -Actual costs may differ from these estimates. These estimates are based on the prompt highlighted in the section on [batch processing](#consider-batch-processing). Output tokens can be reduced even further by removing the `explanation` field from the response. + + Actual costs may differ from these estimates. These estimates are based on the prompt highlighted in the section on + + [batch processing](#consider-batch-processing) + + . Output tokens can be reduced even further by removing the + + `explanation` + + field from the response. + ### Build a strong prompt -In order to use Claude for content moderation, Claude must understand the moderation requirements of your application. Let’s start by writing a prompt that allows you to define your moderation needs: +To use Claude for content moderation, Claude must understand the moderation requirements of your application. Start by writing a prompt that allows you to define your moderation needs: -```python Python nocheck hidelines={1} -import anthropic +```python Python import json # Initialize the Anthropic client @@ -182,7 +216,7 @@ for comment in user_comments: In this example, the `moderate_message` function contains an assessment prompt that includes the unsafe content categories and the message to evaluate. The prompt asks Claude to assess whether the message should be moderated, based on the unsafe categories defined above. -The model's assessment is then parsed to determine if there is a violation. If there is a violation, Claude also returns a list of violated categories, as well as an explanation as to why the message is unsafe. +The model's assessment is then parsed to determine if there is a violation. If there is a violation, Claude also returns a list of violated categories and an explanation as to why the message is unsafe. ### Evaluate your prompt @@ -190,8 +224,7 @@ Content moderation is a classification problem. Thus, you can use the same techn One additional consideration is that instead of treating content moderation as a binary classification problem, you may instead create multiple categories to represent various risk levels. Creating multiple risk levels allows you to adjust the aggressiveness of your moderation. For example, you might want to automatically block user queries that are deemed high risk, while users with many medium risk queries are flagged for human review. -```python Python nocheck hidelines={1} -import anthropic +```python Python import json # Initialize the Anthropic client @@ -267,19 +300,19 @@ This code implements an `assess_risk_level` function that uses Claude to evaluat Within the function, a prompt is generated for Claude, including the message to be assessed, the unsafe categories, and specific instructions for evaluating the risk level. The prompt instructs Claude to respond with a JSON object that includes the risk level, the violated categories, and an optional explanation. -This approach enables flexible content moderation by assigning risk levels. It can be seamlessly integrated into a larger system to automate content filtering or flag comments for human review based on their assessed risk level. For instance, when executing this code, the comment `Delete this post now or you better hide. I am coming after you and your family.` is identified as high risk due to its dangerous threat. Conversely, the comment `Stay away from the 5G cellphones!! They are using 5G to control you.` is categorized as medium risk. +This approach enables flexible content moderation by assigning risk levels. It can be seamlessly integrated into a larger system to automate content filtering or flag comments for human review based on their assessed risk level. For instance, when running this code, the comment `Delete this post now or you better hide. I am coming after you and your family.` is identified as high risk because of its dangerous threat. Conversely, the comment `Stay away from the 5G cellphones!! They are using 5G to control you.` is categorized as medium risk. ### Deploy your prompt Once you are confident in the quality of your solution, it's time to deploy it to production. Here are some best practices to follow when using content moderation in production: -1. **Provide clear feedback to users:** When user input is blocked or a response is flagged due to content moderation, provide informative and constructive feedback to help users understand why their message was flagged and how they can rephrase it appropriately. In the earlier coding examples, this is done through the `explanation` field in the Claude response. +1. **Provide clear feedback to users:** When user input is blocked or a response is flagged because of content moderation, provide informative and constructive feedback to help users understand why their message was flagged and how they can rephrase it appropriately. In the earlier coding examples, this is done through the `explanation` field in the Claude response. 2. **Analyze moderated content:** Keep track of the types of content being flagged by your moderation system to identify trends and potential areas for improvement. 3. **Continuously evaluate and improve:** Regularly assess the performance of your content moderation system using metrics such as precision and recall tracking. Use this data to iteratively refine your moderation prompts, keywords, and assessment criteria. -___ +*** ## Improve performance @@ -289,8 +322,7 @@ In complex scenarios, it may be helpful to consider additional strategies to imp In addition to listing the unsafe categories in the prompt, further improvements can be made by providing definitions and phrases related to each category. -```python Python nocheck hidelines={1} -import anthropic +```python Python import json # Initialize the Anthropic client @@ -387,8 +419,7 @@ Notably, the definition for the `Specialized Advice` category now specifies the To reduce costs in situations where real-time moderation isn't necessary, consider moderating messages in batches. Include multiple messages within the prompt's context, and ask Claude to assess which messages should be moderated. -```python Python nocheck hidelines={1} -import anthropic +```python Python import json # Initialize the Anthropic client @@ -457,15 +488,14 @@ Explanation: {violation["explanation"]} """) ``` -In this example, the `batch_moderate_messages` function handles the moderation of an entire batch of messages with a single Claude API call. -Inside the function, a prompt is created that includes the list of messages to evaluate and the unsafe content categories. The prompt directs Claude to return a JSON object listing all messages that contain violations. Each message in the response is identified by its id, which corresponds to the message's position in the input list. -Keep in mind that finding the optimal batch size for your specific needs may require some experimentation. While larger batch sizes can lower costs, they might also lead to a slight decrease in quality. Additionally, you may need to increase the `max_tokens` parameter in the Claude API call to accommodate longer responses. For details on the maximum number of tokens your chosen model can output, refer to the [model comparison table](/docs/en/about-claude/models/overview#latest-models-comparison). +In this example, the `batch_moderate_messages` function handles the moderation of an entire batch of messages with a single Claude API call. Inside the function, a prompt is created that includes the list of messages to evaluate and the unsafe content categories. The prompt directs Claude to return a JSON object listing all messages that contain violations. Each message in the response is identified by its `id`, which corresponds to the message's position in the input list. Keep in mind that finding the optimal batch size for your specific needs may require some experimentation. While larger batch sizes can lower costs, they might also lead to a slight decrease in quality. Additionally, you may need to increase the `max_tokens` parameter in the Claude API call to accommodate longer responses. For details on the maximum number of tokens your chosen model can output, refer to the [model comparison table](/docs/en/about-claude/models/overview#latest-models-comparison). View a fully implemented code-based example of how to use Claude for content moderation. + Explore the guardrails guide for techniques to moderate interactions with Claude. - \ No newline at end of file + diff --git a/content/en/about-claude/use-case-guides/customer-support-chat.md b/content/en/about-claude/use-case-guides/customer-support-chat.md index de9bf98ef..790b95385 100644 --- a/content/en/about-claude/use-case-guides/customer-support-chat.md +++ b/content/en/about-claude/use-case-guides/customer-support-chat.md @@ -1,46 +1,55 @@ # Customer support agent -This guide walks through how to leverage Claude's advanced conversational capabilities to handle customer inquiries in real time, providing 24/7 support, reducing wait times, and managing high support volumes with accurate responses and positive interactions. +Build a customer support chatbot with Claude that answers product questions, stays on topic, and generates quotes through tool use. --- +## Prerequisites + +To follow this guide, you need: + +* A Claude API key (set as the `ANTHROPIC_API_KEY` environment variable) +* Python 3.9 or later + +Install the required packages: + +```bash +pip install anthropic streamlit python-dotenv +``` + ## Before building with Claude ### Decide whether to use Claude for support chat Here are some key indicators that you should employ an LLM like Claude to automate portions of your customer support process: -
- + + Claude excels at handling a large number of similar questions efficiently, freeing up human agents for more complex issues. - -
-
+ + Claude can quickly retrieve, process, and combine information from vast knowledge bases, while human agents may need time to research or consult multiple sources. - -
-
+ + Claude can provide round-the-clock support without fatigue, whereas staffing human agents for continuous coverage can be costly and challenging. - -
-
+ + Claude can handle sudden increases in query volume without the need for hiring and training additional staff. - -
-
+ + You can instruct Claude to consistently represent your brand's tone and values, whereas human agents may vary in their communication styles. - -
+ + Some considerations for choosing Claude over other LLMs: -- You prioritize natural, nuanced conversation: Claude's sophisticated language understanding allows for more natural, context-aware conversations that feel more human-like than chats with other LLMs. -- You often receive complex and open-ended queries: Claude can handle a wide range of topics and inquiries without generating canned responses or requiring extensive programming of permutations of user utterances. -- You need scalable multilingual support: Claude's multilingual capabilities allow it to engage in conversations in over 200 languages without the need for separate chatbots or extensive translation processes for each supported language. +* You prioritize natural, nuanced conversation: Claude's sophisticated language understanding allows for more natural, context-aware conversations that feel more human-like than chats with other LLMs. +* You often receive complex and open-ended queries: Claude can handle a wide range of topics and inquiries without generating canned responses or requiring extensive programming of permutations of user utterances. +* You need scalable multilingual support: Claude's multilingual capabilities allow it to engage in conversations in over 200 languages without the need for separate chatbots or extensive translation processes for each supported language. ### Define your ideal chat interaction @@ -49,51 +58,72 @@ Outline an ideal customer interaction to define how and when you expect the cust Here is an example chat interaction for car insurance customer support: * **Customer:** Initiates support chat experience - * **Claude:** Warmly greets customer and initiates conversation + * **Claude:** Warmly greets customer and initiates conversation + * **Customer:** Asks about insurance for their new electric car - * **Claude:** Provides relevant information about electric vehicle coverage + * **Claude:** Provides relevant information about electric vehicle coverage + * **Customer:** Asks questions related to unique needs for electric vehicle insurances - * **Claude:** Responds with accurate and informative answers and provides links to the sources + * **Claude:** Responds with accurate and informative answers and provides links to the sources + * **Customer:** Asks off-topic questions unrelated to insurance or cars - * **Claude:** Clarifies it does not discuss unrelated topics and steers the user back to car insurance + * **Claude:** Clarifies it does not discuss unrelated topics and steers the user back to car insurance + * **Customer:** Expresses interest in an insurance quote - * **Claude:** Ask a set of questions to determine the appropriate quote, adapting to their responses - * **Claude:** Sends a request to use the quote generation API tool along with necessary information collected from the user - * **Claude:** Receives the response information from the API tool use, synthesizes the information into a natural response, and presents the provided quote to the user + + * **Claude:** Ask a set of questions to determine the appropriate quote, adapting to their responses + * **Claude:** Sends a request to use the quote generation API tool along with necessary information collected from the user + * **Claude:** Receives the response information from the API tool use, synthesizes the information into a natural response, and presents the provided quote to the user + * **Customer:** Asks follow up questions - * **Claude:** Answers follow up questions as needed - * **Claude:** Guides the customer to the next steps in the insurance process and closes out the conversation -In the real example that you write for your own use case, you might find it useful to write out the actual words in this interaction so that you can also get a sense of the ideal tone, response length, and level of detail you want Claude to have. + * **Claude:** Answers follow up questions as needed + * **Claude:** Guides the customer to the next steps in the insurance process and closes out the conversation + + + In the real example that you write for your own use case, you might find it useful to write out the actual words in this interaction so that you can also get a sense of the ideal tone, response length, and level of detail you want Claude to have. + ### Break the interaction into unique tasks Customer support chat is a collection of multiple different tasks, from question answering to information retrieval to taking action on requests, wrapped up in a single customer interaction. Before you start building, break down your ideal customer interaction into every task you want Claude to be able to perform. This ensures you can prompt and evaluate Claude for every task, and gives you a good sense of the range of interactions you need to account for when writing test cases. -Customers sometimes find it helpful to visualize this as an interaction flowchart of possible conversation inflection points depending on user requests. + + Customers sometimes find it helpful to visualize this as an interaction flowchart of possible conversation inflection points depending on user requests. + -Here are the key tasks associated with the example insurance interaction above: +Here are the key tasks associated with the example insurance interaction: 1. Greeting and general guidance - - Warmly greet the customer and initiate conversation - - Provide general information about the company and interaction - -2. Product Information - - Provide information about electric vehicle coverage - This will require that Claude have the necessary information in its context, and might imply that a [RAG integration](https://platform.claude.com/cookbook/capabilities-retrieval-augmented-generation-guide) is necessary. - - Answer questions related to unique electric vehicle insurance needs - - Answer follow-up questions about the quote or insurance details - - Offer links to sources when appropriate - -3. Conversation Management - - Stay on topic (car insurance) - - Redirect off-topic questions back to relevant subjects - -4. Quote Generation - - Ask appropriate questions to determine quote eligibility - - Adapt questions based on customer responses - - Submit collected information to quote generation API - - Present the provided quote to the customer + + * Warmly greet the customer and initiate conversation + * Provide general information about the company and interaction + +2. Product information + + * Provide information about electric vehicle coverage + + This will require that Claude have the necessary information in its context, and might imply that a + + [RAG integration](https://platform.claude.com/cookbook/capabilities-retrieval-augmented-generation-guide) + + is necessary. + + * Answer questions related to unique electric vehicle insurance needs + * Answer follow-up questions about the quote or insurance details + * Offer links to sources when appropriate + +3. Conversation management + + * Stay on topic (car insurance) + * Redirect off-topic questions back to relevant subjects + +4. Quote generation + + * Ask appropriate questions to determine quote eligibility + * Adapt questions based on customer responses + * Submit collected information to quote generation API + * Present the provided quote to the customer ### Establish success criteria @@ -101,64 +131,55 @@ Work with your support team to [define success criteria and write detailed evalu Here are criteria and benchmarks that can be used to evaluate how successfully Claude performs the defined tasks: -
- + + This metric evaluates how accurately Claude understands customer inquiries across various topics. Measure this by reviewing a sample of conversations and assessing whether Claude has the correct interpretation of customer intent, critical next steps, what successful resolution looks like, and more. Aim for a comprehension accuracy of 95% or higher. - -
-
+ + This assesses how well Claude's response addresses the customer's specific question or issue. Evaluate a set of conversations and rate the relevance of each response (using LLM-based grading for scale). Target a relevance score of 90% or above. - -
-
+ + Assess the correctness of general company and product information provided to the user, based on the information provided to Claude in context. Target 100% accuracy in this introductory information. - -
-
+ + Track the frequency and relevance of links or sources offered. Target providing relevant sources in 80% of interactions where additional information could be beneficial. - -
-
+ + Measure how well Claude stays on topic, such as the topic of car insurance in the example implementation. Aim for 95% of responses to be directly related to car insurance or the customer's specific query. - -
-
+ + Measure how successful Claude is at determining when to generate informational content and how relevant that content is. For example, in this implementation, you would be determining how well Claude understands when to generate a quote and how accurate that quote is. Target 100% accuracy, as this is vital information for a successful customer interaction. - -
-
+ + This measures Claude's ability to recognize when a query needs human intervention and escalate appropriately. Track the percentage of correctly escalated conversations versus those that should have been escalated but weren't. Aim for an escalation accuracy of 95% or higher. - -
+ + Here are criteria and benchmarks that can be used to evaluate the business impact of employing Claude for support: -
- + + This assesses Claude's ability to maintain or improve customer sentiment throughout the conversation. Use sentiment analysis tools to measure sentiment at the beginning and end of each conversation. Aim for maintained or improved sentiment in 90% of interactions. - -
-
+ + The percentage of customer inquiries successfully handled by the chatbot without human intervention. Typically aim for 70-80% deflection rate, depending on the complexity of inquiries. - -
-
+ + A measure of how satisfied customers are with their chatbot interaction. Usually done through post-interaction surveys. Aim for a CSAT score of 4 out of 5 or higher. - -
-
+ + The average time it takes for the chatbot to resolve an inquiry. This varies widely based on the complexity of issues, but generally, aim for a lower AHT compared to human agents. - -
+ + ## How to implement Claude as a customer service agent @@ -166,13 +187,13 @@ Here are criteria and benchmarks that can be used to evaluate the business impac The choice of model depends on the trade-offs between cost, accuracy, and response time. -For customer support chat, Claude Opus 4.7 is well suited to balance intelligence, latency, and cost. However, for instances where you have conversation flow with multiple prompts including RAG, tool use, and/or long-context prompts, Claude Haiku 4.5 may be more suitable to optimize for latency. +For customer support chat, Claude Opus 4.8 is well suited to balance intelligence, latency, and cost. However, for instances where you have conversation flow with multiple prompts including RAG, tool use, or long-context prompts, Claude Haiku 4.5 may be more suitable to optimize for latency. ### Build a strong prompt Using Claude for customer support requires Claude having enough direction and context to respond appropriately, while having enough flexibility to handle a wide range of customer inquiries. -Start by writing the elements of a strong prompt, starting with a system prompt: +Start by writing the elements of a strong prompt, beginning with a system prompt. Create a file called `config.py` and add each of the following blocks to it: ```python IDENTITY = """You are Eva, a friendly and knowledgeable AI assistant for Acme Insurance @@ -181,11 +202,19 @@ Acme's insurance offerings, which include car insurance and electric car insurance. You can also help customers get quotes for their insurance needs.""" ``` -While you may be tempted to put all your information inside a system prompt as a way to separate instructions from the user conversation, Claude actually works best with the bulk of its prompt content written inside the first `User` turn (with the only exception being role prompting). Read more at [Giving Claude a role with a system prompt](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#give-claude-a-role). + + While you may be tempted to put all your information inside a system prompt as a way to separate instructions from the user conversation, Claude actually works best with the bulk of its prompt content written inside the first + + `User` + + turn (with the only exception being role prompting). Read more at -It's best to break down complex prompts into subsections and write one part at a time. For each task, you might find greater success by following a step by step process to define the parts of the prompt Claude would need to do the task well. For this car insurance customer support example, you'll be writing piecemeal all the parts for a prompt starting with the "Greeting and general guidance" task. This also makes debugging your prompt easier as you can more quickly adjust individual parts of the overall prompt. + [Giving Claude a role with a system prompt](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#give-claude-a-role) -Put all of these pieces in a file called `config.py`. + . + + +It's best to break down complex prompts into subsections and write one part at a time. For each task, you might find greater success by following a step-by-step process to define the parts of the prompt Claude would need to do the task well. For this car insurance customer support example, you'll be writing piecemeal all the parts for a prompt starting with the "Greeting and general guidance" task. This also makes debugging your prompt easier as you can more quickly adjust individual parts of the overall prompt. ```python STATIC_GREETINGS_AND_GENERAL = """ @@ -320,8 +349,7 @@ Once you provide this information, I'll use our quoting tool to generate a perso """ ``` -You will also want to include any important instructions outlining Do's and Don'ts for how Claude should interact with the customer. -This may draw from brand guardrails or support policies. +You will also want to include any important instructions outlining dos and don'ts for how Claude should interact with the customer. This may draw from brand guardrails or support policies. ```python ADDITIONAL_GUARDRAILS = """Please adhere to the following guardrails: @@ -337,7 +365,7 @@ You only provide information and guidance. Now combine all these sections into a single string to use as your prompt. -```python nocheck +```python TASK_SPECIFIC_INSTRUCTIONS = " ".join( [ STATIC_GREETINGS_AND_GENERAL, @@ -351,15 +379,21 @@ TASK_SPECIFIC_INSTRUCTIONS = " ".join( ### Add dynamic and agentic capabilities with tool use -Claude is capable of taking actions and retrieving information dynamically using client-side tool use functionality. Start by listing any external tools or APIs the prompt should utilize. +Claude is capable of taking actions and retrieving information dynamically using client-side tool use functionality. Start by listing any external tools or APIs the prompt should use. For this example, start with one tool for calculating the quote. -As a reminder, this tool will not perform the actual calculation, it will just signal to the application that a tool should be used with whatever arguments specified. + + As a reminder, this tool will not perform the actual calculation, it will just signal to the application that a tool should be used with whatever arguments specified. + -Example insurance quote calculator: +Add the model name, the tool definition, and a stub implementation to `config.py`: ```python +import time + +MODEL = "claude-opus-4-8" + TOOLS = [ { "name": "get_quote", @@ -398,13 +432,13 @@ def get_quote(make, model, year, mileage, driver_age): ### Deploy your prompts -It's hard to know how well your prompt works without deploying it in a test production setting and [running evaluations](/docs/en/test-and-evaluate/develop-tests) so let's build a small application using the prompt, the Anthropic SDK, and streamlit for a user interface. +It's hard to know how well your prompt works without deploying it in a test production setting and [running evaluations](/docs/en/test-and-evaluate/develop-tests). Build a small application using the prompt, the Anthropic SDK, and Streamlit for a user interface. In a file called `chatbot.py`, start by setting up the ChatBot class, which will encapsulate the interactions with the Anthropic SDK. The class should have two main methods: `generate_message` and `process_user_input`. -```python nocheck +```python from anthropic import Anthropic from config import IDENTITY, TOOLS, MODEL, get_quote from dotenv import load_dotenv @@ -506,7 +540,7 @@ Test deploying this code with Streamlit using a main method. This `main()` funct Do this in a file called `app.py` -```python nocheck +```python import streamlit as st from chatbot import ChatBot from config import TASK_SPECIFIC_INSTRUCTIONS @@ -554,7 +588,13 @@ streamlit run app.py Prompting often requires testing and optimization for it to be production ready. To determine the readiness of your solution, evaluate the chatbot performance using a systematic process combining quantitative and qualitative methods. Creating a [strong empirical evaluation](/docs/en/test-and-evaluate/develop-tests#building-evals-and-test-cases) based on your defined success criteria will allow you to optimize your prompts. -The [Claude Console](/dashboard) now features an Evaluation tool that allows you to test your prompts under various scenarios. + + The + + [Claude Console](/dashboard) + + now features an Evaluation tool that lets you test your prompts under various scenarios. + ### Improve performance @@ -562,35 +602,36 @@ In complex scenarios, it may be helpful to consider additional strategies to imp #### Reduce long context latency with RAG -When dealing with large amounts of static and dynamic context, including all information in the prompt can lead to high costs, slower response times, and reaching context window limits. In this scenario, implementing Retrieval Augmented Generation (RAG) techniques can significantly improve performance and efficiency. +When dealing with large amounts of static and dynamic context, including all information in the prompt can lead to high costs, slower response times, and reaching context window limits. In this scenario, implementing Retrieval Augmented Generation (RAG) techniques can improve performance and efficiency. By using [embedding models like Voyage](/docs/en/build-with-claude/embeddings) to convert information into vector representations, you can create a more scalable and responsive system. This approach allows for dynamic retrieval of relevant information based on the current query, rather than including all possible context in every prompt. -Implementing RAG for support use cases [RAG recipe](https://platform.claude.com/cookbook/capabilities-retrieval-augmented-generation-guide) has been shown to increase accuracy, reduce response times, and reduce API costs in systems with extensive context requirements. +Implementing RAG for support use cases has been shown to increase accuracy, reduce response times, and reduce API costs in systems with extensive context requirements. See the [RAG recipe](https://platform.claude.com/cookbook/capabilities-retrieval-augmented-generation-guide) for a worked example. #### Integrate real-time data with tool use -When dealing with queries that require real-time information, such as account balances or policy details, embedding-based RAG approaches are not sufficient. Instead, you can leverage tool use to significantly enhance your chatbot's ability to provide accurate, real-time responses. For example, you can use tool use to look up customer information, retrieve order details, and cancel orders on behalf of the customer. +When dealing with queries that require real-time information, such as account balances or policy details, embedding-based RAG approaches are not sufficient. Instead, tool use can enhance your chatbot's ability to provide accurate, real-time responses. For example, you can use tool use to look up customer information, retrieve order details, and cancel orders on behalf of the customer. -This approach, [outlined in the tool use: customer service agent recipe](https://platform.claude.com/cookbook/tool-use-customer-service-agent), allows you to seamlessly integrate live data into your Claude's responses and provide a more personalized and efficient customer experience. +This approach, [outlined in the tool use: customer service agent recipe](https://platform.claude.com/cookbook/tool-use-customer-service-agent), lets you integrate live data into Claude's responses and provide a more personalized and efficient customer experience. #### Strengthen input and output guardrails -When deploying a chatbot, especially in customer service scenarios, it's crucial to prevent risks associated with misuse, out-of-scope queries, and inappropriate responses. While Claude is inherently resilient to such scenarios, here are additional steps to strengthen your chatbot guardrails: +When deploying a chatbot, especially in customer service scenarios, it's important to prevent risks associated with misuse, out-of-scope queries, and inappropriate responses. While Claude is inherently resilient to such scenarios, here are additional steps to strengthen your chatbot guardrails: -- [Reduce hallucination](/docs/en/test-and-evaluate/strengthen-guardrails/reduce-hallucinations): Implement fact-checking mechanisms and [citations](https://platform.claude.com/cookbook/misc-using-citations) to ground responses in provided information. -- Cross-check information: Verify that the agent's responses align with your company's policies and known facts. -- Avoid contractual commitments: Ensure the agent doesn't make promises or enter into agreements it's not authorized to make. -- [Mitigate jailbreaks](/docs/en/test-and-evaluate/strengthen-guardrails/mitigate-jailbreaks): Use methods like harmlessness screens and input validation to prevent users from exploiting model vulnerabilities, aiming to generate inappropriate content. -- Avoid mentioning competitors: Implement a competitor mention filter to maintain brand focus and not mention any competitor's products or services. -- [Increase output consistency](/docs/en/test-and-evaluate/strengthen-guardrails/increase-consistency): Prevent Claude from changing style or going out of character, even during long, complex interactions. -- Remove Personally Identifiable Information (PII): Unless explicitly required and authorized, strip out any PII from responses. +* [Reduce hallucination](/docs/en/test-and-evaluate/strengthen-guardrails/reduce-hallucinations): Implement fact-checking mechanisms and [citations](https://platform.claude.com/cookbook/misc-using-citations) to ground responses in provided information. +* Cross-check information: Verify that the agent's responses align with your company's policies and known facts. +* Avoid contractual commitments: Ensure the agent doesn't make promises or enter into agreements it's not authorized to make. +* [Mitigate jailbreaks](/docs/en/test-and-evaluate/strengthen-guardrails/mitigate-jailbreaks): Use methods like harmlessness screens and input validation to prevent users from exploiting model vulnerabilities, aiming to generate inappropriate content. +* Avoid mentioning competitors: Implement a competitor mention filter to maintain brand focus and not mention any competitor's products or services. +* [Increase output consistency](/docs/en/test-and-evaluate/strengthen-guardrails/increase-consistency): Prevent Claude from changing style or going out of character, even during long, complex interactions. +* Remove Personally Identifiable Information (PII): Unless explicitly required and authorized, strip out any PII from responses. #### Reduce perceived response time with streaming -When dealing with potentially lengthy responses, implementing streaming can significantly improve user engagement and satisfaction. In this scenario, users receive the answer progressively instead of waiting for the entire response to be generated. +When dealing with potentially lengthy responses, implementing streaming can improve user engagement and satisfaction. In this scenario, users receive the answer progressively instead of waiting for the entire response to be generated. Here is how to implement streaming: + 1. Use the [Anthropic Streaming API](/docs/en/build-with-claude/streaming) to support streaming responses. 2. Set up your frontend to handle incoming chunks of text. 3. Display each chunk as it arrives, simulating real-time typing. @@ -598,33 +639,45 @@ Here is how to implement streaming: In some cases, streaming enables the use of more advanced models with higher base latencies, as the progressive display mitigates the impact of longer processing times. -#### Scale your Chatbot +#### Scale your chatbot -As the complexity of your Chatbot grows, your application architecture can evolve to match. Before you add further layers to your architecture, consider the following less exhaustive options: +As the complexity of your chatbot grows, your application architecture can evolve to match. Before you add further layers to your architecture, consider the following less exhaustive options: -- Ensure that you are making the most out of your prompts and optimizing through prompt engineering. Use the [prompt engineering guides](/docs/en/build-with-claude/prompt-engineering/overview) to write the most effective prompts. -- Add additional [tools](/docs/en/build-with-claude/tool-use) to the prompt (which can include [prompt chains](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#chain-complex-prompts)) and see if you can achieve the functionality required. +* Ensure that you are making the most out of your prompts and optimizing through prompt engineering. Use the [prompt engineering guides](/docs/en/build-with-claude/prompt-engineering/overview) to write the most effective prompts. +* Add additional [tools](/docs/en/agents-and-tools/tool-use/overview) to the prompt (which can include [prompt chains](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#chain-complex-prompts)) and see if you can achieve the functionality required. -If your Chatbot handles incredibly varied tasks, you may want to consider adding a [separate intent classifier](https://platform.claude.com/cookbook/capabilities-classification-guide) to route the initial customer query. For the existing application, this would involve creating a decision tree that would route customer queries through the classifier and then to specialized conversations (with their own set of tools and system prompts). Note, this method requires an additional call to Claude that can increase latency. +If your chatbot handles incredibly varied tasks, you may want to consider adding a [separate intent classifier](https://platform.claude.com/cookbook/capabilities-classification-guide) to route the initial customer query. For the existing application, this would involve creating a decision tree that would route customer queries through the classifier and then to specialized conversations (with their own set of tools and system prompts). Note, this method requires an additional call to Claude that can increase latency. ### Integrate Claude into your support workflow -While the examples above have focused on Python functions callable within a Streamlit environment, deploying Claude for real-time support chatbot requires an API service. +While these examples have focused on Python functions callable within a Streamlit environment, deploying Claude for real-time support chatbot requires an API service. Here's how you can approach this: 1. Create an API wrapper: Develop a simple API wrapper around your classification function. For example, you can use Flask API or Fast API to wrap your code into a HTTP Service. Your HTTP service could accept the user input and return the Assistant response in its entirety. Thus, your service could have the following characteristics: - - Server-Sent Events (SSE): SSE allows for real-time streaming of responses from the server to the client. This is crucial for providing a smooth, interactive experience when working with LLMs. - - Caching: Implementing caching can significantly improve response times and reduce unnecessary API calls. - - Context retention: Maintaining context when a user navigates away and returns is important for continuity in conversations. + + * Server-Sent Events (SSE): SSE allows for real-time streaming of responses from the server to the client. This provides a smooth, interactive experience when working with LLMs. + * Caching: Implementing caching can improve response times and reduce unnecessary API calls. + * Context retention: Maintaining context when a user navigates away and returns is important for continuity in conversations. 2. Build a web interface: Implement a user-friendly web UI for interacting with the Claude-powered agent. +## Next steps + - - Visit the RAG cookbook recipe for more example code and detailed guidance. + + Give Claude access to your APIs so it can take action on behalf of customers. - - Explore the Citations cookbook recipe for how to ensure accuracy and explainability of information. + + + Build evaluations to measure your support agent against the success criteria you defined. + + + + Stream responses so customers see answers as they generate. + + + + Refine your system prompt and examples for better task performance. - \ No newline at end of file + diff --git a/content/en/about-claude/use-case-guides/legal-summarization.md b/content/en/about-claude/use-case-guides/legal-summarization.md index 20cbb74e3..16c759df5 100644 --- a/content/en/about-claude/use-case-guides/legal-summarization.md +++ b/content/en/about-claude/use-case-guides/legal-summarization.md @@ -10,28 +10,35 @@ This guide walks through how to leverage Claude's advanced natural language proc ### Decide whether to use Claude for legal summarization -Here are some key indicators that you should employ an LLM like Claude to summarize legal documents: - -
-Large-scale document review can be time-consuming and expensive when done manually. Claude can process and summarize vast amounts of legal documents rapidly, significantly reducing the time and cost associated with document review. This capability is particularly valuable for tasks like due diligence, contract analysis, or litigation discovery, where efficiency is crucial. -
-
-Claude can efficiently extract and categorize important metadata from legal documents, such as parties involved, dates, contract terms, or specific clauses. This automated extraction can help organize information, making it easier to search, analyze, and manage large document sets. It's especially useful for contract management, compliance checks, or creating searchable databases of legal information. -
-
-Claude can generate structured summaries that follow predetermined formats, making it easier for legal professionals to quickly grasp the key points of various documents. These standardized summaries can improve readability, facilitate comparison between documents, and enhance overall comprehension, especially when dealing with complex legal language or technical jargon. -
-
-When creating legal summaries, proper attribution and citation are crucial to ensure credibility and compliance with legal standards. Claude can be prompted to include accurate citations for all referenced legal points, making it easier for legal professionals to review and verify the summarized information. -
-
-Claude can assist in legal research by quickly analyzing large volumes of case law, statutes, and legal commentary. It can identify relevant precedents, extract key legal principles, and summarize complex legal arguments. This capability can significantly speed up the research process, allowing legal professionals to focus on higher-level analysis and strategy development. -
+Here are some key indicators that you should employ an LLM such as Claude to summarize legal documents: + + + + Large-scale document review can be time-consuming and expensive when done manually. Claude can process and summarize vast amounts of legal documents rapidly, significantly reducing the time and cost associated with document review. This capability is particularly valuable for tasks like due diligence, contract analysis, or litigation discovery, where efficiency is crucial. + + + + Claude can efficiently extract and categorize important metadata from legal documents, such as parties involved, dates, contract terms, or specific clauses. This automated extraction can help organize information, making it easier to search, analyze, and manage large document sets. It's especially useful for contract management, compliance checks, or creating searchable databases of legal information. + + + + Claude can generate structured summaries that follow predetermined formats, making it easier for legal professionals to quickly grasp the key points of various documents. These standardized summaries can improve readability, facilitate comparison between documents, and enhance overall comprehension, especially when dealing with complex legal language or technical jargon. + + + + When creating legal summaries, proper attribution and citation are crucial to ensure credibility and compliance with legal standards. Claude can be prompted to include accurate citations for all referenced legal points, making it easier for legal professionals to review and verify the summarized information. + + + + Claude can assist in legal research by quickly analyzing large volumes of case law, statutes, and legal commentary. It can identify relevant precedents, extract key legal principles, and summarize complex legal arguments. This capability can significantly speed up the research process, allowing legal professionals to focus on higher-level analysis and strategy development. + + ### Determine the details you want the summarization to extract + There is no single correct summary for any given document. Without clear direction, it can be difficult for Claude to determine which details to include. To achieve optimal results, identify the specific information you want to include in the summary. -For instance, when summarizing a sublease agreement, you might wish to extract the following key points: +For instance, when summarizing a sublease agreement, you might want to extract the following key points: ```python details_to_extract = [ @@ -46,68 +53,85 @@ details_to_extract = [ ### Establish success criteria -Evaluating the quality of summaries is a notoriously challenging task. Unlike many other natural language processing tasks, evaluation of summaries often lacks clear-cut, objective metrics. The process can be highly subjective, with different readers valuing different aspects of a summary. Here are criteria you may wish to consider when assessing how well Claude performs legal summarization. - -
-The summary should accurately represent the facts, legal concepts, and key points in the document. -
-
-Terminology and references to statutes, case law, or regulations must be correct and aligned with legal standards. -
-
- The summary should condense the legal document to its essential points without losing important details. -
-
-If summarizing multiple documents, the LLM should maintain a consistent structure and approach to each summary. -
-
-The text should be clear and easy to understand. If the audience is not legal experts, the summarization should not include legal jargon that could confuse the audience. -
-
-The summary should present an unbiased and fair depiction of the legal arguments and positions. -
+Evaluating the quality of summaries is a notoriously challenging task. Unlike many other natural language processing tasks, evaluation of summaries often lacks clear-cut, objective metrics. The process can be highly subjective, with different readers valuing different aspects of a summary. Here are criteria you may want to consider when assessing how well Claude performs legal summarization. + + + + The summary should accurately represent the facts, legal concepts, and key points in the document. + + + + Terminology and references to statutes, case law, or regulations must be correct and aligned with legal standards. + + + + The summary should condense the legal document to its essential points without losing important details. + + + + If summarizing multiple documents, the LLM should maintain a consistent structure and approach to each summary. + + + + The text should be clear and easy to understand. If the audience is not legal experts, the summarization should not include legal jargon that could confuse the audience. + + + + The summary should present an unbiased and fair depiction of the legal arguments and positions. + + See the guide on [establishing success criteria](/docs/en/test-and-evaluate/develop-tests) for more information. ---- +*** ## How to summarize legal documents using Claude ### Select the right Claude model -Model accuracy is extremely important when summarizing legal documents. Claude Opus 4.7 is an excellent choice for use cases such as this where high accuracy is required. If the size and quantity of your documents is large such that costs start to become a concern, you can also try using a smaller model like Claude Haiku 4.5. +Model accuracy is extremely important when summarizing legal documents. Claude Opus 4.8 is an excellent choice for use cases such as this where high accuracy is required. If the size and quantity of your documents is large such that costs start to become a concern, you can also try using a smaller model such as Claude Haiku 4.5. To help estimate these costs, the following is a comparison of the cost to summarize 1,000 sublease agreements using both Opus and Haiku: * **Content size** - * Number of agreements: 1,000 - * Characters per agreement: 300,000 - * Total characters: 300M + + * Number of agreements: 1,000 + * Characters per agreement: 300,000 + * Total characters: 300M * **Estimated tokens** - * Input tokens: 86M (assuming 1 token per 3.5 characters) - * Output tokens per summary: 350 - * Total output tokens: 350,000 -* **Claude Opus 4.7 estimated cost** - * Input token cost: 86 MTok * \$5.00/MTok = \$430.00 USD - * Output token cost: 0.35 MTok * \$25.00/MTok = \$8.75 USD - * Total cost: \$430.00 + \$8.75 = \$438.75 USD + * Input tokens: 86M (assuming 1 token per 3.5 characters) + * Output tokens per summary: 350 + * Total output tokens: 350,000 + +* **Claude Opus 4.8 estimated cost** + + * Input token cost: 86 MTok \* $5.00/MTok = $430.00 USD + * Output token cost: 0.35 MTok \* $25.00/MTok = $8.75 USD + * Total cost: $430.00 + $8.75 = $438.75 USD * **Claude Haiku 4.5 estimated cost** - * Input token cost: 86 MTok * \$1.00/MTok = \$86.00 USD - * Output token cost: 0.35 MTok * \$5.00/MTok = \$1.75 USD - * Total cost: \$86.00 + \$1.75 = \$87.75 USD -Actual costs may differ from these estimates. These estimates are based on the example highlighted in the section on [prompting](#build-a-strong-prompt). + * Input token cost: 86 MTok \* $1.00/MTok = $86.00 USD + * Output token cost: 0.35 MTok \* $5.00/MTok = $1.75 USD + * Total cost: $86.00 + $1.75 = $87.75 USD + + + Actual costs may differ from these estimates. These estimates are based on the example highlighted in the + + [Build a strong prompt](#build-a-strong-prompt) + + section. + ### Transform documents into a format that Claude can process Before you begin summarizing documents, you need to prepare your data. This involves extracting text from PDFs, cleaning the text, and ensuring it's ready to be processed by Claude. -Here is a demonstration of this process on a sample pdf: +Here is a demonstration of this process on a sample PDF: -```python nocheck +```python from io import BytesIO import re @@ -129,7 +153,7 @@ def get_llm_text(pdf_file): # Create the full URL from the GitHub repository -url = "https://raw.githubusercontent.com/anthropics/anthropic-cookbook/main/skills/summarization/data/Sample Sublease Agreement.pdf" +url = "https://raw.githubusercontent.com/anthropics/claude-cookbooks/main/capabilities/summarization/data/Sample Sublease Agreement.pdf" url = url.replace(" ", "%20") # Download the PDF file into memory @@ -142,25 +166,23 @@ document_text = get_llm_text(pdf_file) print(document_text[:50000]) ``` -In this example, you first download a pdf of a sample sublease agreement used in the [summarization cookbook](https://platform.claude.com/cookbook/capabilities-summarization-guide). This agreement was sourced from a publicly available sublease agreement from the [sec.gov website](https://www.sec.gov/Archives/edgar/data/1045425/000119312507044370/dex1032.htm). +In this example, you first download a PDF of a sample sublease agreement used in the [summarization cookbook](https://platform.claude.com/cookbook/capabilities-summarization-guide). This agreement was sourced from a publicly available sublease agreement from the [sec.gov website](https://www.sec.gov/Archives/edgar/data/1045425/000119312507044370/dex1032.htm). -The example uses the pypdf library to extract the contents of the pdf and convert it to text. The text data is then cleaned by removing page numbers and extra whitespace. +The example uses the pypdf library to extract the contents of the PDF and convert it to text. The text data is then cleaned by removing page numbers and extra whitespace. ### Build a strong prompt -Claude can adapt to various summarization styles. You can change the details of the prompt to guide Claude to be more or less verbose, include more or less technical terminology, or provide a higher or lower level summary of the context at hand. +Claude can adapt to various summarization styles. You can change the details of the prompt to guide Claude to be more or less verbose, include more or less technical terminology, or provide a higher- or lower-level summary of the context at hand. Here’s an example of how to create a prompt that ensures the generated summaries follow a consistent structure when analyzing sublease agreements: -```python Python nocheck hidelines={1..2} -import anthropic - +```python Python # Initialize the Anthropic client client = anthropic.Anthropic() def summarize_document( - text, details_to_extract, model="claude-opus-4-7", max_tokens=1000 + text, details_to_extract, model="claude-opus-4-8", max_tokens=1000 ): # Format the details to extract to be placed within the prompt's context details_to_extract_str = "\n".join(details_to_extract) @@ -207,23 +229,33 @@ Because the code outputs each section of the summary within tags, each section c ### Evaluate your prompt -Prompting often requires testing and optimization for it to be production ready. To determine the readiness of your solution, evaluate the quality of your summaries using a systematic process combining quantitative and qualitative methods. Creating a [strong empirical evaluation](/docs/en/test-and-evaluate/develop-tests#building-evals-and-test-cases) based on your defined success criteria allows you to optimize your prompts. Here are some metrics you may wish to include within your empirical evaluation: - -
-This measures the overlap between the generated summary and an expert-created reference summary. This metric primarily focuses on recall and is useful for evaluating content coverage. -
-
-While originally developed for machine translation, this metric can be adapted for summarization tasks. BLEU scores measure the precision of n-gram matches between the generated summary and reference summaries. A higher score indicates that the generated summary contains similar phrases and terminology to the reference summary. -
-
-This metric involves creating vector representations (embeddings) of both the generated and reference summaries. The similarity between these embeddings is then calculated, often using cosine similarity. Higher similarity scores indicate that the generated summary captures the semantic meaning and context of the reference summary, even if the exact wording differs. -
-
-This method involves using an LLM such as Claude to evaluate the quality of generated summaries against a scoring rubric. The rubric can be tailored to your specific needs, assessing key factors like accuracy, completeness, and coherence. For guidance on implementing LLM-based grading, view these [tips](/docs/en/test-and-evaluate/develop-tests#tips-for-llm-based-grading). -
-
-In addition to creating the reference summaries, legal experts can also evaluate the quality of the generated summaries. While this is expensive and time-consuming at scale, this is often done on a few summaries as a sanity check before deploying to production. -
+Prompting often requires testing and optimization for it to be production ready. To determine the readiness of your solution, evaluate the quality of your summaries using a systematic process combining quantitative and qualitative methods. Creating a [strong empirical evaluation](/docs/en/test-and-evaluate/develop-tests#building-evals-and-test-cases) based on your defined success criteria allows you to optimize your prompts. Here are some metrics you may want to include within your empirical evaluation: + + + + This measures the overlap between the generated summary and an expert-created reference summary. This metric primarily focuses on recall and is useful for evaluating content coverage. + + + + Although originally developed for machine translation, this metric can be adapted for summarization tasks. BLEU scores measure the precision of n-gram matches between the generated summary and reference summaries. A higher score indicates that the generated summary contains similar phrases and terminology to the reference summary. + + + + This metric involves creating vector representations (embeddings) of both the generated and reference summaries. The similarity between these embeddings is then calculated, often using cosine similarity. Higher similarity scores indicate that the generated summary captures the semantic meaning and context of the reference summary, even if the exact wording differs. + + + + This method involves using an LLM such as Claude to evaluate the quality of generated summaries against a scoring rubric. The rubric can be tailored to your specific needs, assessing key factors such as accuracy, completeness, and coherence. For implementation guidance, see + + [Tips for LLM-based grading](/docs/en/test-and-evaluate/develop-tests#tips-for-llm-based-grading) + + . + + + + In addition to creating the reference summaries, legal experts can also evaluate the quality of the generated summaries. Although this is expensive and time-consuming at scale, this is often done on a few summaries as a validation check before deploying to production. + + ### Deploy your prompt @@ -231,11 +263,11 @@ Here are some additional considerations to keep in mind as you deploy your solut 1. **Ensure no liability:** Understand the legal implications of errors in the summaries, which could lead to legal liability for your organization or clients. Provide disclaimers or legal notices clarifying that the summaries are generated by AI and should be reviewed by legal professionals. -2. **Handle diverse document types:** This guide discusses how to extract text from PDFs. In the real-world, documents may come in a variety of formats (PDFs, Word documents, text files, etc.). Ensure your data extraction pipeline can convert all of the file formats you expect to receive. +2. **Handle diverse document types:** This guide discusses how to extract text from PDFs. In the real-world, documents may come in a variety of formats (such as PDFs, Word documents, and text files). Ensure your data extraction pipeline can convert all of the file formats you expect to receive. 3. **Parallelize API calls to Claude:** Long documents with a large number of tokens may require up to a minute for Claude to generate a summary. For large document collections, you may want to send API calls to Claude in parallel so that the summaries can be completed in a reasonable timeframe. Refer to Anthropic’s [rate limits](/docs/en/api/rate-limits#rate-limits) to determine the maximum amount of API calls that can be performed in parallel. ---- +*** ## Improve performance @@ -243,13 +275,11 @@ In complex scenarios, it may be helpful to consider additional strategies to imp ### Perform meta-summarization to summarize long documents -Legal summarization often involves handling long documents or many related documents at once, such that you surpass Claude’s context window. You can use a chunking method known as meta-summarization in order to handle this use case. This technique involves breaking down documents into smaller, manageable chunks and then processing each chunk separately. You can then combine the summaries of each chunk to create a meta-summary of the entire document. +Legal summarization often involves handling long documents or many related documents at once, such that you surpass Claude’s context window. You can use a chunking method known as meta-summarization to handle this use case. This technique involves breaking down documents into smaller, manageable chunks and then processing each chunk separately. You can then combine the summaries of each chunk to create a meta-summary of the entire document. Here's an example of how to perform meta-summarization: -```python Python nocheck hidelines={1..2} -import anthropic - +```python Python # Initialize the Anthropic client client = anthropic.Anthropic() @@ -259,7 +289,7 @@ def chunk_text(text, chunk_size=20000): def summarize_long_document( - text, details_to_extract, model="claude-opus-4-7", max_tokens=1000 + text, details_to_extract, model="claude-opus-4-8", max_tokens=1000 ): # Format the details to extract to be placed within the prompt's context details_to_extract_str = "\n".join(details_to_extract) @@ -314,7 +344,7 @@ The `summarize_long_document` function builds upon the earlier `summarize_docume The code achieves this by applying the `summarize_document` function to each chunk of 20,000 characters within the original document. The individual summaries are then combined, and a final summary is created from these chunk summaries. -Note that the `summarize_long_document` function isn't strictly necessary for the example pdf, as the entire document fits within Claude's context window. However, it becomes essential for documents exceeding Claude's context window or when summarizing multiple related documents together. Regardless, this meta-summarization technique often captures additional important details in the final summary that were missed in the earlier single-summary approach. +Note that the `summarize_long_document` function isn't strictly necessary for the example PDF, as the entire document fits within Claude's context window. However, it becomes essential for documents exceeding Claude's context window or when summarizing multiple related documents together. Regardless, this meta-summarization technique often captures additional important details in the final summary that were missed in the earlier single-summary approach. ### Use summary indexed documents to explore a large collection of documents @@ -328,17 +358,24 @@ Another advanced technique to improve Claude's ability to generate summaries is 2. **Curate a dataset:** Once you've identified these issues, compile a dataset of these problematic examples. This dataset should include the original legal documents alongside your corrected summaries, ensuring that Claude learns the desired behavior. -3. **Perform fine-tuning:** Fine-tuning involves retraining the model on your curated dataset to adjust its weights and parameters. This retraining helps Claude better understand the specific requirements of your legal domain, improving its ability to summarize documents according to your standards. +3. **Perform fine-tuning:** Fine-tuning involves retraining the model on your curated dataset to adjust its weights and parameters. This retraining helps Claude better adapt to the specific requirements of your legal domain, improving its ability to summarize documents according to your standards. 4. **Iterative improvement:** Fine-tuning is not a one-time process. As Claude continues to generate summaries, you can iteratively add new examples where it has underperformed, further refining its capabilities. Over time, this continuous feedback loop will result in a model that is highly specialized for your legal summarization tasks. -Fine-tuning is currently only available via Amazon Bedrock. Additional details are available in the [AWS launch blog](https://aws.amazon.com/blogs/machine-learning/fine-tune-anthropics-claude-3-haiku-in-amazon-bedrock-to-boost-model-accuracy-and-quality/). + + Fine-tuning is currently only available through Amazon Bedrock. Additional details are available in the + + [AWS launch blog](https://aws.amazon.com/blogs/machine-learning/fine-tune-anthropics-claude-3-haiku-in-amazon-bedrock-to-boost-model-accuracy-and-quality/) + + . + View a fully implemented code-based example of how to use Claude to summarize contracts. + Explore the Citations cookbook recipe for guidance on how to ensure accuracy and explainability of information. - \ No newline at end of file + diff --git a/content/en/about-claude/use-case-guides/overview.md b/content/en/about-claude/use-case-guides/overview.md index 3339d44f8..8fa1ec6b3 100644 --- a/content/en/about-claude/use-case-guides/overview.md +++ b/content/en/about-claude/use-case-guides/overview.md @@ -1,5 +1,7 @@ # Guides to common use cases +Explore production guides for building common Claude use cases: ticket routing, customer support agents, content moderation, and legal summarization. + --- Claude is designed to excel in a variety of tasks. Explore these in-depth production guides to learn how to build common use cases with Claude. @@ -8,13 +10,16 @@ Claude is designed to excel in a variety of tasks. Explore these in-depth produc Best practices for using Claude to classify and route customer support tickets at scale. + Build intelligent, context-aware chatbots with Claude to enhance customer support interactions. + Techniques and best practices for using Claude to perform content filtering and general content moderation. + Summarize legal documents using Claude to extract key information and expedite research. - \ No newline at end of file + diff --git a/content/en/about-claude/use-case-guides/ticket-routing.md b/content/en/about-claude/use-case-guides/ticket-routing.md index cb1cd7c09..5c58028d7 100644 --- a/content/en/about-claude/use-case-guides/ticket-routing.md +++ b/content/en/about-claude/use-case-guides/ticket-routing.md @@ -4,54 +4,56 @@ This guide walks through how to harness Claude's advanced natural language under --- -## Define whether to use Claude for ticket routing +## Prerequisites + +* A Claude API key and the Python SDK installed +* Access to, and familiarity with, your existing support ticketing system +* A sample set of historical support tickets for testing -Here are some key indicators that you should use an LLM like Claude instead of traditional ML approaches for your classification task: +## Define whether to use Claude for ticket routing -
+Here are some key indicators that you should use an LLM like Claude instead of traditional ML approaches for your classification task: - Traditional ML processes require massive labeled datasets. Claude's pre-trained model can effectively classify tickets with just a few dozen labeled examples, significantly reducing data preparation time and costs. - -
-
+ + + Traditional ML processes require massive labeled datasets. Claude's pre-trained model can effectively classify tickets with just a few dozen labeled examples, significantly reducing data preparation time and costs. + - Once a traditional ML approach has been established, changing it is a laborious and data-intensive undertaking. On the other hand, as your product or customer needs evolve, Claude can easily adapt to changes in class definitions or new classes without extensive relabeling of training data. - -
-
+ + Once a traditional ML approach has been established, changing it is a laborious and data-intensive undertaking. On the other hand, as your product or customer needs evolve, Claude can easily adapt to changes in class definitions or new classes without extensive relabeling of training data. + - Traditional ML models often struggle with unstructured data and require extensive feature engineering. Claude's advanced language understanding allows for accurate classification based on content and context, rather than relying on strict ontological structures. - -
-
+ + Traditional ML models often struggle with unstructured data and require extensive feature engineering. Claude's advanced language understanding allows for accurate classification based on content and context, rather than relying on strict ontological structures. + - Traditional ML approaches often rely on bag-of-words models or simple pattern matching. Claude excels at understanding and applying underlying rules when classes are defined by conditions rather than examples. - -
-
+ + Traditional ML approaches often rely on bag-of-words models or simple pattern matching. Claude excels at understanding and applying underlying rules when classes are defined by conditions rather than examples. + - Many traditional ML models provide little insight into their decision-making process. Claude can provide human-readable explanations for its classification decisions, building trust in the automation system and facilitating easy adaptation if needed. - -
-
+ + Many traditional ML models provide little insight into their decision-making process. Claude can provide human-readable explanations for its classification decisions, building trust in the automation system and facilitating easy adaptation if needed. + - Traditional ML systems often struggle with outliers and ambiguous inputs, frequently misclassifying them or defaulting to a catch-all category. Claude's natural language processing capabilities allow it to better interpret context and nuance in support tickets, potentially reducing the number of misrouted or unclassified tickets that require manual intervention. - -
-
+ + Traditional ML systems often struggle with outliers and ambiguous inputs, frequently misclassifying them or defaulting to a catch-all category. Claude's natural language processing capabilities allow it to better interpret context and nuance in support tickets, potentially reducing the number of misrouted or unclassified tickets that require manual intervention. + - Traditional ML approaches typically require separate models or extensive translation processes for each supported language. Claude's multilingual capabilities allow it to classify tickets in various languages without the need for separate models or extensive translation processes, streamlining support for global customer bases. - -
+ + Traditional ML approaches typically require separate models or extensive translation processes for each supported language. Claude's multilingual capabilities allow it to classify tickets in various languages without the need for separate models or extensive translation processes, streamlining support for global customer bases. + + *** -## Build and deploy your LLM support workflow +## Build and deploy your LLM support workflow ### Understand your current support approach -Before diving into automation, it's crucial to understand your existing ticketing system. Start by investigating how your support team currently handles ticket routing. + +Before you automate, it's crucial to understand your existing ticketing system. Start by investigating how your support team currently handles ticket routing. Consider questions like: + * What criteria are used to determine what SLA/service offering is applied? * Is ticket routing used to determine which tier of support or product specialist a ticket goes to? * Are there any automated rules or workflows already in place? In what cases do they fail? @@ -61,101 +63,91 @@ Consider questions like: The more you know about how humans handle certain cases, the better you can work with Claude to do the task. ### Define user intent categories + A well-defined list of user intent categories is crucial for accurate support ticket classification with Claude. Claude’s ability to route tickets effectively within your system is directly proportional to how well-defined your system’s categories are. Here are some example user intent categories and subcategories. -
- - * Hardware problem - * Software bug - * Compatibility issue - * Performance problem - -
-
- - * Password reset - * Account access issues - * Billing inquiries - * Subscription changes - -
-
- - * Feature inquiries - * Product compatibility questions - * Pricing information - * Availability inquiries - -
-
- - * How-to questions - * Feature usage assistance - * Best practices advice - * Troubleshooting guidance - -
-
- - * Bug reports - * Feature requests - * General feedback or suggestions - * Complaints - -
-
- - * Order status inquiries - * Shipping information - * Returns and exchanges - * Order modifications - -
-
- - * Installation assistance - * Upgrade requests - * Maintenance scheduling - * Service cancellation - -
-
- - * Data privacy inquiries - * Suspicious activity reports - * Security feature assistance - -
-
- - * Regulatory compliance questions - * Terms of service inquiries - * Legal documentation requests - -
-
- - * Critical system failures - * Urgent security issues - * Time-sensitive problems - -
-
- - * Product training requests - * Documentation inquiries - * Webinar or workshop information - -
-
- - * Integration assistance - * API usage questions - * Third-party compatibility inquiries - -
+ + + * Hardware problem + * Software bug + * Compatibility issue + * Performance problem + + + + * Password reset + * Account access issues + * Billing inquiries + * Subscription changes + + + + * Feature inquiries + * Product compatibility questions + * Pricing information + * Availability inquiries + + + + * How-to questions + * Feature usage assistance + * Best practices advice + * Troubleshooting guidance + + + + * Bug reports + * Feature requests + * General feedback or suggestions + * Complaints + + + + * Order status inquiries + * Shipping information + * Returns and exchanges + * Order modifications + + + + * Installation assistance + * Upgrade requests + * Maintenance scheduling + * Service cancellation + + + + * Data privacy inquiries + * Suspicious activity reports + * Security feature assistance + + + + * Regulatory compliance questions + * Terms of service inquiries + * Legal documentation requests + + + + * Critical system failures + * Urgent security issues + * Time-sensitive problems + + + + * Product training requests + * Documentation inquiries + * Webinar or workshop information + + + + * Integration assistance + * API usage questions + * Third-party compatibility inquiries + + In addition to intent, ticket routing and prioritization may also be influenced by other factors such as urgency, customer type, SLAs, or language. Be sure to consider other routing criteria when building your automated routing system. @@ -165,114 +157,99 @@ Work with your support team to [define clear success criteria](/docs/en/test-and Here are some standard criteria and benchmarks when using LLMs for support ticket routing: -
- - This metric assesses how consistently Claude classifies similar tickets over time. It's crucial for maintaining routing reliability. Measure this by periodically testing the model with a set of standardized inputs and aiming for a consistency rate of 95% or higher. - -
-
+ + + This metric assesses how consistently Claude classifies similar tickets over time. It's crucial for maintaining routing reliability. Measure this by periodically testing the model with a set of standardized inputs and aiming for a consistency rate of 95% or higher. + - This measures how quickly Claude can adapt to new categories or changing ticket patterns. Test this by introducing new ticket types and measuring the time it takes for the model to achieve satisfactory accuracy (e.g., >90%) on these new categories. Aim for adaptation within 50-100 sample tickets. - -
-
+ + This measures how quickly Claude can adapt to new categories or changing ticket patterns. Test this by introducing new ticket types and measuring the time it takes for the model to achieve satisfactory accuracy (for example, >90%) on these new categories. Aim for adaptation within 50–100 sample tickets. + - This assesses Claude's ability to accurately route tickets in multiple languages. Measure the routing accuracy across different languages, aiming for no more than a 5-10% drop in accuracy for non-primary languages. - -
-
+ + This assesses Claude's ability to accurately route tickets in multiple languages. Measure the routing accuracy across different languages, aiming for no more than a 5–10% drop in accuracy for non-primary languages. + - This evaluates Claude's performance on unusual or complex tickets. Create a test set of edge cases and measure the routing accuracy, aiming for at least 80% accuracy on these challenging inputs. - -
-
+ + This evaluates Claude's performance on unusual or complex tickets. Create a test set of edge cases and measure the routing accuracy, aiming for at least 80% accuracy on these challenging inputs. + - This measures Claude's fairness in routing across different customer demographics. Regularly audit routing decisions for potential biases, aiming for consistent routing accuracy (within 2-3%) across all customer groups. - -
-
+ + This measures Claude's fairness in routing across different customer demographics. Regularly audit routing decisions for potential biases, aiming for consistent routing accuracy (within 2–3%) across all customer groups. + - In situations where minimizing token count is crucial, this criteria assesses how well Claude performs with minimal context. Measure routing accuracy with varying amounts of context provided, aiming for 90%+ accuracy with just the ticket title and a brief description. - -
-
+ + In situations where minimizing token count is crucial, this criteria assesses how well Claude performs with minimal context. Measure routing accuracy with varying amounts of context provided, aiming for 90%+ accuracy with just the ticket title and a brief description. + - This evaluates the quality and relevance of Claude's explanations for its routing decisions. Human raters can score explanations on a scale (e.g., 1-5), with the goal of achieving an average score of 4 or higher. - -
+ + This evaluates the quality and relevance of Claude's explanations for its routing decisions. Human raters can score explanations on a scale (for example, 1–5), with the goal of achieving an average score of 4 or higher. + + Here are some common success criteria that may be useful regardless of whether an LLM is used: -
- - Routing accuracy measures how often tickets are correctly assigned to the appropriate team or individual on the first try. This is typically measured as a percentage of correctly routed tickets out of total tickets. Industry benchmarks often aim for 90-95% accuracy, though this can vary based on the complexity of the support structure. - -
-
- - This metric tracks how quickly tickets are assigned after being submitted. Faster assignment times generally lead to quicker resolutions and improved customer satisfaction. Best-in-class systems often achieve average assignment times of under 5 minutes, with many aiming for near-instantaneous routing (which is possible with LLM implementations). - -
-
- - The rerouting rate indicates how often tickets need to be reassigned after initial routing. A lower rate suggests more accurate initial routing. Aim for a rerouting rate below 10%, with top-performing systems achieving rates as low as 5% or less. - -
-
- - This measures the percentage of tickets resolved during the first interaction with the customer. Higher rates indicate efficient routing and well-prepared support teams. Industry benchmarks typically range from 70-75%, with top performers achieving rates of 80% or higher. - -
-
- - Average handling time measures how long it takes to resolve a ticket from start to finish. Efficient routing can significantly reduce this time. Benchmarks vary widely by industry and complexity, but many organizations aim to keep average handling time under 24 hours for non-critical issues. - -
-
- - Often measured through post-interaction surveys, these scores reflect overall customer happiness with the support process. Effective routing contributes to higher satisfaction. Aim for CSAT scores of 90% or higher, with top performers often achieving 95%+ satisfaction rates. - -
-
- - This measures how often tickets need to be escalated to higher tiers of support. Lower escalation rates often indicate more accurate initial routing. Strive for an escalation rate below 20%, with best-in-class systems achieving rates of 10% or less. - -
-
- - This metric looks at how many tickets agents can handle effectively after implementing the routing solution. Improved routing should increase productivity. Measure this by tracking tickets resolved per agent per day or hour, aiming for a 10-20% improvement after implementing a new routing system. - -
-
- - This measures the percentage of potential tickets resolved through self-service options before entering the routing system. Higher rates indicate effective pre-routing triage. Aim for a deflection rate of 20-30%, with top performers achieving rates of 40% or higher. - -
-
- - This metric calculates the average cost to resolve each support ticket. Efficient routing should help reduce this cost over time. While benchmarks vary widely, many organizations aim to reduce cost per ticket by 10-15% after implementing an improved routing system. - -
+ + + Routing accuracy measures how often tickets are correctly assigned to the appropriate team or individual on the first try. This is typically measured as a percentage of correctly routed tickets out of total tickets. Industry benchmarks often aim for 90–95% accuracy, though this can vary based on the complexity of the support structure. + + + + This metric tracks how quickly tickets are assigned after being submitted. Faster assignment times generally lead to quicker resolutions and improved customer satisfaction. Best-in-class systems often achieve average assignment times of under 5 minutes, with many aiming for near-instantaneous routing (which is possible with LLM implementations). + + + + The rerouting rate indicates how often tickets need to be reassigned after initial routing. A lower rate suggests more accurate initial routing. Aim for a rerouting rate below 10%, with top-performing systems achieving rates as low as 5% or less. + + + + This measures the percentage of tickets resolved during the first interaction with the customer. Higher rates indicate efficient routing and well-prepared support teams. Industry benchmarks typically range from 70–75%, with top performers achieving rates of 80% or higher. + + + + Average handling time measures how long it takes to resolve a ticket from start to finish. Efficient routing can significantly reduce this time. Benchmarks vary widely by industry and complexity, but many organizations aim to keep average handling time under 24 hours for non-critical issues. + + + + Often measured through post-interaction surveys, these scores reflect overall customer happiness with the support process. Effective routing contributes to higher satisfaction. Aim for CSAT scores of 90% or higher, with top performers often achieving 95%+ satisfaction rates. + + + + This measures how often tickets need to be escalated to higher tiers of support. Lower escalation rates often indicate more accurate initial routing. Strive for an escalation rate below 20%, with best-in-class systems achieving rates of 10% or less. + + + + This metric looks at how many tickets agents can handle effectively after implementing the routing solution. Improved routing should increase productivity. Measure this by tracking tickets resolved per agent per day or hour, aiming for a 10–20% improvement after implementing a new routing system. + + + + This measures the percentage of potential tickets resolved through self-service options before entering the routing system. Higher rates indicate effective pre-routing triage. Aim for a deflection rate of 20–30%, with top performers achieving rates of 40% or higher. + + + + This metric calculates the average cost to resolve each support ticket. Efficient routing should help reduce this cost over time. While benchmarks vary widely, many organizations aim to reduce cost per ticket by 10–15% after implementing an improved routing system. + + ### Choose the right Claude model The choice of model depends on the trade-offs between cost, accuracy, and response time. -Many customers have found `claude-haiku-4-5-20251001` an ideal model for ticket routing, as it is the fastest and most cost-effective model in the Claude 4 family while still delivering excellent results. If your classification problem requires deep subject matter expertise or a large volume of intent categories complex reasoning, you may opt for the [larger Sonnet model](/docs/en/about-claude/models). +Many customers have found `claude-haiku-4-5-20251001` an ideal model for ticket routing, as it is the fastest and most cost-effective model in the Claude 4 family while still delivering excellent results. If your classification problem requires deep subject matter expertise or a large volume of intent categories, or complex reasoning, you may opt for the [larger Sonnet model](/docs/en/about-claude/models). ### Build a strong prompt Ticket routing is a type of classification task. Claude analyzes the content of a support ticket and classifies it into predefined categories based on the issue type, urgency, required expertise, or other relevant factors. -Let’s write a ticket classification prompt. Our initial prompt should contain the contents of the user request and return both the reasoning and the intent. +Write a ticket classification prompt. The initial prompt should contain the contents of the user request and return both the reasoning and the intent. -Try the [prompt generator](/docs/en/prompt-generator) on the [Claude Console](/login) to have Claude write a first draft for you. + Try the [prompt generator](/docs/en/prompt-generator) on the [Claude Console](/login) to have Claude write a first draft for you. Here's an example ticket routing classification prompt: -```python nocheck +```python def classify_support_request(ticket_contents): # Define the prompt for the classification task classification_prompt = f"""You will be acting as a customer support ticket classification system. Your task is to analyze customer support requests and output the appropriate classification intent for each request, along with your reasoning. @@ -329,23 +306,23 @@ def classify_support_request(ticket_contents): """ ``` -Let's break down the key components of this prompt: -* We use Python f-strings to create the prompt template, allowing the `ticket_contents` to be inserted into the `` tags. -* We give Claude a clearly defined role as a classification system that carefully analyzes the ticket content to determine the customer's core intent and needs. -* We instruct Claude on proper output formatting, in this case to provide its reasoning and analysis inside `` tags, followed by the appropriate classification label inside `` tags. -* We specify the valid intent categories: "Support, Feedback, Complaint", "Order Tracking", and "Refund/Exchange". -* We include a few examples (a.k.a. few-shot prompting) to illustrate how the output should be formatted, which improves accuracy and consistency. +Here are the key components of this prompt: -The reason we want to have Claude split its response into various XML tag sections is so that we can use regular expressions to separately extract the reasoning and intent from the output. This allows us to create targeted next steps in the ticket routing workflow, such as using only the intent to decide which person to route the ticket to. +* The prompt template is a Python f-string, allowing the `ticket_contents` to be inserted into the `` tags. +* The prompt gives Claude a clearly defined role as a classification system that carefully analyzes the ticket content to determine the customer's core intent and needs. +* The prompt instructs Claude on proper output formatting, in this case to provide its reasoning and analysis inside `` tags, followed by the appropriate classification label inside `` tags. +* The prompt specifies the valid intent categories: "Support, Feedback, Complaint", "Order Tracking", and "Refund/Exchange". +* The prompt includes a few examples (a.k.a. few-shot prompting) to illustrate how the output should be formatted, which improves accuracy and consistency. + +Having Claude split its response into separate XML tag sections lets you use regular expressions to extract the reasoning and intent from the output independently. This lets you create targeted next steps in the ticket routing workflow, such as using only the intent to decide which person to route the ticket to. ### Deploy your prompt It’s hard to know how well your prompt works without deploying it in a test production setting and [running evaluations](/docs/en/test-and-evaluate/develop-tests). -Let’s build the deployment structure. Start by defining the method signature for wrapping our call to Claude. We'll take the method we’ve already begun to write, which has `ticket_contents` as input, and now return a tuple of `reasoning` and `intent` as output. If you have an existing automation using traditional ML, you'll want to follow that method signature instead. +Build the deployment structure. Start by defining the method signature for wrapping the call to Claude. Extend the method you began writing earlier, which takes `ticket_contents` as input, so that it now returns a tuple of `reasoning` and `intent` as output. If you have an existing automation using traditional ML, you'll want to follow that method signature instead. -```python Python nocheck hidelines={1} -import anthropic +```python Python import re # Create an instance of the Claude API client @@ -385,12 +362,13 @@ def classify_support_request(ticket_contents): ``` This code: + * Creates a client instance using your API key. * Defines a `classify_support_request` function that takes a `ticket_contents` string. -* Sends the `ticket_contents` to Claude for classification using the `classification_prompt` +* Sends the `ticket_contents` to Claude for classification using the `classification_prompt`. * Returns the model's `reasoning` and `intent` extracted from the response. -Since we need to wait for the entire reasoning and intent text to be generated before parsing, we set `stream=False` (the default). +Because the entire reasoning and intent text must be generated before parsing, the example sets `stream=False` (the default). *** @@ -402,16 +380,16 @@ To run your evaluation, you need test cases to run it on. The rest of this guide ### Build an evaluation function -Our example evaluation for this guide measures Claude’s performance along three key metrics: +The example evaluation for this guide measures Claude’s performance along three key metrics: + * Accuracy * Cost per classification You may need to assess Claude on other axes depending on what factors that are important to you. -To assess this, we first have to modify the script we wrote and add a function to compare the predicted intent with the actual intent and calculate the percentage of correct predictions. We also have to add in cost calculation and time measurement functionality. +To assess this, first modify the script to add a function that compares the predicted intent with the actual intent and calculates the percentage of correct predictions. Then add cost calculation and time measurement functionality. -```python Python nocheck hidelines={1} -import anthropic +```python Python import re # Create an instance of the Claude API client @@ -454,13 +432,15 @@ def classify_support_request(request, actual_intent): return reasoning, intent, correct, usage ``` -Let’s break down the edits we’ve made: -* We added the `actual_intent` from our test cases into the `classify_support_request` method and set up a comparison to assess whether Claude’s intent classification matches our golden intent classification. -* We extracted usage statistics for the API call to calculate cost based on input and output tokens used +Here is a breakdown of the edits: + +* The `classify_support_request` method now takes the `actual_intent` from the test cases and compares it against Claude’s intent classification to assess whether they match. +* The method extracts usage statistics for the API call to calculate cost based on input and output tokens used. ### Run your evaluation -A proper evaluation requires clear thresholds and benchmarks to determine what is a good result. The script above gives us the runtime values for accuracy, response time, and cost per classification, but we still would need clearly established thresholds. For example: +A proper evaluation requires clear thresholds and benchmarks to determine what is a good result. The preceding script returns the runtime values for accuracy, response time, and cost per classification, but you still need clearly established thresholds. For example: + * **Accuracy:** 95% (out of 100 tests) * **Cost per classification:** 50% reduction on average (across 100 tests) from current routing method @@ -475,6 +455,7 @@ In complex scenarios, it may be helpful to consider additional strategies to imp ### Use a taxonomic hierarchy for cases with 20+ intent categories As the number of classes grows, the number of examples required also expands, potentially making the prompt unwieldy. As an alternative, you can consider implementing a hierarchical classification system using a mixture of classifiers. + 1. Organize your intents in a taxonomic tree structure. 2. Create a series of classifiers at every level of the tree, enabling a cascading routing approach. @@ -484,7 +465,7 @@ For example, you might have a top-level classifier that broadly categorizes tick * **Pros - greater nuance and accuracy:** You can create different prompts for each parent path, allowing for more targeted and context-specific classification. This can lead to improved accuracy and more nuanced handling of customer requests. -* **Cons - increased latency:** Be advised that multiple classifiers can lead to increased latency, and we recommend implementing this approach with our fastest model, Haiku. +* **Cons - increased latency:** Be advised that multiple classifiers can lead to increased latency, and Anthropic recommends implementing this approach with the fastest model, Haiku. ### Use vector databases and similarity search retrieval to handle highly variable tickets @@ -492,50 +473,53 @@ Despite providing examples being the most effective way to improve performance, In this scenario, you could employ a vector database to do similarity searches from a dataset of examples and retrieve the most relevant examples for a given query. -This approach, outlined in detail in our [classification recipe](https://platform.claude.com/cookbook/capabilities-classification-guide), has been shown to improve performance from 71% accuracy to 93% accuracy. +This approach, outlined in detail in the [classification recipe](https://platform.claude.com/cookbook/capabilities-classification-guide), has been shown to improve performance from 71% accuracy to 93% accuracy. ### Account specifically for expected edge cases -Here are some scenarios where Claude may misclassify tickets (there may be others that are unique to your situation). In these scenarios,consider providing explicit instructions or examples in the prompt of how Claude should handle the edge case: +Here are some scenarios where Claude may misclassify tickets (there may be others that are unique to your situation). In these scenarios, consider providing explicit instructions or examples in the prompt of how Claude should handle the edge case: -
+ + + Customers often express needs indirectly. For example, "I've been waiting for my package for over two weeks now" may be an indirect request for order status. - Customers often express needs indirectly. For example, "I've been waiting for my package for over two weeks now" may be an indirect request for order status. - * **Solution:** Provide Claude with some real customer examples of these kinds of requests, along with what the underlying intent is. You can get even better results if you include a classification rationale for particularly nuanced ticket intents, so that Claude can better generalize the logic to other tickets. - -
-
+ * **Solution:** Provide Claude with some real customer examples of these kinds of requests, along with what the underlying intent is. You can get even better results if you include a classification rationale for particularly nuanced ticket intents, so that Claude can better generalize the logic to other tickets. + - When customers express dissatisfaction, Claude may prioritize addressing the emotion over solving the underlying problem. - * **Solution:** Provide Claude with directions on when to prioritize customer sentiment or not. It can be something as simple as “Ignore all customer emotions. Focus only on analyzing the intent of the customer’s request and what information the customer might be asking for.” - -
-
+ + When customers express dissatisfaction, Claude may prioritize addressing the emotion over solving the underlying problem. - When customers present multiple issues in a single interaction, Claude may have difficulty identifying the primary concern. - * **Solution:** Clarify the prioritization of intents so thatClaude can better rank the extracted intents and identify the primary concern. - -
+ * **Solution:** Provide Claude with directions on when to prioritize customer sentiment or not. It can be something as simple as “Ignore all customer emotions. Focus only on analyzing the intent of the customer’s request and what information the customer might be asking for.” + + + + When customers present multiple issues in a single interaction, Claude may have difficulty identifying the primary concern. + + * **Solution:** Clarify the prioritization of intents so that Claude can better rank the extracted intents and identify the primary concern. + + *** ## Integrate Claude into your greater support workflow -Proper integration requires that you make some decisions regarding how your Claude-based ticket routing script fits into the architecture of your greater ticket routing system.There are two ways you could do this: -* **Push-based:** The support ticket system you’re using (e.g. Zendesk) triggers your code by sending a webhook event to your routing service, which then classifies the intent and routes it. - * This approach is more web-scalable, but needs you to expose a public endpoint. -* **Pull-Based:** Your code pulls for the latest tickets based on a given schedule and routes them at pull time. - * This approach is easier to implement but might make unnecessary calls to the support ticket system when the pull frequency is too high or might be overly slow when the pull frequency is too low. +Proper integration requires that you make some decisions regarding how your Claude-based ticket routing script fits into the architecture of your greater ticket routing system. There are two ways you could do this: + +* **Push-based:** The support ticket system you’re using (for example, Zendesk) triggers your code by sending a webhook event to your routing service, which then classifies the intent and routes it. + * This approach is more web-scalable, but needs you to expose a public endpoint. +* **Pull-based:** Your code pulls for the latest tickets based on a given schedule and routes them at pull time. + * This approach is easier to implement but might make unnecessary calls to the support ticket system when the pull frequency is too high or might be overly slow when the pull frequency is too low. For either of these approaches, you need to wrap your script in a service. The choice of approach depends on what APIs your support ticketing system provides. *** - - Visit our classification cookbook for more example code and detailed eval guidance. - - - Begin building and evaluating your workflow on the Claude Console. - - \ No newline at end of file + + Visit the classification cookbook for more example code and detailed eval guidance. + + + + Begin building and evaluating your workflow on the Claude Console. + + diff --git a/content/en/agents-and-tools/agent-skills/claude-api-skill.md b/content/en/agents-and-tools/agent-skills/claude-api-skill.md index 274fc0b5d..ea311cefb 100644 --- a/content/en/agents-and-tools/agent-skills/claude-api-skill.md +++ b/content/en/agents-and-tools/agent-skills/claude-api-skill.md @@ -6,14 +6,14 @@ An open-source Agent Skill that provides Claude with up-to-date API reference ma The `claude-api` skill is an open-source [Agent Skill](/docs/en/agents-and-tools/agent-skills/overview) that provides Claude with detailed, up-to-date reference material for building applications on two Anthropic surfaces: -- **Messages API** — the primary surface for single requests, streaming chat, tool use, batch processing, prompt caching, structured outputs, and custom agent loops. -- **Claude Managed Agents (beta):** An Anthropic-hosted surface for server-managed stateful agents with Anthropic-hosted tool execution, persistent agent configs, and per-session containers. +* **Messages API:** The primary surface for single requests, streaming chat, tool use, batch processing, prompt caching, structured outputs, and custom agent loops. +* **Claude Managed Agents (beta):** An Anthropic-hosted surface for server-managed stateful agents with Anthropic-hosted tool execution, persistent agent configs, and per-session sandboxes. -It covers 8 programming languages for the Messages API (Python, TypeScript, Java, Go, Ruby, C#, PHP, and cURL) and 7 languages for Managed Agents (Python, TypeScript, Java, Go, Ruby, PHP, and cURL — C# is not currently supported). +It covers eight programming languages for both the Messages API and Managed Agents: Python, TypeScript, C#, Go, Java, PHP, Ruby, and cURL. -The skill comes bundled with [Claude Code](https://code.claude.com/docs/en/overview) and is also available in the open-source [Anthropic skills repository](https://github.com/anthropics/skills/tree/main/skills/claude-api), where you can install it in any environment that supports Agent Skills. +The skill comes bundled with [Claude Code](https://code.claude.com/docs/en/overview) and is also available in the open-source [Anthropic skills repository](https://github.com/anthropics/skills), where you can install it in any environment that supports Agent Skills. -The skill uses [progressive disclosure](/docs/en/agents-and-tools/agent-skills/overview#three-types-of-skill-content-three-levels-of-loading) to keep context efficient: Claude loads only the documentation relevant to your project's language, surface (Messages API or Managed Agents), and the specific task at hand (tool use, streaming, batches, and so on), rather than loading everything at once. +The skill uses [progressive disclosure](/docs/en/agents-and-tools/agent-skills/overview#how-skills-work) to keep context efficient: Claude loads only the documentation relevant to your project's language, surface (Messages API or Managed Agents), and the specific task at hand (tool use, streaming, batches, and so on), rather than loading everything at once. ## What the skill provides @@ -21,21 +21,21 @@ When triggered, the skill equips Claude with: **For the Messages API:** -- **Language-specific SDK documentation:** Installation, quick start, common patterns, and error handling for your project's language -- **Tool use guidance:** Language-specific examples and [conceptual foundations](/docs/en/agents-and-tools/tool-use/overview) for function calling, including the beta tool runner where available -- **Streaming patterns:** Implementation details for building chat UIs and handling incremental display -- **Batch processing:** Offline batch processing at 50% cost -- **Prompt caching:** Prefix-stability design, breakpoint placement, and silent-invalidator audit -- **Model migration:** Step-by-step guidance for migrating to newer Claude models (including the breaking changes and behavior shifts on [Claude Opus 4.7](/docs/en/about-claude/models/migration-guide#migrating-to-claude-opus-4-7)) -- **Current model information:** Model IDs, context window sizes, and pricing -- **Common pitfalls:** Detailed guidance on avoiding frequent mistakes when integrating with the API +* **Language-specific SDK documentation:** Installation, quick start, common patterns, and error handling for your project's language +* **Tool use guidance:** Language-specific examples and [conceptual foundations](/docs/en/agents-and-tools/tool-use/overview) for function calling, including the beta tool runner where available +* **Streaming patterns:** Implementation details for building chat UIs and handling incremental display +* **Batch processing:** Offline batch processing at 50% cost +* **Prompt caching:** Prefix-stability design, breakpoint placement, and silent-invalidator audit +* **Model migration:** Step-by-step guidance for migrating to newer Claude models (including the breaking changes and behavior shifts on [Claude Opus 4.8](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-47)) +* **Current model information:** Model IDs, context window sizes, and pricing +* **Common pitfalls:** Detailed guidance on avoiding frequent mistakes when integrating with the API **For Managed Agents (beta):** -- **Onboarding flow:** An interview-driven walkthrough for setting up a new Managed Agent from scratch, available via the `/claude-api managed-agents-onboard` subcommand -- **Language-specific Managed Agents docs:** Creating persistent agents, starting sessions, streaming events, and handling tool confirmations for Python, TypeScript, Java, Go, Ruby, PHP, and cURL -- **Client patterns:** Lossless stream reconnect, `processed_at` queued/processed gate, interrupt handling, file-mount gotchas, and credential handling -- **Deployment constraints:** Managed Agents is available on the Claude API and [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) only (not on Amazon Bedrock, Vertex AI, or Microsoft Foundry). The skill routes other deployments to the Messages API and tool use instead. +* **Onboarding flow:** An interview-driven walkthrough for setting up a new Managed Agent from scratch, available through the `/claude-api managed-agents-onboard` subcommand +* **Language-specific Managed Agents docs:** Creating persistent agents, starting sessions, streaming events, and handling tool confirmations for Python, TypeScript, C#, Go, Java, PHP, Ruby, and cURL +* **Client patterns:** Lossless stream reconnect, `processed_at` queued/processed gate, interrupt handling, file-mount gotchas, and credential handling +* **Deployment constraints:** Managed Agents is available on the Claude API and [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) only (not on Amazon Bedrock, Google Cloud, or Microsoft Foundry). The skill routes other deployments to the Messages API and tool use instead. ## When the skill activates @@ -43,9 +43,9 @@ The skill activates in two ways: **Automatic activation** occurs when: -- Your code imports an Anthropic SDK (`anthropic` for Python, `@anthropic-ai/sdk` for TypeScript/JavaScript) -- You ask Claude to help build, debug, or optimize something with the Claude API, an Anthropic SDK, or Managed Agents -- You add, modify, or tune a Claude feature in a file (prompt caching, adaptive thinking, compaction, tool use, batch, files, citations, memory) or a model reference +* Your code imports an Anthropic SDK (`anthropic` for Python, `@anthropic-ai/sdk` for TypeScript/JavaScript) +* You ask Claude to help build, debug, or optimize something with the Claude API, an Anthropic SDK, or Managed Agents +* You add, modify, or tune a Claude feature in a file (prompt caching, adaptive thinking, compaction, tool use, batch, files, citations, memory) or a model reference **Manual invocation** by typing `/claude-api` (with optional subcommand or prose) in any environment where the skill is installed. @@ -56,14 +56,14 @@ The skill does not activate for general programming tasks, ML/data-science work, The skill detects your project's language automatically by examining project files (for example, `requirements.txt` for Python, `tsconfig.json` for TypeScript, `go.mod` for Go) and loads the appropriate documentation. | Language | Messages API SDK | Tool runner | Managed Agents | -|------------|------------------|-------------|----------------| +| ---------- | ---------------- | ----------- | -------------- | | Python | Yes | Yes (beta) | Yes (beta) | | TypeScript | Yes | Yes (beta) | Yes (beta) | -| Java | Yes | No | Yes (beta) | -| Go | Yes | No | Yes (beta) | +| C# | Yes | Yes (beta) | Yes (beta) | +| Go | Yes | Yes (beta) | Yes (beta) | +| Java | Yes | Yes (beta) | Yes (beta) | +| PHP | Yes | Yes (beta) | Yes (beta) | | Ruby | Yes | Yes (beta) | Yes (beta) | -| C# | Yes | No | No | -| PHP | Yes | No | Yes (beta) | | cURL | Yes | N/A | Yes (beta) | If your project uses multiple languages, Claude asks which one applies. For unsupported languages (Rust, Swift, C++), the skill provides cURL/raw HTTP examples. @@ -76,7 +76,7 @@ The skill ships with [Claude Code](https://code.claude.com/docs/en/overview) and You can also invoke it directly: -```text +```text wrap /claude-api ``` @@ -90,53 +90,55 @@ The skill source is available in the [Anthropic skills repository](https://githu npx skills add https://github.com/anthropics/skills --skill claude-api ``` -Or install it as a Claude Code [plugin](https://code.claude.com/docs/en/plugins): +Or install it as a [Claude Code plugin](https://code.claude.com/docs/en/plugins): -```bash +```text wrap /plugin marketplace add anthropics/skills /plugin install claude-api@anthropic-agent-skills ``` ## Migrating to a newer Claude model -The Claude API skill can perform Claude model migrations across a codebase. Invoke it directly with `/claude-api migrate`: +The Claude API skill can perform Claude model migrations across a code base. Invoke it directly with `/claude-api migrate`: -```text -/claude-api migrate this project to claude-opus-4-7 +```text wrap +/claude-api migrate this project to claude-opus-4-8 ``` You can also pass a specific scope up front to skip the scope-confirmation question: -```text -/claude-api migrate everything under src/ to claude-opus-4-7 -/claude-api migrate apps/api.py and apps/worker.py to claude-opus-4-7 +```text wrap +/claude-api migrate everything under src/ to claude-opus-4-8 +/claude-api migrate apps/api.py and apps/worker.py to claude-opus-4-8 ``` -When the scope is ambiguous (for example, a bare `/claude-api migrate to claude-opus-4-7`), the skill asks you to choose between the entire working directory, a specific subdirectory, or an explicit file list before editing any files. This applies to both Messages API and Managed Agents callers. +When the scope is ambiguous (for example, a bare `/claude-api migrate to claude-opus-4-8`), the skill asks you to choose between the entire working directory, a specific subdirectory, or an explicit file list before editing any files. This applies to both Messages API and Managed Agents callers. The skill handles: -- **Model ID swaps**, including typed SDK constants (`Model.CLAUDE_OPUS_4_6` → `Model.CLAUDE_OPUS_4_7`) across all supported languages, and classifies each file as a caller, a model definer, or an opaque string reference before editing -- **Breaking parameter changes**, such as removing `temperature`, `top_p`, and `top_k` for Claude Opus 4.7, and converting `thinking: {type: "enabled", budget_tokens: N}` to `thinking: {type: "adaptive"}` -- **Prefill replacement**, converting assistant-message prefill patterns to [structured outputs](/docs/en/build-with-claude/structured-outputs) where applicable -- **Beta header cleanup**, removing headers that are GA on the target model (for example, `effort-2025-11-24`, `fine-grained-tool-streaming-2025-05-14`, `interleaved-thinking-2025-05-14`) and switching back from `client.beta.messages.create` to `client.messages.create` -- **Effort calibration**, recommending an `output_config.effort` starting point for the target model (for example, `xhigh` for coding and agentic use cases on Claude Opus 4.7) -- **Prompt-behavior tuning**, flagging length-control, tool-triggering, subagent, and instruction-following prompts that may behave differently on the target model -- **Silent default handling**, opting back into thinking summarization (`thinking.display: "summarized"`) when reasoning is surfaced to users on Claude Opus 4.7 +* **Model ID swaps**, including typed SDK constants (`Model.CLAUDE_OPUS_4_7` → `Model.CLAUDE_OPUS_4_8`) across all supported languages, and classifies each file as a caller, a model definer, or an opaque string reference before editing +* **Cloud platform detection**, preserving platform-specific model ID formats (for example, the `anthropic.` prefix on Amazon Bedrock) and skipping changes for features that are unavailable on partner-operated platforms +* **Breaking parameter changes**, such as removing `temperature`, `top_p`, and `top_k` for Claude Opus 4.8 and Claude Opus 4.7, and converting `thinking: {type: "enabled", budget_tokens: N}` to `thinking: {type: "adaptive"}` +* **Prefill replacement**, converting assistant-message prefill patterns to [structured outputs](/docs/en/build-with-claude/structured-outputs) where applicable +* **Beta header cleanup**, removing headers that are GA on the target model (for example, `effort-2025-11-24`, `fine-grained-tool-streaming-2025-05-14`, `interleaved-thinking-2025-05-14`) and switching back from `client.beta.messages.create` to `client.messages.create` +* **Effort calibration**, recommending an `output_config.effort` starting point for the target model (for example, `xhigh` for coding and agentic use cases on Claude Opus 4.8 and Claude Opus 4.7) +* **Prompt-behavior tuning**, flagging length-control, tool-triggering, subagent, and instruction-following prompts that may behave differently on the target model +* **Silent default handling**, opting back into thinking summarization (`thinking.display: "summarized"`) when reasoning is surfaced to users on Claude Opus 4.8 and Claude Opus 4.7 +* **Refusal fallback configuration**, adding `stop_reason: "refusal"` handling before reading response content and setting up a [fallback retry path](/docs/en/build-with-claude/refusals-and-fallback) when the target is Claude Fable 5 (the server-side `fallbacks` parameter, the SDK refusal-fallback middleware, or a fallback-credit retry), and updating fallback code written against earlier preview shapes As it edits, the skill explains each change and its motivation inline. On completion, it produces a checklist of items that require manual verification (typically integration tests, length-control prompt tuning, and cost/rate-limit re-baselining). -For the full list of model-specific changes the skill applies, see [Migrating to Claude Opus 4.7](/docs/en/about-claude/models/migration-guide#migrating-to-claude-opus-4-7). +For the full list of model-specific changes the skill applies, see [Migrating to Claude Opus 4.8 from Claude Opus 4.7](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-47). ## Setting up a Managed Agent To scaffold a new Managed Agent from scratch, invoke the `managed-agents-onboard` subcommand: -```text +```text wrap /claude-api managed-agents-onboard ``` -The skill runs an interview that walks you through the Managed Agents mental model (Agent configs versus Sessions), templates an agent config, configures environments and tools, sets up the session loop, and emits runnable code for your language. The skill also covers the mandatory **Agent (once) → Session (every run)** flow — `model`, `system`, and `tools` live on the agent, never on the session, and agents should be created once and referenced by ID. +The skill runs an interview that walks you through the Managed Agents mental model (Agent configs versus Sessions), templates an agent config, configures environments and tools, sets up the session loop, and emits runnable code for your language. The skill also covers the mandatory **Agent (once) → Session (every run)** flow: `model`, `system`, and `tools` live on the agent, never on the session, and agents should be created once and referenced by ID. Managed Agents requires the `managed-agents-2026-04-01` beta header, which the SDK sets automatically for all `client.beta.agents.*`, `client.beta.environments.*`, `client.beta.sessions.*`, and `client.beta.vaults.*` calls. @@ -145,17 +147,20 @@ Managed Agents requires the `managed-agents-2026-04-01` beta header, which the S Here are examples of tasks the skill helps Claude handle: **Building a chat application:** -```text + +```text wrap Build a streaming chat UI with the Claude API in TypeScript ``` **Migrating an existing project:** -```text -/claude-api migrate this codebase to claude-opus-4-7 and re-tune effort + +```text wrap +/claude-api migrate this codebase to claude-opus-4-8 and re-tune effort ``` **Onboarding a new Managed Agent:** -```text + +```text wrap /claude-api managed-agents-onboard ``` @@ -164,25 +169,15 @@ In each case, the skill loads the relevant language-specific documentation and g ## Next steps - + Learn about how Agent Skills work and the progressive disclosure model - + + Browse the official Anthropic SDKs for all supported languages - + + Explore the public Anthropic skills repository on GitHub - \ No newline at end of file + diff --git a/content/en/api/admin.md b/content/en/api/admin.md index b2b540cdc..9f825ac4d 100644 --- a/content/en/api/admin.md +++ b/content/en/api/admin.md @@ -4297,11 +4297,11 @@ curl https://api.anthropic.com/v1/organizations/workspaces/$WORKSPACE_ID/service # API Keys -## Get API Key +## Retrieve API Key (Admin API) **get** `/v1/organizations/api_keys/{api_key_id}` -Get API Key +Retrieve information about a single API key in your organization, looked up by its ID. This Admin API endpoint requires an Admin API key, is intended for programmatic key management, and never returns the key's secret value. To view or create your own API keys, go to [API keys](https://platform.claude.com/settings/keys) in the Claude Console. ### Path Parameters diff --git a/content/en/api/admin/api_keys.md b/content/en/api/admin/api_keys.md index 7cb652c4f..f00df7ba5 100644 --- a/content/en/api/admin/api_keys.md +++ b/content/en/api/admin/api_keys.md @@ -1,10 +1,10 @@ # API Keys -## Get API Key +## Retrieve API Key (Admin API) **get** `/v1/organizations/api_keys/{api_key_id}` -Get API Key +Retrieve information about a single API key in your organization, looked up by its ID. This Admin API endpoint requires an Admin API key, is intended for programmatic key management, and never returns the key's secret value. To view or create your own API keys, go to [API keys](https://platform.claude.com/settings/keys) in the Claude Console. ### Path Parameters diff --git a/content/en/api/admin/api_keys/retrieve.md b/content/en/api/admin/api_keys/retrieve.md index 21180766b..6ee4720b7 100644 --- a/content/en/api/admin/api_keys/retrieve.md +++ b/content/en/api/admin/api_keys/retrieve.md @@ -1,8 +1,8 @@ -## Get API Key +## Retrieve API Key (Admin API) **get** `/v1/organizations/api_keys/{api_key_id}` -Get API Key +Retrieve information about a single API key in your organization, looked up by its ID. This Admin API endpoint requires an Admin API key, is intended for programmatic key management, and never returns the key's secret value. To view or create your own API keys, go to [API keys](https://platform.claude.com/settings/keys) in the Claude Console. ### Path Parameters diff --git a/content/en/api/beta-headers.md b/content/en/api/beta-headers.md index b63b9d18f..f6a51d783 100644 --- a/content/en/api/beta-headers.md +++ b/content/en/api/beta-headers.md @@ -1,15 +1,13 @@ # Beta headers -Documentation for using beta headers with the Claude API +Access experimental features before general availability with the `anthropic-beta` header or the SDKs' `betas` parameter. --- Beta headers allow you to access experimental features and new model capabilities before they become part of the standard API. -These features are subject to change and may be modified or removed in future releases. - -Beta headers are often used in conjunction with the [beta namespace in the client SDKs](/docs/en/api/client-sdks#beta-namespace-in-client-sdks) + Each [client SDK](/docs/en/cli-sdks-libraries/overview) exposes a `beta` namespace for calling the API with beta features enabled. ## How to use beta headers @@ -18,72 +16,148 @@ To access beta features, include the `anthropic-beta` header in your API request ```http POST /v1/messages -Content-Type: application/json -X-API-Key: YOUR_API_KEY +x-api-key: YOUR_API_KEY +anthropic-version: 2023-06-01 anthropic-beta: BETA_FEATURE_NAME +content-type: application/json ``` -When using the SDK, you can specify beta headers in the request options: +Each feature's documentation states the exact beta name to send. The [API overview](/docs/en/api/overview) lists the APIs currently in beta. + +The following examples show the same request with cURL, the `ant` CLI, and the SDKs. The SDKs take beta names in the `betas` parameter and send the `anthropic-beta` header for you: + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "anthropic-beta: files-api-2025-04-14" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-opus-4-8", + "max_tokens": 1024, + "messages": [ + {"role": "user", "content": "Hello, Claude"} + ] + }' + ``` + + ```bash CLI + ant beta:messages create \ + --beta files-api-2025-04-14 \ + --model claude-opus-4-8 \ + --max-tokens 1024 \ + --message '{role: user, content: "Hello, Claude"}' + ``` + + ```python Python + client = Anthropic() + + response = client.beta.messages.create( + model="claude-opus-4-8", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude"}], + betas=["files-api-2025-04-14"], + ) + + print(response.content) + ``` + + ```typescript TypeScript + const client = new Anthropic(); + + const msg = await client.beta.messages.create({ + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + betas: ["files-api-2025-04-14"] + }); + + console.log(msg.content); + ``` + + ```csharp C# + var client = new AnthropicClient(); + + var message = await client.Beta.Messages.Create( + new MessageCreateParams + { + Model = "claude-opus-4-8", + MaxTokens = 1024, + Messages = [new() { Role = Role.User, Content = "Hello, Claude" }], + Betas = ["files-api-2025-04-14"], + } + ); + + Console.WriteLine(string.Join("\n", message.Content)); + ``` + + ```go Go + client := anthropic.NewClient() + + message, err := client.Beta.Messages.New(context.TODO(), anthropic.BetaMessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 1024, + Messages: []anthropic.BetaMessageParam{ + anthropic.NewBetaUserMessage(anthropic.NewBetaTextBlock("Hello, Claude")), + }, + Betas: []anthropic.AnthropicBeta{anthropic.AnthropicBetaFilesAPI2025_04_14}, + }) + if err != nil { + panic(err) + } -```bash cURL -curl https://api.anthropic.com/v1/messages \ - -H "x-api-key: $ANTHROPIC_API_KEY" \ - -H "anthropic-version: 2023-06-01" \ - -H "anthropic-beta: files-api-2025-04-14" \ - -H "content-type: application/json" \ - -d '{ - "model": "claude-opus-4-7", - "max_tokens": 1024, - "messages": [ - {"role": "user", "content": "Hello, Claude"} - ] - }' -``` + fmt.Printf("%+v\n", message.Content) + ``` -```bash CLI -ant beta:messages create \ - --beta files-api-2025-04-14 \ - --model claude-opus-4-7 \ - --max-tokens 1024 \ - --message '{role: user, content: "Hello, Claude"}' -``` + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); -```python Python hidelines={1..2} -from anthropic import Anthropic + MessageCreateParams params = MessageCreateParams.builder() + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(1024) + .addUserMessage("Hello, Claude") + .addBeta(AnthropicBeta.FILES_API_2025_04_14) + .build(); -client = Anthropic() + BetaMessage message = client.beta().messages().create(params); + System.out.println(message.content()); + ``` -response = client.beta.messages.create( - model="claude-opus-4-7", - max_tokens=1024, - messages=[{"role": "user", "content": "Hello, Claude"}], - betas=["files-api-2025-04-14"], -) -``` + ```php PHP + $client = new Client(); -```typescript TypeScript hidelines={1..2} -import Anthropic from "@anthropic-ai/sdk"; + $message = $client->beta->messages->create( + maxTokens: 1024, + messages: [['role' => 'user', 'content' => 'Hello, Claude']], + model: 'claude-opus-4-8', + betas: ['files-api-2025-04-14'], + ); -const anthropic = new Anthropic(); + echo $message; + ``` -const msg = await anthropic.beta.messages.create({ - model: "claude-opus-4-7", - max_tokens: 1024, - messages: [{ role: "user", content: "Hello, Claude" }], - betas: ["files-api-2025-04-14"] -}); -``` + ```ruby Ruby + client = Anthropic::Client.new + message = client.beta.messages.create( + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [{role: "user", content: "Hello, Claude"}], + betas: ["files-api-2025-04-14"] + ) + + puts(message.content) + ``` -Beta features are experimental and may: -- Have breaking changes with notice -- Be deprecated or removed -- Have different rate limits or pricing -- Not be available in all regions + Beta features are experimental and may: + + * Have breaking changes with notice + * Be deprecated or removed + * Have different rate limits or pricing + * Not be available in all regions ### Multiple beta features @@ -94,40 +168,53 @@ To use multiple beta features in a single request, include all feature names in anthropic-beta: feature1,feature2,feature3 ``` +When using an SDK, list each feature in the `betas` parameter (for example, `betas=["feature1", "feature2"]`). With the CLI, pass a single `--beta` flag with the feature names separated by commas (for example, `--beta feature1,feature2`). Avoid repeating the flag: currently only the first flag's value takes effect. + ### Endpoint-specific headers -Some beta features are scoped to specific endpoints rather than individual request parameters and require a feature-specific beta header on every request: +Some beta APIs are scoped to specific endpoints and require a feature-specific beta header on every request: -| Endpoints | Beta header | -| --- | --- | +| Endpoints | Beta header | +| ------------------------------------------------ | --------------------------- | | `/v1/agents`, `/v1/sessions`, `/v1/environments` | `managed-agents-2026-04-01` | +| `/v1/tunnels` | `mcp-tunnels-2026-06-22` | +| `/v1/memory_stores` and sub-resources | `agent-memory-2026-07-22` | -See the [Managed Agents overview](/docs/en/managed-agents/overview) for details. +The SDKs' `beta` namespaces add these headers automatically. Add them yourself only when making raw HTTP requests. See the [Managed Agents overview](/docs/en/managed-agents/overview), [Using agent memory](/docs/en/managed-agents/memory), and the [MCP tunnels reference](/docs/en/agents-and-tools/mcp-tunnels/reference#tunnels-api) for details. + +Endpoint-specific headers that apply to the same endpoint aren't always combinable. On memory store endpoints, `agent-memory-2026-07-22` replaces `managed-agents-2026-04-01`: sending both on the same request returns a `400` error. The client SDKs send the correct header for each endpoint automatically. ### Version naming conventions -Beta feature names typically follow the pattern: `feature-name-YYYY-MM-DD`, where the date indicates when the beta version was released. Always use the exact beta feature name as documented. +Beta feature names typically follow the pattern `feature-name-YYYY-MM-DD`, where the date indicates when the beta was released. Always use the exact beta feature name as documented. ## Error handling -If you use an invalid or unavailable beta header, you'll receive an error response: +If you use an invalid beta name, or a beta your organization doesn't have access to, you'll receive a `400` error response: ```json Output { "type": "error", "error": { "type": "invalid_request_error", - "message": "Unsupported beta header: invalid-beta-name" - } + "message": "Unexpected value(s) `invalid-beta-name` for the `anthropic-beta` header. Please consult our documentation at platform.claude.com/docs or try again without the header." + }, + "request_id": "req_011CcnGfC9fELffo2EALu4Wd" } ``` ## Getting help -For questions about beta features: +For updates to beta features, see the [release notes](/docs/en/release-notes/overview). For help with production issues, contact [support](https://support.claude.com/). + +## Next steps -1. Check the documentation for the specific feature -2. Review the [API changelog](/docs/en/api/versioning) for updates -3. Contact support for assistance with production usage + + + Understand the HTTP status codes, error response shape, and request IDs the Claude API returns, and handle errors with the SDKs' typed exceptions. + -Remember that beta features are provided "as-is" and may not have the same SLA guarantees as stable API features. \ No newline at end of file + + Explore the Claude API's features, including the APIs currently in beta. + + diff --git a/content/en/api/claude-code/routines-fire.md b/content/en/api/claude-code/routines-fire.md index c1b51450a..08b6c74d1 100644 --- a/content/en/api/claude-code/routines-fire.md +++ b/content/en/api/claude-code/routines-fire.md @@ -1,38 +1,38 @@ -# Trigger a routine via API +# Trigger a routine through the API Start a Claude Code routine session on demand by sending an authenticated POST request. --- -This is an experimental API. Request and response shapes, rate limits, and token semantics may change. Breaking changes ship behind new dated beta header versions, and the two previous header versions continue to work so that callers have time to migrate. + This is an experimental API. Request and response shapes, rate limits, and token semantics might change. Breaking changes ship behind new dated beta header versions, and the two previous header versions continue to work so that callers have time to migrate. [Claude Code](https://code.claude.com/docs) is Anthropic's agentic coding tool. [Claude Code on the web](https://code.claude.com/docs/en/claude-code-on-the-web) runs Claude Code sessions on Anthropic-managed cloud infrastructure at claude.ai/code, and a [routine](https://code.claude.com/docs/en/routines) is a saved configuration there: a prompt, one or more repositories, and connectors, packaged so it can run unattended on a schedule, in response to GitHub events, or when called over HTTP. This endpoint is the HTTP entry point. POSTing to it starts a new run of an existing routine and returns the resulting session ID and URL. Typical callers are alerting systems, CI pipelines, and internal tools that need to start a Claude Code session programmatically. -Calling this endpoint requires a claude.ai account on a Pro, Max, Team, or Enterprise plan with [Claude Code on the web](https://code.claude.com/docs/en/claude-code-on-the-web) enabled. Authenticate with a per-routine bearer token created in the Claude Code web UI rather than an Anthropic API key. +Calling this endpoint requires a claude.ai account on a Pro, Max, Team, or Enterprise plan with [Claude Code on the web](https://code.claude.com/docs/en/claude-code-on-the-web) enabled. Authenticate with a per-routine bearer token created in the Claude Code web UI rather than a Claude API key. ## Differences from the Claude Platform The routine fire endpoint belongs to the Claude Code product surface, which differs from the Claude Platform APIs and SDKs in a few ways: -| Aspect | This endpoint | Other Anthropic APIs | -| :--- | :--- | :--- | -| Authentication | `Authorization: Bearer` with a per-routine token (`sk-ant-oat01-...`) created at [claude.ai/code/routines](https://claude.ai/code/routines) | `x-api-key` with an Anthropic API key from Claude Console | -| Token scope | One routine only; no read access | Workspace-level | -| SDK support | None | Available in all [client SDKs](/docs/en/api/client-sdks) | -| Billing | Claude Code subscription usage on claude.ai | Claude Platform usage | -| Path namespace | `/v1/claude_code/...` | `/v1/...` | -| Stability | Experimental; requires `anthropic-beta: experimental-cc-routine-2026-04-01` | Stable or standard beta | +| Aspect | This endpoint | Claude Platform APIs | +| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------- | +| Authentication | `Authorization: Bearer` with a per-routine token (`sk-ant-oat01-...`) created at [claude.ai/code/routines](https://claude.ai/code/routines) | `x-api-key` with a Claude API key from Claude Console | +| Token scope | One routine only; no read access | Workspace-level | +| SDK support | None | Available in all [client SDKs](/docs/en/cli-sdks-libraries/overview) | +| Billing | Claude Code subscription usage on claude.ai | Claude Platform usage | +| Path namespace | `/v1/claude_code/...` | `/v1/...` | +| Stability | Experimental; requires `anthropic-beta: experimental-cc-routine-2026-04-01` | Stable or standard beta | ## Before you begin To call this endpoint, you need: -1. A routine created at [claude.ai/code/routines](https://claude.ai/code/routines) -2. A bearer token generated for that routine: open the routine for editing, click **Add another trigger** under **Select a trigger**, choose **API**, then click **Generate token** in the modal. The token is shown once and cannot be retrieved later. +1. A routine created at [claude.ai/code/routines](https://claude.ai/code/routines). +2. A bearer token generated for that routine: open the routine for editing, click **Add another trigger** under **Select a trigger**, choose **API**, then click **Generate token** in the modal window. The token is shown once and cannot be retrieved later. See [Add an API trigger](https://code.claude.com/docs/en/routines#add-an-api-trigger) in the Claude Code documentation for the full setup walkthrough. @@ -44,7 +44,7 @@ POST https://api.anthropic.com/v1/claude_code/routines/{routine_id}/fire Every request must include the `anthropic-beta: experimental-cc-routine-2026-04-01` header. Requests without it return `400 invalid_request_error`. -The Claude Code web UI provides the full URL alongside the token when you add an API trigger, so most integrations store both as secrets and call the endpoint directly. The examples below show a shell call and a GitHub Actions step that triggers the routine on CI failure. +The Claude Code web UI provides the full URL alongside the token when you add an API trigger, so most integrations store both as secrets and call the endpoint directly. The following examples show a shell call and a GitHub Actions step that triggers the routine on CI failure. ```bash cURL curl -X POST https://api.anthropic.com/v1/claude_code/routines/$ROUTINE_ID/fire \ @@ -73,24 +73,24 @@ The request returns once the session is created. It does not stream session outp ### Headers -| Name | Required | Description | -| :--- | :--- | :--- | -| `Authorization` | Yes | `Bearer `. The per-routine token created in the Claude Code web UI, prefixed `sk-ant-oat01-`. | -| `anthropic-beta` | Yes | Must include `experimental-cc-routine-2026-04-01`. | -| `anthropic-version` | Yes | The [API version](/docs/en/api/versioning), for example `2023-06-01`. | -| `Content-Type` | When body is present | `application/json`. | +| Name | Required | Description | +| ------------------- | -------------------- | ---------------------------------------------------------------------------------------------------- | +| `Authorization` | Yes | `Bearer `. The per-routine token created in the Claude Code web UI, prefixed `sk-ant-oat01-`. | +| `anthropic-beta` | Yes | Must include `experimental-cc-routine-2026-04-01`. | +| `anthropic-version` | Yes | The [API version](/docs/en/api/versioning), for example `2023-06-01`. | +| `Content-Type` | When body is present | `application/json`. | ### Path parameters -| Name | Type | Description | -| :--- | :--- | :--- | -| `routine_id` | string | The routine's identifier. Despite the parameter name, the value is prefixed `trig_` rather than `routine_`. Included in the URL the modal shows when you add an API trigger. | +| Name | Type | Description | +| ------------ | ------ | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `routine_id` | string | The routine's identifier. Despite the parameter name, the value is prefixed `trig_` rather than `routine_`. Included in the URL the modal window shows when you add an API trigger. | ### Request body -| Field | Type | Required | Description | -| :--- | :--- | :--- | :--- | -| `text` | string | No | Initial context for this run, such as an alert body, a failing log line, or a git diff. The value is freeform text and is not parsed; if you send JSON or another structured payload, the routine receives it as a literal string. Passed to the routine alongside its saved prompt. Maximum 65,536 characters. | +| Field | Type | Required | Description | +| ------ | ------ | -------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `text` | string | No | Initial context for this run, such as an alert body, a failing log line, or a git diff. The value is freeform text and is not parsed; if you send JSON or another structured payload, the routine receives it as a literal string. Passed to the routine alongside its saved prompt. Maximum 65,536 characters. | The body is optional. Unknown fields in the body are ignored. @@ -106,10 +106,10 @@ A successful request returns `200 OK` with the new session details: } ``` -| Field | Type | Description | -| :--- | :--- | :--- | -| `type` | string | Always `routine_fire`. | -| `claude_code_session_id` | string | The ID of the Claude Code session created for this run. | +| Field | Type | Description | +| ------------------------- | ------ | ------------------------------------------------------------------------------------------------------------------------ | +| `type` | string | Always `routine_fire`. | +| `claude_code_session_id` | string | The ID of the Claude Code session created for this run. | | `claude_code_session_url` | string | A link to the session on claude.ai. Open it in a browser to watch the run, review changes, or continue the conversation. | ### Errors @@ -126,15 +126,15 @@ Errors use the standard Anthropic [error envelope](/docs/en/api/errors): } ``` -| HTTP status | Error type | Cause | -| :--- | :--- | :--- | -| 400 | `invalid_request_error` | Missing or invalid `anthropic-beta` header, `text` exceeds 65,536 characters, or the routine is [paused](https://code.claude.com/docs/en/routines#edit-and-control-routines). | -| 401 | `authentication_error` | No bearer token in the `Authorization` header, or the token does not match this routine. | -| 403 | `permission_error` | The account or organization does not have access to this endpoint. | -| 404 | `not_found_error` | The routine does not exist. | -| 429 | `rate_limit_error` | The account's routine run limit or usage limit has been reached. The response includes a `Retry-After` header indicating when the window resets. | -| 500 | `api_error` | An unexpected server error. | -| 503 | `overloaded_error` | The service is temporarily overloaded. Retry after a short delay. The Claude Platform returns 529 for this error type; this endpoint returns 503. | +| HTTP status | Error type | Cause | +| ----------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 400 | `invalid_request_error` | Missing or invalid `anthropic-beta` header, `text` exceeds 65,536 characters, or the routine is paused (see [Edit and control routines](https://code.claude.com/docs/en/routines#edit-and-control-routines)). | +| 401 | `authentication_error` | No bearer token in the `Authorization` header, or the token does not match this routine. | +| 403 | `permission_error` | The account or organization does not have access to this endpoint. | +| 404 | `not_found_error` | The routine does not exist. | +| 429 | `rate_limit_error` | The account's routine run limit or usage limit has been reached. The response includes a `Retry-After` header indicating when the window resets. | +| 500 | `api_error` | An unexpected server error. Retry with exponential backoff; if the error persists, contact support with the request ID. | +| 503 | `overloaded_error` | The service is temporarily overloaded. Retry after a short delay. The Claude Platform returns 529 for this error type; this endpoint returns 503. | ## Authentication @@ -150,7 +150,7 @@ Each successful request creates a new session. There is no idempotency key. If a Routine runs count against a per-account daily allowance that varies by plan, and the resulting sessions draw down the same Claude Code subscription usage as interactive sessions. When either limit is reached, the endpoint returns `429 rate_limit_error` with a `Retry-After` header. Organizations with extra usage enabled continue past the included allowance on metered overage. -Your remaining daily runs are shown at [claude.ai/code/routines](https://claude.ai/code/routines). For how routine usage interacts with subscription limits and extra usage billing, see [Usage and limits](https://code.claude.com/docs/en/routines#usage-and-limits) in the Claude Code documentation. +View your remaining daily runs at [claude.ai/code/routines](https://claude.ai/code/routines). For how routine usage interacts with subscription limits and extra usage billing, see [Usage and limits](https://code.claude.com/docs/en/routines#usage-and-limits) in the Claude Code documentation. ## SDK support @@ -158,6 +158,6 @@ This endpoint is not in the Anthropic SDKs. Its token model differs from API key ## See also -- [Automate work with routines](https://code.claude.com/docs/en/routines) in the Claude Code documentation -- [Beta headers](/docs/en/api/beta-headers) -- [Errors](/docs/en/api/errors) \ No newline at end of file +* [Automate work with routines](https://code.claude.com/docs/en/routines) in the Claude Code documentation +* [Beta headers](/docs/en/api/beta-headers) +* [Errors](/docs/en/api/errors) diff --git a/content/en/api/claude-platform-on-aws-iam-actions.md b/content/en/api/claude-platform-on-aws-iam-actions.md index d64ea3cd3..86ca24879 100644 --- a/content/en/api/claude-platform-on-aws-iam-actions.md +++ b/content/en/api/claude-platform-on-aws-iam-actions.md @@ -8,14 +8,14 @@ Claude Platform on AWS uses AWS IAM for access control. Every API route maps to ## Service details -| Attribute | Value | -| :--- | :--- | +| Attribute | Value | +| ---------------------- | ------------------------ | | **IAM service prefix** | `aws-external-anthropic` | -| **Resource types** | `workspace` | +| **Resource types** | `workspace` | Workspace ARN format: -```text +```text wrap arn:aws:aws-external-anthropic:{region}:{account-id}:workspace/{workspace-id} ``` @@ -23,308 +23,354 @@ The ARN region is always populated and matches the region the workspace is bound ## Actions -The service defines 58 actions. Actions follow the AWS `VerbNoun` convention and use verb discipline so that `Get*` and `List*` wildcards produce a clean read-only boundary. +The service defines 65 actions. Actions follow the AWS `VerbNoun` convention and use verb discipline so that `Get*` and `List*` wildcards produce a clean read-only boundary. ### Inference -| Action | Routes authorized | -| :--- | :--- | -| `CreateInference` | `POST /v1/messages` | -| `CountTokens` | `POST /v1/messages/count_tokens` | +| Action | Routes authorized | +| ----------------- | -------------------------------- | +| `CreateInference` | `POST /v1/messages` | +| `CountTokens` | `POST /v1/messages/count_tokens` | ### Batch processing -| Action | Routes authorized | -| :--- | :--- | -| `CreateBatchInference` | `POST /v1/messages/batches` | -| `GetBatchInference` | `GET /v1/messages/batches/{id}`
`GET /v1/messages/batches/{id}/results` | -| `ListBatchInferences` | `GET /v1/messages/batches` | -| `CancelBatchInference` | `POST /v1/messages/batches/{id}/cancel` | -| `DeleteBatchInference` | `DELETE /v1/messages/batches/{id}` | +| Action | Routes authorized | +| ---------------------- | ----------------------------------------------------------------------- | +| `CreateBatchInference` | `POST /v1/messages/batches` | +| `GetBatchInference` | `GET /v1/messages/batches/{id}` `GET /v1/messages/batches/{id}/results` | +| `ListBatchInferences` | `GET /v1/messages/batches` | +| `CancelBatchInference` | `POST /v1/messages/batches/{id}/cancel` | +| `DeleteBatchInference` | `DELETE /v1/messages/batches/{id}` | -`GetBatchInference` authorizes both reading batch metadata and downloading batch results. The `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, and `AnthropicLimitedAccess` policies' `Get*` wildcards include this action. + `GetBatchInference` authorizes both reading batch metadata and downloading batch results. The `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, and `AnthropicLimitedAccess` policies' `Get*` wildcards include this action. ### Models -| Action | Routes authorized | -| :--- | :--- | -| `GetModel` | `GET /v1/models/{id}` | -| `ListModels` | `GET /v1/models` | +| Action | Routes authorized | +| ------------ | --------------------- | +| `GetModel` | `GET /v1/models/{id}` | +| `ListModels` | `GET /v1/models` | ### Files -| Action | Routes authorized | -| :--- | :--- | -| `CreateFile` | `POST /v1/files` | -| `GetFile` | `GET /v1/files/{id}`
`GET /v1/files/{id}/content` | -| `ListFiles` | `GET /v1/files` | -| `DeleteFile` | `DELETE /v1/files/{id}` | +| Action | Routes authorized | +| ------------ | ------------------------------------------------- | +| `CreateFile` | `POST /v1/files` | +| `GetFile` | `GET /v1/files/{id}` `GET /v1/files/{id}/content` | +| `ListFiles` | `GET /v1/files` | +| `DeleteFile` | `DELETE /v1/files/{id}` | -`GetFile` authorizes both metadata and content download. A principal with read-only access can download file bytes, not just list files. + `GetFile` authorizes both metadata and content download. A principal with read-only access can download file bytes, not just list files. ### Skills -| Action | Routes authorized | -| :--- | :--- | -| `CreateSkill` | `POST /v1/skills` | -| `GetSkill` | `GET /v1/skills/{id}`
`GET /v1/skills/{id}/versions`
`GET /v1/skills/{id}/versions/{version}` | -| `ListSkills` | `GET /v1/skills` | -| `UpdateSkill` | `POST /v1/skills/{id}/versions`
`DELETE /v1/skills/{id}/versions/{version}` | -| `DeleteSkill` | `DELETE /v1/skills/{id}` | +| Action | Routes authorized | +| ------------- | ---------------------------------------------------------------------------------------------------------------------------------------------- | +| `CreateSkill` | `POST /v1/skills` | +| `GetSkill` | `GET /v1/skills/{id}` `GET /v1/skills/{id}/versions` `GET /v1/skills/{id}/versions/{version}` `GET /v1/skills/{id}/versions/{version}/content` | +| `ListSkills` | `GET /v1/skills` | +| `UpdateSkill` | `POST /v1/skills/{id}/versions` `DELETE /v1/skills/{id}/versions/{version}` | +| `DeleteSkill` | `DELETE /v1/skills/{id}` | -Creating or deleting an individual skill version maps to `UpdateSkill`, not `CreateSkill` or `DeleteSkill`. A policy that denies `aws-external-anthropic:Delete*` still allows version deletion, and a policy that denies `aws-external-anthropic:Create*` still allows version creation. Deny `UpdateSkill` and `CreateSkill` as well if you need to prevent any skill mutation. + `GetSkill` authorizes both skill metadata and skill-content download. A principal with read-only access can download skill bytes, not just list skills. + + + + Creating or deleting an individual skill version maps to `UpdateSkill`, not `CreateSkill` or `DeleteSkill`. A policy that denies `aws-external-anthropic:Delete*` still allows version deletion, and a policy that denies `aws-external-anthropic:Create*` still allows version creation. Deny `UpdateSkill` and `CreateSkill` as well if you need to prevent any skill mutation. ### Agents -| Action | Routes authorized | -| :--- | :--- | -| `CreateAgent` | `POST /v1/agents` | -| `GetAgent` | `GET /v1/agents/{id}`
`GET /v1/agents/{id}/versions` | -| `ListAgents` | `GET /v1/agents` | -| `UpdateAgent` | `POST /v1/agents/{id}` | -| `ArchiveAgent` | `POST /v1/agents/{id}/archive` | +| Action | Routes authorized | +| -------------- | ---------------------------------------------------- | +| `CreateAgent` | `POST /v1/agents` | +| `GetAgent` | `GET /v1/agents/{id}` `GET /v1/agents/{id}/versions` | +| `ListAgents` | `GET /v1/agents` | +| `UpdateAgent` | `POST /v1/agents/{id}` | +| `ArchiveAgent` | `POST /v1/agents/{id}/archive` | -Agents support only archive, not hard delete. A policy that denies `aws-external-anthropic:Delete*` does not block `ArchiveAgent`. Deny `ArchiveAgent`, `UpdateAgent`, and `CreateAgent` if you need to prevent any agent mutation. + Agents support only archive, not hard delete. A policy that denies `aws-external-anthropic:Delete*` does not block `ArchiveAgent`. Deny `ArchiveAgent`, `UpdateAgent`, and `CreateAgent` if you need to prevent any agent mutation. ### Sessions -| Action | Routes authorized | -| :--- | :--- | -| `CreateSession` | `POST /v1/sessions` | -| `GetSession` | `GET /v1/sessions/{id}`
`GET /v1/sessions/{id}/events`
`GET /v1/sessions/{id}/events/stream`
`GET /v1/sessions/{id}/resources`
`GET /v1/sessions/{id}/resources/{id}` | -| `ListSessions` | `GET /v1/sessions` | -| `UpdateSession` | `POST /v1/sessions/{id}`
`POST /v1/sessions/{id}/events`
`POST /v1/sessions/{id}/resources`
`POST /v1/sessions/{id}/resources/{id}`
`DELETE /v1/sessions/{id}/resources/{id}` | -| `ArchiveSession` | `POST /v1/sessions/{id}/archive` | -| `DeleteSession` | `DELETE /v1/sessions/{id}` | +| Action | Routes authorized | +| ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `CreateSession` | `POST /v1/sessions` | +| `GetSession` | `GET /v1/sessions/{id}` `GET /v1/sessions/{id}/events` `GET /v1/sessions/{id}/events/stream` `GET /v1/sessions/{id}/resources` `GET /v1/sessions/{id}/resources/{id}` | +| `ListSessions` | `GET /v1/sessions` | +| `UpdateSession` | `POST /v1/sessions/{id}` `POST /v1/sessions/{id}/events` `POST /v1/sessions/{id}/resources` `POST /v1/sessions/{id}/resources/{id}` `DELETE /v1/sessions/{id}/resources/{id}` | +| `ArchiveSession` | `POST /v1/sessions/{id}/archive` | +| `DeleteSession` | `DELETE /v1/sessions/{id}` | -`GetSession` authorizes reading session metadata, the full event stream (conversation history), and session resources. The `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, and `AnthropicLimitedAccess` policies' `Get*` wildcards include this action. + `GetSession` authorizes reading session metadata, the full event stream (conversation history), and session resources. The `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, and `AnthropicLimitedAccess` policies' `Get*` wildcards include this action. -Creating, updating, or deleting an individual session sub-resource (events or session resources) maps to `UpdateSession`, not `CreateSession` or `DeleteSession`. A policy that denies `aws-external-anthropic:Delete*` still allows sub-resource deletion, and a policy that denies `aws-external-anthropic:Create*` still allows sub-resource creation. Deny `UpdateSession`, `CreateSession`, and `ArchiveSession` as well if you need to prevent any session mutation. + Creating, updating, or deleting an individual session sub-resource (events or session resources) maps to `UpdateSession`, not `CreateSession` or `DeleteSession`. A policy that denies `aws-external-anthropic:Delete*` still allows sub-resource deletion, and a policy that denies `aws-external-anthropic:Create*` still allows sub-resource creation. Deny `UpdateSession`, `CreateSession`, and `ArchiveSession` as well if you need to prevent any session mutation. ### Environments -| Action | Routes authorized | -| :--- | :--- | -| `CreateEnvironment` | `POST /v1/environments` | -| `GetEnvironment` | `GET /v1/environments/{id}` | -| `ListEnvironments` | `GET /v1/environments` | -| `UpdateEnvironment` | `POST /v1/environments/{id}` | -| `ArchiveEnvironment` | `POST /v1/environments/{id}/archive` | -| `DeleteEnvironment` | `DELETE /v1/environments/{id}` | +| Action | Routes authorized | +| ------------------------ | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `CreateEnvironment` | `POST /v1/environments` | +| `GetEnvironment` | `GET /v1/environments/{id}` `GET /v1/environments/{id}/work` `GET /v1/environments/{id}/work/{work_id}` `GET /v1/environments/{id}/work/stats` | +| `ListEnvironments` | `GET /v1/environments` | +| `UpdateEnvironment` | `POST /v1/environments/{id}` | +| `ArchiveEnvironment` | `POST /v1/environments/{id}/archive` | +| `DeleteEnvironment` | `DELETE /v1/environments/{id}` | +| `ProcessEnvironmentWork` | `GET /v1/environments/{id}/work/poll` `POST /v1/environments/{id}/work/{work_id}` `POST /v1/environments/{id}/work/{work_id}/ack` `POST /v1/environments/{id}/work/{work_id}/heartbeat` `POST /v1/environments/{id}/work/{work_id}/stop` | + + + A policy that denies `aws-external-anthropic:Delete*` does not block `ArchiveEnvironment`. `ProcessEnvironmentWork` is not matched by `Create*`, `Update*`, `Delete*`, or `Archive*` wildcards. Deny `ArchiveEnvironment`, `UpdateEnvironment`, `CreateEnvironment`, and `ProcessEnvironmentWork` as well if you need to prevent any environment mutation. + -A policy that denies `aws-external-anthropic:Delete*` does not block `ArchiveEnvironment`. Deny `ArchiveEnvironment`, `UpdateEnvironment`, and `CreateEnvironment` as well if you need to prevent any environment mutation. + `ProcessEnvironmentWork` authorizes a [self-hosted sandbox](/docs/en/managed-agents/self-hosted-sandboxes) worker to poll for, acknowledge, heartbeat, stop, and post results on environment work items. Grant it only to principals that run self-hosted environment workers. The `AnthropicSelfHostedEnvironmentAccess` managed policy includes this action. ### Vaults -| Action | Routes authorized | -| :--- | :--- | -| `CreateVault` | `POST /v1/vaults` | -| `GetVault` | `GET /v1/vaults/{id}`
`GET /v1/vaults/{id}/credentials`
`GET /v1/vaults/{id}/credentials/{id}` | -| `ListVaults` | `GET /v1/vaults` | -| `UpdateVault` | `POST /v1/vaults/{id}`
`POST /v1/vaults/{id}/credentials`
`POST /v1/vaults/{id}/credentials/{id}`
`POST /v1/vaults/{id}/credentials/{id}/archive`
`DELETE /v1/vaults/{id}/credentials/{id}` | -| `ArchiveVault` | `POST /v1/vaults/{id}/archive` | -| `DeleteVault` | `DELETE /v1/vaults/{id}` | +| Action | Routes authorized | +| -------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `CreateVault` | `POST /v1/vaults` | +| `GetVault` | `GET /v1/vaults/{id}` `GET /v1/vaults/{id}/credentials` `GET /v1/vaults/{id}/credentials/{id}` | +| `ListVaults` | `GET /v1/vaults` | +| `UpdateVault` | `POST /v1/vaults/{id}` `POST /v1/vaults/{id}/credentials` `POST /v1/vaults/{id}/credentials/{id}` `POST /v1/vaults/{id}/credentials/{id}/archive` `DELETE /v1/vaults/{id}/credentials/{id}` | +| `ArchiveVault` | `POST /v1/vaults/{id}/archive` | +| `DeleteVault` | `DELETE /v1/vaults/{id}` | -Creating, updating, archiving, or deleting an individual vault credential maps to `UpdateVault`. Reading a credential maps to `GetVault`. Vault credential secrets are not exposed: secret fields are write-only and are never returned by `GetVault` (see [Authenticate with vaults](/docs/en/managed-agents/vaults)). A policy that denies `aws-external-anthropic:Delete*` still allows credential deletion, and a policy that denies `aws-external-anthropic:Create*` still allows credential creation. Deny `UpdateVault`, `CreateVault`, and `ArchiveVault` as well if you need to prevent any vault mutation. + Creating, updating, archiving, or deleting an individual vault credential maps to `UpdateVault`. Reading a credential maps to `GetVault`. Vault credential secrets are not exposed: secret fields are write-only and are never returned by `GetVault` (see [Authenticate with vaults](/docs/en/managed-agents/vaults)). A policy that denies `aws-external-anthropic:Delete*` still allows credential deletion, and a policy that denies `aws-external-anthropic:Create*` still allows credential creation. Deny `UpdateVault`, `CreateVault`, and `ArchiveVault` as well if you need to prevent any vault mutation. ### Memory stores -| Action | Routes authorized | -| :--- | :--- | -| `CreateMemoryStore` | `POST /v1/memory_stores` | -| `GetMemoryStore` | `GET /v1/memory_stores/{id}`
`GET /v1/memory_stores/{id}/memories`
`GET /v1/memory_stores/{id}/memories/{id}`
`GET /v1/memory_stores/{id}/memory_versions`
`GET /v1/memory_stores/{id}/memory_versions/{id}` | -| `ListMemoryStores` | `GET /v1/memory_stores` | -| `UpdateMemoryStore` | `POST /v1/memory_stores/{id}`
`POST /v1/memory_stores/{id}/memories`
`POST /v1/memory_stores/{id}/memories/{id}`
`DELETE /v1/memory_stores/{id}/memories/{id}`
`POST /v1/memory_stores/{id}/memory_versions/{id}/redact` | -| `ArchiveMemoryStore` | `POST /v1/memory_stores/{id}/archive` | -| `DeleteMemoryStore` | `DELETE /v1/memory_stores/{id}` | +| Action | Routes authorized | +| -------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `CreateMemoryStore` | `POST /v1/memory_stores` | +| `GetMemoryStore` | `GET /v1/memory_stores/{id}` `GET /v1/memory_stores/{id}/memories` `GET /v1/memory_stores/{id}/memories/{id}` `GET /v1/memory_stores/{id}/memory_versions` `GET /v1/memory_stores/{id}/memory_versions/{id}` | +| `ListMemoryStores` | `GET /v1/memory_stores` | +| `UpdateMemoryStore` | `POST /v1/memory_stores/{id}` `POST /v1/memory_stores/{id}/memories` `POST /v1/memory_stores/{id}/memories/{id}` `DELETE /v1/memory_stores/{id}/memories/{id}` `POST /v1/memory_stores/{id}/memory_versions/{id}/redact` | +| `ArchiveMemoryStore` | `POST /v1/memory_stores/{id}/archive` | +| `DeleteMemoryStore` | `DELETE /v1/memory_stores/{id}` | -`GetMemoryStore` authorizes reading store metadata, all memories, and memory version history. The `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, and `AnthropicLimitedAccess` policies' `Get*` wildcards include this action. + `GetMemoryStore` authorizes reading store metadata, all memories, and memory version history. The `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, and `AnthropicLimitedAccess` policies' `Get*` wildcards include this action. -Creating, updating, or deleting an individual memory and redacting a memory version both map to `UpdateMemoryStore`, not `CreateMemoryStore` or `DeleteMemoryStore`. A policy that denies `aws-external-anthropic:Delete*` still allows individual-memory deletion and memory-version redaction, and a policy that denies `aws-external-anthropic:Create*` still allows individual-memory creation. Deny `UpdateMemoryStore`, `CreateMemoryStore`, and `ArchiveMemoryStore` as well if you need to prevent any memory-store mutation. + Creating, updating, or deleting an individual memory and redacting a memory version both map to `UpdateMemoryStore`, not `CreateMemoryStore` or `DeleteMemoryStore`. A policy that denies `aws-external-anthropic:Delete*` still allows individual-memory deletion and memory-version redaction, and a policy that denies `aws-external-anthropic:Create*` still allows individual-memory creation. Deny `UpdateMemoryStore`, `CreateMemoryStore`, and `ArchiveMemoryStore` as well if you need to prevent any memory-store mutation. + + +### Webhooks + +| Action | Routes authorized | +| --------------------- | -------------------------------------------------- | +| `CreateWebhook` | `POST /v1/webhooks` | +| `GetWebhook` | `GET /v1/webhooks/{id}` | +| `ListWebhooks` | `GET /v1/webhooks` | +| `UpdateWebhook` | `POST /v1/webhooks/{id}` | +| `DeleteWebhook` | `DELETE /v1/webhooks/{id}` | +| `RotateWebhookSecret` | `POST /v1/webhooks/{id}/regenerate_signing_secret` | + + + Webhook signing secrets are write-only. `GetWebhook` returns webhook metadata only; it does not return the signing secret. + + + + `RotateWebhookSecret` is not matched by `aws-external-anthropic:Create*`, `Update*`, or `Delete*` wildcards. A policy that denies those patterns still allows secret rotation. Deny `RotateWebhookSecret`, `UpdateWebhook`, `CreateWebhook`, and `DeleteWebhook` if you need to prevent any webhook mutation. ### User profiles -| Action | Routes authorized | -| :--- | :--- | -| `CreateUserProfile` | `POST /v1/user_profiles` | -| `GetUserProfile` | `GET /v1/user_profiles/{id}` | -| `ListUserProfiles` | `GET /v1/user_profiles` | +| Action | Routes authorized | +| ------------------- | ----------------------------- | +| `CreateUserProfile` | `POST /v1/user_profiles` | +| `GetUserProfile` | `GET /v1/user_profiles/{id}` | +| `ListUserProfiles` | `GET /v1/user_profiles` | | `UpdateUserProfile` | `POST /v1/user_profiles/{id}` | -IAM action matching is case-insensitive. The wildcard `aws-external-anthropic:*File` matches `CreateFile`, `GetFile`, and `DeleteFile`, but does not match `ListFiles` (which ends in "files", not "file"). It also over-matches `CreateUserProfile`, `GetUserProfile`, and `UpdateUserProfile` because "Profile" ends in "file". If you intend to grant or deny only Files API actions, enumerate them explicitly (`CreateFile`, `GetFile`, `ListFiles`, `DeleteFile`) rather than using a `*File` suffix pattern. + IAM action matching is case-insensitive. The wildcard `aws-external-anthropic:*File` matches `CreateFile`, `GetFile`, and `DeleteFile`, but does not match `ListFiles` (which ends in "files", not "file"). It also over-matches `CreateUserProfile`, `GetUserProfile`, and `UpdateUserProfile` because "Profile" ends in "file". If you intend to grant or deny only Files API actions, enumerate them explicitly (`CreateFile`, `GetFile`, `ListFiles`, `DeleteFile`) rather than using a `*File` suffix pattern. ### Workspaces -| Action | Routes authorized | -| :--- | :--- | -| `CreateWorkspace` | `POST /v1/organizations/workspaces` | -| `GetWorkspace` | `GET /v1/organizations/workspaces/{id}` | -| `ListWorkspaces` | `GET /v1/organizations/workspaces` | -| `UpdateWorkspace` | `POST /v1/organizations/workspaces/{id}` | +| Action | Routes authorized | +| ------------------ | ------------------------------------------------ | +| `CreateWorkspace` | `POST /v1/organizations/workspaces` | +| `GetWorkspace` | `GET /v1/organizations/workspaces/{id}` | +| `ListWorkspaces` | `GET /v1/organizations/workspaces` | +| `UpdateWorkspace` | `POST /v1/organizations/workspaces/{id}` | | `ArchiveWorkspace` | `POST /v1/organizations/workspaces/{id}/archive` | -Workspaces support only archive, not hard delete. A policy that denies `aws-external-anthropic:Delete*` does not block `ArchiveWorkspace`. Deny `ArchiveWorkspace`, `UpdateWorkspace`, and `CreateWorkspace` if you need to prevent any workspace mutation. + Workspaces support only archive, not hard delete. A policy that denies `aws-external-anthropic:Delete*` does not block `ArchiveWorkspace`. Deny `ArchiveWorkspace`, `UpdateWorkspace`, and `CreateWorkspace` if you need to prevent any workspace mutation. ### Authentication -| Action | Routes authorized | -| :--- | :--- | -| `CallWithBearerToken` | (none) | +| Action | Routes authorized | +| --------------------- | ----------------- | +| `CallWithBearerToken` | (none) | `CallWithBearerToken` is an authentication-layer permission that authorizes a principal to authenticate through an API key (bearer token) rather than AWS SigV4. It does not map to a route. Grant it alongside the route-mapped actions you want the API key holder to perform. ### Console access -| Action | Routes authorized | -| :--- | :--- | -| `AssumeConsole` | (none) | +| Action | Routes authorized | +| --------------- | ----------------- | +| `AssumeConsole` | (none) | `AssumeConsole` authorizes a principal to open the Claude Console for a Claude Platform on AWS workspace through the AWS Console federation flow. It does not map to a route. Grant it to principals who should be able to click **Open Claude Console** on the Claude Platform on AWS service page in the AWS Console. The Claude Console role (Admin or Developer) is assigned separately by your Anthropic account representative; it is not derived from the principal's IAM permissions. See [Using the Claude Console](/docs/en/build-with-claude/claude-platform-on-aws#using-the-claude-console) for the sign-in flow and role descriptions. ## Route-to-action mapping -The following table lists every route on Claude Platform on AWS and the IAM action required to call it. Each IAM action also authorizes requests that use the `anthropic-beta` header; beta variants of a route do not require a separate IAM action. CloudTrail classifies each action as either a Data event (high-volume, data-plane operations) or a Management event (control-plane operations). Vault actions are classified as Management events because vaults hold credentials and benefit from default-on audit logging. Workspace actions are also classified as Management events because they are organization-scoped control-plane operations. All other actions, including inference, batch, model, file, skill, user profile, and the remaining Claude Managed Agents actions, are classified as Data events. - -| Method | Route | IAM action | CloudTrail event type | -| :--- | :--- | :--- | :--- | -| `POST` | `/v1/messages` | `CreateInference` | Data | -| `POST` | `/v1/messages/count_tokens` | `CountTokens` | Data | -| `POST` | `/v1/messages/batches` | `CreateBatchInference` | Data | -| `GET` | `/v1/messages/batches` | `ListBatchInferences` | Data | -| `GET` | `/v1/messages/batches/{id}` | `GetBatchInference` | Data | -| `GET` | `/v1/messages/batches/{id}/results` | `GetBatchInference` | Data | -| `POST` | `/v1/messages/batches/{id}/cancel` | `CancelBatchInference` | Data | -| `DELETE` | `/v1/messages/batches/{id}` | `DeleteBatchInference` | Data | -| `GET` | `/v1/models` | `ListModels` | Data | -| `GET` | `/v1/models/{id}` | `GetModel` | Data | -| `POST` | `/v1/files` | `CreateFile` | Data | -| `GET` | `/v1/files` | `ListFiles` | Data | -| `GET` | `/v1/files/{id}` | `GetFile` | Data | -| `GET` | `/v1/files/{id}/content` | `GetFile` | Data | -| `DELETE` | `/v1/files/{id}` | `DeleteFile` | Data | -| `POST` | `/v1/skills` | `CreateSkill` | Data | -| `GET` | `/v1/skills` | `ListSkills` | Data | -| `GET` | `/v1/skills/{id}` | `GetSkill` | Data | -| `DELETE` | `/v1/skills/{id}` | `DeleteSkill` | Data | -| `POST` | `/v1/skills/{id}/versions` | `UpdateSkill` | Data | -| `GET` | `/v1/skills/{id}/versions` | `GetSkill` | Data | -| `GET` | `/v1/skills/{id}/versions/{version}` | `GetSkill` | Data | -| `DELETE` | `/v1/skills/{id}/versions/{version}` | `UpdateSkill` | Data | -| `POST` | `/v1/user_profiles` | `CreateUserProfile` | Data | -| `GET` | `/v1/user_profiles` | `ListUserProfiles` | Data | -| `GET` | `/v1/user_profiles/{id}` | `GetUserProfile` | Data | -| `POST` | `/v1/user_profiles/{id}` | `UpdateUserProfile` | Data | -| `POST` | `/v1/organizations/workspaces` | `CreateWorkspace` | Management | -| `GET` | `/v1/organizations/workspaces` | `ListWorkspaces` | Management | -| `GET` | `/v1/organizations/workspaces/{id}` | `GetWorkspace` | Management | -| `POST` | `/v1/organizations/workspaces/{id}` | `UpdateWorkspace` | Management | -| `POST` | `/v1/organizations/workspaces/{id}/archive` | `ArchiveWorkspace` | Management | -| `POST` | `/v1/agents` | `CreateAgent` | Data | -| `GET` | `/v1/agents` | `ListAgents` | Data | -| `GET` | `/v1/agents/{id}` | `GetAgent` | Data | -| `POST` | `/v1/agents/{id}` | `UpdateAgent` | Data | -| `POST` | `/v1/agents/{id}/archive` | `ArchiveAgent` | Data | -| `GET` | `/v1/agents/{id}/versions` | `GetAgent` | Data | -| `POST` | `/v1/sessions` | `CreateSession` | Data | -| `GET` | `/v1/sessions` | `ListSessions` | Data | -| `GET` | `/v1/sessions/{id}` | `GetSession` | Data | -| `POST` | `/v1/sessions/{id}` | `UpdateSession` | Data | -| `POST` | `/v1/sessions/{id}/archive` | `ArchiveSession` | Data | -| `DELETE` | `/v1/sessions/{id}` | `DeleteSession` | Data | -| `GET` | `/v1/sessions/{id}/events` | `GetSession` | Data | -| `POST` | `/v1/sessions/{id}/events` | `UpdateSession` | Data | -| `GET` | `/v1/sessions/{id}/events/stream` | `GetSession` | Data | -| `GET` | `/v1/sessions/{id}/resources` | `GetSession` | Data | -| `GET` | `/v1/sessions/{id}/resources/{id}` | `GetSession` | Data | -| `POST` | `/v1/sessions/{id}/resources` | `UpdateSession` | Data | -| `POST` | `/v1/sessions/{id}/resources/{id}` | `UpdateSession` | Data | -| `DELETE` | `/v1/sessions/{id}/resources/{id}` | `UpdateSession` | Data | -| `POST` | `/v1/environments` | `CreateEnvironment` | Data | -| `GET` | `/v1/environments` | `ListEnvironments` | Data | -| `GET` | `/v1/environments/{id}` | `GetEnvironment` | Data | -| `POST` | `/v1/environments/{id}` | `UpdateEnvironment` | Data | -| `POST` | `/v1/environments/{id}/archive` | `ArchiveEnvironment` | Data | -| `DELETE` | `/v1/environments/{id}` | `DeleteEnvironment` | Data | -| `POST` | `/v1/vaults` | `CreateVault` | Management | -| `GET` | `/v1/vaults` | `ListVaults` | Management | -| `GET` | `/v1/vaults/{id}` | `GetVault` | Management | -| `POST` | `/v1/vaults/{id}` | `UpdateVault` | Management | -| `POST` | `/v1/vaults/{id}/archive` | `ArchiveVault` | Management | -| `DELETE` | `/v1/vaults/{id}` | `DeleteVault` | Management | -| `GET` | `/v1/vaults/{id}/credentials` | `GetVault` | Management | -| `POST` | `/v1/vaults/{id}/credentials` | `UpdateVault` | Management | -| `GET` | `/v1/vaults/{id}/credentials/{id}` | `GetVault` | Management | -| `POST` | `/v1/vaults/{id}/credentials/{id}` | `UpdateVault` | Management | -| `POST` | `/v1/vaults/{id}/credentials/{id}/archive` | `UpdateVault` | Management | -| `DELETE` | `/v1/vaults/{id}/credentials/{id}` | `UpdateVault` | Management | -| `POST` | `/v1/memory_stores` | `CreateMemoryStore` | Data | -| `GET` | `/v1/memory_stores` | `ListMemoryStores` | Data | -| `GET` | `/v1/memory_stores/{id}` | `GetMemoryStore` | Data | -| `POST` | `/v1/memory_stores/{id}` | `UpdateMemoryStore` | Data | -| `POST` | `/v1/memory_stores/{id}/archive` | `ArchiveMemoryStore` | Data | -| `DELETE` | `/v1/memory_stores/{id}` | `DeleteMemoryStore` | Data | -| `POST` | `/v1/memory_stores/{id}/memories` | `UpdateMemoryStore` | Data | -| `GET` | `/v1/memory_stores/{id}/memories` | `GetMemoryStore` | Data | -| `GET` | `/v1/memory_stores/{id}/memories/{id}` | `GetMemoryStore` | Data | -| `POST` | `/v1/memory_stores/{id}/memories/{id}` | `UpdateMemoryStore` | Data | -| `DELETE` | `/v1/memory_stores/{id}/memories/{id}` | `UpdateMemoryStore` | Data | -| `GET` | `/v1/memory_stores/{id}/memory_versions` | `GetMemoryStore` | Data | -| `GET` | `/v1/memory_stores/{id}/memory_versions/{id}` | `GetMemoryStore` | Data | -| `POST` | `/v1/memory_stores/{id}/memory_versions/{id}/redact` | `UpdateMemoryStore` | Data | +The following table lists every route on Claude Platform on AWS and the IAM action required to call it. Each IAM action also authorizes requests that use the `anthropic-beta` header; beta variants of a route do not require a separate IAM action. CloudTrail classifies each action as either a Data event (high-volume, data-plane operations) or a Management event (control-plane operations). Vault and webhook actions are classified as Management events because they hold secrets (vault credentials and webhook signing secrets) and benefit from default-on audit logging. Workspace actions are also classified as Management events because they are organization-scoped control-plane operations. All other actions, including inference, batch, model, file, skill, user profile, and the remaining Claude Managed Agents actions, are classified as Data events. + +| Method | Route | IAM action | CloudTrail event type | +| -------- | ---------------------------------------------------- | ------------------------ | --------------------- | +| `POST` | `/v1/messages` | `CreateInference` | Data | +| `POST` | `/v1/messages/count_tokens` | `CountTokens` | Data | +| `POST` | `/v1/messages/batches` | `CreateBatchInference` | Data | +| `GET` | `/v1/messages/batches` | `ListBatchInferences` | Data | +| `GET` | `/v1/messages/batches/{id}` | `GetBatchInference` | Data | +| `GET` | `/v1/messages/batches/{id}/results` | `GetBatchInference` | Data | +| `POST` | `/v1/messages/batches/{id}/cancel` | `CancelBatchInference` | Data | +| `DELETE` | `/v1/messages/batches/{id}` | `DeleteBatchInference` | Data | +| `GET` | `/v1/models` | `ListModels` | Data | +| `GET` | `/v1/models/{id}` | `GetModel` | Data | +| `POST` | `/v1/files` | `CreateFile` | Data | +| `GET` | `/v1/files` | `ListFiles` | Data | +| `GET` | `/v1/files/{id}` | `GetFile` | Data | +| `GET` | `/v1/files/{id}/content` | `GetFile` | Data | +| `DELETE` | `/v1/files/{id}` | `DeleteFile` | Data | +| `POST` | `/v1/skills` | `CreateSkill` | Data | +| `GET` | `/v1/skills` | `ListSkills` | Data | +| `GET` | `/v1/skills/{id}` | `GetSkill` | Data | +| `DELETE` | `/v1/skills/{id}` | `DeleteSkill` | Data | +| `POST` | `/v1/skills/{id}/versions` | `UpdateSkill` | Data | +| `GET` | `/v1/skills/{id}/versions` | `GetSkill` | Data | +| `GET` | `/v1/skills/{id}/versions/{version}` | `GetSkill` | Data | +| `GET` | `/v1/skills/{id}/versions/{version}/content` | `GetSkill` | Data | +| `DELETE` | `/v1/skills/{id}/versions/{version}` | `UpdateSkill` | Data | +| `POST` | `/v1/user_profiles` | `CreateUserProfile` | Data | +| `GET` | `/v1/user_profiles` | `ListUserProfiles` | Data | +| `GET` | `/v1/user_profiles/{id}` | `GetUserProfile` | Data | +| `POST` | `/v1/user_profiles/{id}` | `UpdateUserProfile` | Data | +| `POST` | `/v1/organizations/workspaces` | `CreateWorkspace` | Management | +| `GET` | `/v1/organizations/workspaces` | `ListWorkspaces` | Management | +| `GET` | `/v1/organizations/workspaces/{id}` | `GetWorkspace` | Management | +| `POST` | `/v1/organizations/workspaces/{id}` | `UpdateWorkspace` | Management | +| `POST` | `/v1/organizations/workspaces/{id}/archive` | `ArchiveWorkspace` | Management | +| `POST` | `/v1/agents` | `CreateAgent` | Data | +| `GET` | `/v1/agents` | `ListAgents` | Data | +| `GET` | `/v1/agents/{id}` | `GetAgent` | Data | +| `POST` | `/v1/agents/{id}` | `UpdateAgent` | Data | +| `POST` | `/v1/agents/{id}/archive` | `ArchiveAgent` | Data | +| `GET` | `/v1/agents/{id}/versions` | `GetAgent` | Data | +| `POST` | `/v1/sessions` | `CreateSession` | Data | +| `GET` | `/v1/sessions` | `ListSessions` | Data | +| `GET` | `/v1/sessions/{id}` | `GetSession` | Data | +| `POST` | `/v1/sessions/{id}` | `UpdateSession` | Data | +| `POST` | `/v1/sessions/{id}/archive` | `ArchiveSession` | Data | +| `DELETE` | `/v1/sessions/{id}` | `DeleteSession` | Data | +| `GET` | `/v1/sessions/{id}/events` | `GetSession` | Data | +| `POST` | `/v1/sessions/{id}/events` | `UpdateSession` | Data | +| `GET` | `/v1/sessions/{id}/events/stream` | `GetSession` | Data | +| `GET` | `/v1/sessions/{id}/resources` | `GetSession` | Data | +| `GET` | `/v1/sessions/{id}/resources/{id}` | `GetSession` | Data | +| `POST` | `/v1/sessions/{id}/resources` | `UpdateSession` | Data | +| `POST` | `/v1/sessions/{id}/resources/{id}` | `UpdateSession` | Data | +| `DELETE` | `/v1/sessions/{id}/resources/{id}` | `UpdateSession` | Data | +| `POST` | `/v1/environments` | `CreateEnvironment` | Data | +| `GET` | `/v1/environments` | `ListEnvironments` | Data | +| `GET` | `/v1/environments/{id}` | `GetEnvironment` | Data | +| `POST` | `/v1/environments/{id}` | `UpdateEnvironment` | Data | +| `POST` | `/v1/environments/{id}/archive` | `ArchiveEnvironment` | Data | +| `DELETE` | `/v1/environments/{id}` | `DeleteEnvironment` | Data | +| `GET` | `/v1/environments/{id}/work` | `GetEnvironment` | Data | +| `GET` | `/v1/environments/{id}/work/poll` | `ProcessEnvironmentWork` | Data | +| `GET` | `/v1/environments/{id}/work/{work_id}` | `GetEnvironment` | Data | +| `GET` | `/v1/environments/{id}/work/stats` | `GetEnvironment` | Data | +| `POST` | `/v1/environments/{id}/work/{work_id}` | `ProcessEnvironmentWork` | Data | +| `POST` | `/v1/environments/{id}/work/{work_id}/ack` | `ProcessEnvironmentWork` | Data | +| `POST` | `/v1/environments/{id}/work/{work_id}/heartbeat` | `ProcessEnvironmentWork` | Data | +| `POST` | `/v1/environments/{id}/work/{work_id}/stop` | `ProcessEnvironmentWork` | Data | +| `POST` | `/v1/vaults` | `CreateVault` | Management | +| `GET` | `/v1/vaults` | `ListVaults` | Management | +| `GET` | `/v1/vaults/{id}` | `GetVault` | Management | +| `POST` | `/v1/vaults/{id}` | `UpdateVault` | Management | +| `POST` | `/v1/vaults/{id}/archive` | `ArchiveVault` | Management | +| `DELETE` | `/v1/vaults/{id}` | `DeleteVault` | Management | +| `GET` | `/v1/vaults/{id}/credentials` | `GetVault` | Management | +| `POST` | `/v1/vaults/{id}/credentials` | `UpdateVault` | Management | +| `GET` | `/v1/vaults/{id}/credentials/{id}` | `GetVault` | Management | +| `POST` | `/v1/vaults/{id}/credentials/{id}` | `UpdateVault` | Management | +| `POST` | `/v1/vaults/{id}/credentials/{id}/archive` | `UpdateVault` | Management | +| `DELETE` | `/v1/vaults/{id}/credentials/{id}` | `UpdateVault` | Management | +| `POST` | `/v1/memory_stores` | `CreateMemoryStore` | Data | +| `GET` | `/v1/memory_stores` | `ListMemoryStores` | Data | +| `GET` | `/v1/memory_stores/{id}` | `GetMemoryStore` | Data | +| `POST` | `/v1/memory_stores/{id}` | `UpdateMemoryStore` | Data | +| `POST` | `/v1/memory_stores/{id}/archive` | `ArchiveMemoryStore` | Data | +| `DELETE` | `/v1/memory_stores/{id}` | `DeleteMemoryStore` | Data | +| `POST` | `/v1/memory_stores/{id}/memories` | `UpdateMemoryStore` | Data | +| `GET` | `/v1/memory_stores/{id}/memories` | `GetMemoryStore` | Data | +| `GET` | `/v1/memory_stores/{id}/memories/{id}` | `GetMemoryStore` | Data | +| `POST` | `/v1/memory_stores/{id}/memories/{id}` | `UpdateMemoryStore` | Data | +| `DELETE` | `/v1/memory_stores/{id}/memories/{id}` | `UpdateMemoryStore` | Data | +| `GET` | `/v1/memory_stores/{id}/memory_versions` | `GetMemoryStore` | Data | +| `GET` | `/v1/memory_stores/{id}/memory_versions/{id}` | `GetMemoryStore` | Data | +| `POST` | `/v1/memory_stores/{id}/memory_versions/{id}/redact` | `UpdateMemoryStore` | Data | +| `GET` | `/v1/webhooks` | `ListWebhooks` | Management | +| `GET` | `/v1/webhooks/{id}` | `GetWebhook` | Management | +| `POST` | `/v1/webhooks` | `CreateWebhook` | Management | +| `POST` | `/v1/webhooks/{id}` | `UpdateWebhook` | Management | +| `DELETE` | `/v1/webhooks/{id}` | `DeleteWebhook` | Management | +| `POST` | `/v1/webhooks/{id}/regenerate_signing_secret` | `RotateWebhookSecret` | Management | Routes not in this table are not available on Claude Platform on AWS. The gateway denies any route not listed here by default. -Workspace routes are the only Admin API routes available on Claude Platform on AWS. The Claude Console Workspaces page is read-only; use the Admin API or the AWS Console to create, update, or archive workspaces. + Workspace routes are the only Admin API routes available on Claude Platform on AWS. The Claude Console Workspaces page is read-only; use the Admin API or the AWS Console to create, update, or archive workspaces. ## Managed policies -AWS provides four managed policies for Claude Platform on AWS. All managed policies apply to `Resource: "*"`. +AWS provides five managed policies for Claude Platform on AWS. All managed policies apply to `Resource: "*"`. -| Policy | Grants | -| :--- | :--- | -| `AnthropicFullAccess` | `aws-external-anthropic:*` | -| `AnthropicReadOnlyAccess` | `Get*`, `List*`, `CallWithBearerToken` | -| `AnthropicInferenceAccess` | `Get*`, `List*`, `CreateInference`, `CreateBatchInference`, `CancelBatchInference`, `DeleteBatchInference`, `CountTokens`, `CallWithBearerToken` | -| `AnthropicLimitedAccess` | All `AnthropicInferenceAccess` actions, plus all Claude Managed Agents actions (agents, sessions, environments, vaults, and memory stores) | +| Policy | Grants | +| -------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `AnthropicFullAccess` | `aws-external-anthropic:*` | +| `AnthropicReadOnlyAccess` | `Get*`, `List*`, `CallWithBearerToken` | +| `AnthropicInferenceAccess` | `Get*`, `List*`, `CreateInference`, `CreateBatchInference`, `CancelBatchInference`, `DeleteBatchInference`, `CountTokens`, `CallWithBearerToken` | +| `AnthropicLimitedAccess` | All `AnthropicInferenceAccess` actions, plus all Claude Managed Agents actions (agents, sessions, environments, vaults, memory stores, webhooks, and self-hosted environment work) | +| `AnthropicSelfHostedEnvironmentAccess` | `GetEnvironment`, `ProcessEnvironmentWork`, `GetSession`, `UpdateSession`, `GetSkill`, `CallWithBearerToken` | -`AnthropicInferenceAccess` is the narrowest managed policy sufficient to run inference. It covers both synchronous and batch inference and, through the `Get*` and `List*` wildcards, grants read access to every API resource in the namespace, including Claude Managed Agents (CMA) resources (agents, sessions, environments, vaults, and memory stores). This includes file content download through `GetFile` (see the [Files](#files) note) and memory contents through `GetMemoryStore`. Vault credential secrets are not exposed: secret fields are write-only and are never returned by `GetVault` (see [Authenticate with vaults](/docs/en/managed-agents/vaults)). `AnthropicInferenceAccess` does not grant file creation or deletion, skill management, user profile management, workspace mutation, or any Claude Managed Agents write action (create, update, archive, delete). To exclude CMA reads, replace `AnthropicInferenceAccess` with a custom policy that enumerates only the specific non-CMA actions you need. +`AnthropicInferenceAccess` is the narrowest managed policy sufficient to run inference. It covers both synchronous and batch inference and, through the `Get*` and `List*` wildcards, grants read access to every API resource in the namespace, including Claude Managed Agents (CMA) resources (agents, sessions, environments, vaults, memory stores, and webhooks). This includes file content download through `GetFile` (see the [Files](#files) note), skill content download through `GetSkill` (see the [Skills](#skills) note), and memory contents through `GetMemoryStore`. Vault credential secrets and webhook signing secrets are not exposed: those fields are write-only and are never returned by `GetVault` or `GetWebhook` (see [Authenticate with vaults](/docs/en/managed-agents/vaults)). `AnthropicInferenceAccess` does not grant file creation or deletion, skill management, user profile management, workspace mutation, or any Claude Managed Agents write action (create, update, archive, delete, process, or rotate). To exclude CMA reads, replace `AnthropicInferenceAccess` with a custom policy that enumerates only the specific non-CMA actions you need. -`AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, and `AnthropicLimitedAccess` all carry the `Get*` and `List*` wildcards, which grant read access to all content in the workspace: file bytes, batch results, session conversation history, and memory contents. Vault credential secrets are not exposed; secret fields are write-only and are never returned by `GetVault`. If your principal should not read existing content, use a custom policy that enumerates only the actions you need. + `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, and `AnthropicLimitedAccess` all carry the `Get*` and `List*` wildcards, which grant read access to all content in the workspace: file bytes, skill content, batch results, session conversation history, and memory contents. Vault credential secrets and webhook signing secrets are not exposed; those fields are write-only and are never returned by `GetVault` or `GetWebhook`. If your principal should not read existing content, use a custom policy that enumerates only the actions you need. `AnthropicLimitedAccess` includes all Claude Managed Agents actions in addition to inference actions. -`AssumeConsole` is not included in `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, or `AnthropicLimitedAccess`. Principals who need Claude Console access require either `AnthropicFullAccess` or a custom policy that grants `aws-external-anthropic:AssumeConsole`. See [Console access](#console-access). +`AnthropicSelfHostedEnvironmentAccess` is the narrowest managed policy sufficient to run a [self-hosted sandbox](/docs/en/managed-agents/self-hosted-sandboxes) worker. Attach it to the principal your environment worker authenticates as. + +`AssumeConsole` is not included in `AnthropicReadOnlyAccess`, `AnthropicInferenceAccess`, `AnthropicLimitedAccess`, or `AnthropicSelfHostedEnvironmentAccess`. Principals who need Claude Console access require either `AnthropicFullAccess` or a custom policy that grants `aws-external-anthropic:AssumeConsole`. See [Console access](#console-access). -`CreateInference` and `CreateBatchInference` are separate actions. Denying one does not block the other. If you intend to prevent all model calls, deny both. + `CreateInference` and `CreateBatchInference` are separate actions. Denying one does not block the other. If you intend to prevent all model calls, deny both. ## Example policies @@ -353,9 +399,9 @@ Grants the minimal permissions for an IAM principal that runs inference against ``` -`ListWorkspaces` is account-scoped (see [Provisioning automation](#provisioning-automation)). If your service account needs to enumerate workspaces, add a separate `Allow` statement for `ListWorkspaces` with `Resource: "*"`. + `ListWorkspaces` is account-scoped (see [Provisioning automation](#provisioning-automation)). If your service account needs to enumerate workspaces, add a separate `Allow` statement for `ListWorkspaces` with `Resource: "*"`. -This policy assumes AWS SigV4 authentication. If the principal authenticates with an API key, add a separate `Allow` statement for `aws-external-anthropic:CallWithBearerToken` with `Resource: "*"`. `CallWithBearerToken` is a route-less action that does not bind to a workspace ARN. See [Per-customer workspace isolation](#per-customer-workspace-isolation) for the two-statement pattern. + This policy assumes AWS SigV4 authentication. If the principal authenticates with an API key, add a separate `Allow` statement for `aws-external-anthropic:CallWithBearerToken` with `Resource: "*"`. `CallWithBearerToken` is a route-less action that does not bind to a workspace ARN. See [Per-customer workspace isolation](#per-customer-workspace-isolation) for the two-statement pattern. ### Per-customer workspace isolation @@ -384,9 +430,9 @@ Restricts a role to a single workspace: ``` -The `aws-external-anthropic:*` wildcard in the first statement includes account-scoped actions (`CreateWorkspace`, `ListWorkspaces`) that the workspace ARN constraint silently filters out. This is consistent with the "isolation" intent (the role cannot create or enumerate workspaces), but the policy contains permissions that have no effect. See [Provisioning automation](#provisioning-automation) for the account-scoped pattern. + The `aws-external-anthropic:*` wildcard in the first statement includes account-scoped actions (`CreateWorkspace`, `ListWorkspaces`) that the workspace ARN constraint silently filters out. This is consistent with the "isolation" intent (the role cannot create or enumerate workspaces), but the policy contains permissions that have no effect. See [Provisioning automation](#provisioning-automation) for the account-scoped pattern. -`CallWithBearerToken` and `AssumeConsole` are route-less actions that do not bind to a workspace ARN. The second statement grants them on `Resource: "*"` so the role can authenticate with an API key and open the Claude Console. Omit this statement if the role uses SigV4 only and does not need Claude Console access. + `CallWithBearerToken` and `AssumeConsole` are route-less actions that do not bind to a workspace ARN. The second statement grants them on `Resource: "*"` so the role can authenticate with an API key and open the Claude Console. Omit this statement if the role uses SigV4 only and does not need Claude Console access. ### Feature lockdown for a ZDR-sensitive workspace @@ -410,13 +456,13 @@ Blocks batch processing and file upload on a specific workspace while leaving sy ``` -This deny blocks creation only. Other file and batch actions are not denied unless you list them as well. For a complete lockdown where the workspace must never hold files or batches, also deny `aws-external-anthropic:GetFile`, `aws-external-anthropic:ListFiles`, `aws-external-anthropic:DeleteFile`, `aws-external-anthropic:GetBatchInference`, `aws-external-anthropic:ListBatchInferences`, `aws-external-anthropic:CancelBatchInference`, and `aws-external-anthropic:DeleteBatchInference`. + This deny blocks creation only. Other file and batch actions are not denied unless you list them as well. For a complete lockdown where the workspace must never hold files or batches, also deny `aws-external-anthropic:GetFile`, `aws-external-anthropic:ListFiles`, `aws-external-anthropic:DeleteFile`, `aws-external-anthropic:GetBatchInference`, `aws-external-anthropic:ListBatchInferences`, `aws-external-anthropic:CancelBatchInference`, and `aws-external-anthropic:DeleteBatchInference`. ### Provisioning automation -The Claude Console Workspaces page is read-only; use the Admin API workspace endpoints or the AWS Console to create, update, or archive workspaces. + The Claude Console Workspaces page is read-only; use the Admin API workspace endpoints or the AWS Console to create, update, or archive workspaces. Grants a CI/CD role the actions needed to create and manage workspaces, without any inference permissions: @@ -444,6 +490,6 @@ Grants a CI/CD role the actions needed to create and manage workspaces, without ## See also -- [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) for setup, authentication, and platform overview -- [AWS IAM User Guide](https://docs.aws.amazon.com/IAM/latest/UserGuide/introduction.html) for IAM policy syntax and evaluation logic -- [AWS CloudTrail User Guide](https://docs.aws.amazon.com/awscloudtrail/latest/userguide/) for audit logging configuration \ No newline at end of file +* [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) for setup, authentication, and platform overview +* [AWS IAM User Guide](https://docs.aws.amazon.com/IAM/latest/UserGuide/introduction.html) for IAM policy syntax and evaluation logic +* [AWS CloudTrail User Guide](https://docs.aws.amazon.com/awscloudtrail/latest/userguide/) for audit logging configuration diff --git a/content/en/api/errors.md b/content/en/api/errors.md index f1627241b..e4d2e0882 100644 --- a/content/en/api/errors.md +++ b/content/en/api/errors.md @@ -1,4 +1,6 @@ -# Errors +# Claude API errors + +Understand the HTTP status codes, error response shape, and request IDs the Claude API returns, and handle errors with the SDKs' typed exceptions. --- @@ -7,40 +9,53 @@ The API follows a predictable HTTP error code format: * 400 - `invalid_request_error`: There was an issue with the format or content of your request. This error type may also be used for other 4XX status codes not listed in this section. -* 401 - `authentication_error`: There's an issue with your API key. On Claude Platform on AWS, this can also indicate a problem with your AWS credentials or SigV4 signature. + +* 401 - `authentication_error`: There's an issue with your API key (for example, it's malformed, revoked, or expired; see [Key expiration](/docs/en/manage-claude/authentication#key-expiration)). On Claude Platform on AWS, this can also indicate a problem with your AWS credentials or SigV4 signature. + * 402 - `billing_error`: There's an issue with your billing or payment information. Check your payment details in the [Claude Console](https://platform.claude.com), or in AWS Marketplace if you're using Claude Platform on AWS. -* 403 - `permission_error`: Your API key does not have permission to use the specified resource. -* 404 - `not_found_error`: The requested resource was not found. + +* 403 - `permission_error`: Your API key does not have permission to use the specified resource. Check your organization's access and workspace settings in the [Claude Console](https://platform.claude.com). + +* 404 - `not_found_error`: The requested resource was not found. Check the endpoint path and any resource IDs in the request URL. + +* 409 - `conflict_error`: The request conflicts with the current state of a resource. For example, the resource was modified concurrently, or a value that must be unique is already in use. Resolve the conflict, then retry the request. + * 413 - `request_too_large`: Request exceeds the maximum allowed number of bytes. See [Request size limits](#request-size-limits) for per-endpoint maximums. + * 429 - `rate_limit_error`: Your account has hit a rate limit. -* 500 - `api_error`: An unexpected error has occurred internal to Anthropic's systems. -* 504 - `timeout_error`: The request timed out while processing. Consider using [streaming](/docs/en/build-with-claude/streaming) for long-running requests. + +* 500 - `api_error`: An unexpected error has occurred internal to Anthropic's systems. Retry the request with exponential backoff; if the error persists, contact support with the [request ID](#request-id). + +* 504 - `timeout_error`: The request timed out while processing. Consider using the [streaming Messages API](/docs/en/build-with-claude/streaming) for long-running requests. See [Long requests](#long-requests) for more options. + * 529 - `overloaded_error`: The API is temporarily overloaded. - 529 errors can occur when APIs experience high traffic across all users. + 529 errors can occur when the API experiences high traffic across all users. - In rare cases, if your organization has a sharp increase in usage, you might see 429 errors because of acceleration limits on the API. To avoid hitting acceleration limits, ramp up your traffic gradually and maintain consistent usage patterns. + In rare cases, if your organization has a sharp increase in usage, you might see 429 errors because of acceleration limits on the API. To avoid hitting acceleration limits, ramp up your traffic gradually and maintain consistent usage patterns. -When receiving a [streaming](/docs/en/build-with-claude/streaming) response over SSE, it's possible that an error can occur after returning a 200 response, in which case error handling wouldn't follow these standard mechanisms. +The official SDKs automatically retry transient failures (such as connection errors, rate limits, and 5xx server errors) with exponential backoff, twice by default, honoring the `retry-after` header when present. Each SDK client accepts a maximum-retries option to configure or disable this behavior. + +When receiving a [streaming](/docs/en/build-with-claude/streaming) response over server-sent events (SSE), an error can occur after the API returns a 200 response. In that case, error handling doesn't follow these standard mechanisms. See [Error events](/docs/en/build-with-claude/streaming#error-events) for the shape of mid-stream errors. ## Request size limits -The API enforces request size limits to ensure optimal performance: +The API enforces request size limits: -| Endpoint type | Maximum request size | -|:---|:---| -| Messages API | 32 MB | -| Token Counting API | 32 MB | -| [Batch API](/docs/en/build-with-claude/batch-processing) | 256 MB | -| [Files API](/docs/en/build-with-claude/files) | 500 MB | +| Endpoint type | Maximum request size | +| -------------------------------------------------------- | -------------------- | +| Messages API | 32 MB | +| Token Counting API | 32 MB | +| [Batch API](/docs/en/build-with-claude/batch-processing) | 256 MB | +| [Files API](/docs/en/build-with-claude/files) | 500 MB | -If you exceed these limits, you'll receive a 413 `request_too_large` error. On the direct Claude API, this error is returned from Cloudflare before the request reaches the API servers. +If you exceed these limits, you'll receive a 413 `request_too_large` error. On the direct Claude API, Cloudflare returns this error before the request reaches the API servers. ## Error shapes -Errors are always returned as JSON, with a top-level `error` object that always includes a `type` and `message` value. The response also includes a `request_id` field for easier tracking and debugging. For example: +The API always returns errors as JSON, with a top-level `error` object that always includes a `type` and `message` value. The response also includes a `request_id` field for easier tracking and debugging. For example: ```json JSON { @@ -55,57 +70,145 @@ Errors are always returned as JSON, with a top-level `error` object that always In accordance with the [versioning](/docs/en/api/versioning) policy, the values within these objects may expand, and it is possible that the `type` values will grow over time. +## SDK error types + +The official SDKs raise typed exceptions for these errors instead of returning raw JSON, and the class names and namespaces differ by language. For example, a 404 surfaces as `anthropic.NotFoundError` in Python, `Anthropic::Errors::NotFoundError` in Ruby, `com.anthropic.errors.NotFoundException` in Java, and as a single `*anthropic.Error` value (branch on `StatusCode`) in Go. Catch the SDK's typed classes rather than string-matching error messages, handling the most specific classes first. Each SDK page documents its full exception hierarchy: + +* [Python](/docs/en/cli-sdks-libraries/sdks/python#handling-errors) · [TypeScript](/docs/en/cli-sdks-libraries/sdks/typescript#handling-errors) · [C#](/docs/en/cli-sdks-libraries/sdks/csharp#error-handling) · [Go](/docs/en/cli-sdks-libraries/sdks/go#error-handling) · [Java](/docs/en/cli-sdks-libraries/sdks/java#error-handling) · [PHP](/docs/en/cli-sdks-libraries/sdks/php#error-handling) · [Ruby](/docs/en/cli-sdks-libraries/sdks/ruby#handling-errors) + ## Request ID -Every API response includes a unique `request-id` header. This header contains a value such as `req_018EeWyXxfu5pfWkrYcMdjWG`. When contacting support about a specific request, include this ID to help quickly resolve your issue. +Every API response includes a unique `request-id` header. This header contains a value such as `req_018EeWyXxfu5pfWkrYcMdjWG`. The same identifier appears as the `request_id` field in [error response bodies](#error-shapes). When contacting support about a specific request, include this ID to help quickly resolve your issue. On [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws), responses include two request IDs: the AWS request ID (`x-amzn-requestid`, primary, indexed in CloudTrail) and the Anthropic request ID (`request-id`, secondary). Use the AWS request ID for CloudTrail lookups and the Anthropic request ID for Anthropic support tickets. -The official SDKs provide the Anthropic request ID as a property on top-level response objects, containing the value of the `request-id` header. On Claude Platform on AWS, use the raw-response accessor to also read the AWS request ID from the HTTP headers: +The Python and TypeScript SDKs expose the request ID as a `_request_id` property on top-level response objects. The C#, Go, Java, and PHP SDKs expose it through their raw-response accessors, which also let you read any other response header. On Claude Platform on AWS, use the raw-response accessor to read the AWS request ID (`x-amzn-requestid`) as well: + ```bash cURL + # Print the response headers (including request-id); discard the body + curl -sS -D - -o /dev/null https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-sonnet-5", + "max_tokens": 1024, + "messages": [{"role": "user", "content": "Hello, Claude"}] + }' + ``` + ```bash CLI # The request-id header is printed to stderr with --debug: ant --debug messages create \ - --model claude-opus-4-7 \ + --model claude-sonnet-5 \ --max-tokens 1024 \ --message '{role: user, content: "Hello, Claude"}' ``` - ```python Python hidelines={1..2} - import anthropic - + ```python Python client = anthropic.Anthropic() message = client.messages.create( - model="claude-opus-4-7", + model="claude-sonnet-5", max_tokens=1024, messages=[{"role": "user", "content": "Hello, Claude"}], ) print(f"Request ID: {message._request_id}") ``` - ```typescript TypeScript hidelines={1..2} - import Anthropic from "@anthropic-ai/sdk"; - + ```typescript TypeScript const client = new Anthropic(); const message = await client.messages.create({ - model: "claude-opus-4-7", + model: "claude-sonnet-5", max_tokens: 1024, messages: [{ role: "user", content: "Hello, Claude" }] }); console.log("Request ID:", message._request_id); ``` - - ```python Python (Claude Platform on AWS) nocheck + ```csharp C# + AnthropicClient client = new(); + + var response = await client.WithRawResponse.Messages.Create(new MessageCreateParams + { + Model = Model.ClaudeSonnet5, + MaxTokens = 1024, + Messages = [new() { Role = Role.User, Content = "Hello, Claude" }] + }); + Console.WriteLine($"Request ID: {response.RequestID}"); + ``` + + ```go Go + client := anthropic.NewClient() + + var response *http.Response + message, err := client.Messages.New( + context.Background(), + anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeSonnet5, + MaxTokens: 1024, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("Hello, Claude")), + }, + }, + option.WithResponseInto(&response), + ) + if err != nil { + panic(err) + } + + fmt.Println("Request ID:", response.Header.Get("request-id")) + fmt.Println(message.Content[0].Text) + ``` + + ```java Java + import com.anthropic.client.AnthropicClient; + import com.anthropic.client.okhttp.AnthropicOkHttpClient; + import com.anthropic.core.http.HttpResponseFor; + import com.anthropic.models.messages.Message; + import com.anthropic.models.messages.MessageCreateParams; + import com.anthropic.models.messages.Model; + + void main() { + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + HttpResponseFor response = client.messages().withRawResponse().create( + MessageCreateParams.builder() + .model(Model.CLAUDE_SONNET_5) + .maxTokens(1024) + .addUserMessage("Hello, Claude") + .build() + ); + + IO.println("Request ID: " + response.requestId().orElse(null)); + } + ``` + + ```php PHP + $client = new Client(); + + $response = $client->messages->raw->create([ + 'model' => 'claude-sonnet-5', + 'maxTokens' => 1024, + 'messages' => [['role' => 'user', 'content' => 'Hello, Claude']], + ]); + echo 'Request ID: ' . $response->getHeaderLine('request-id') . "\n"; + ``` + + ```ruby Ruby + # Accessing raw response headers is not currently supported in the Ruby SDK. + # To read the request-id header, use one of the other SDK examples. + ``` + + ```python Python (Claude Platform on AWS) from anthropic import AnthropicAWS client = AnthropicAWS(aws_region="us-west-2") response = client.messages.with_raw_response.create( - model="claude-opus-4-7", + model="claude-opus-4-8", max_tokens=1024, messages=[{"role": "user", "content": "Hello, Claude"}], ) @@ -114,21 +217,20 @@ The official SDKs provide the Anthropic request ID as a property on top-level re print(f"Anthropic request ID: {message._request_id}") ``` - - ```typescript TypeScript (Claude Platform on AWS) nocheck + ```typescript TypeScript (Claude Platform on AWS) import AnthropicAws from "@anthropic-ai/aws-sdk"; const client = new AnthropicAws({ awsRegion: "us-west-2" }); - const { data: message, response: raw } = await client.messages + const { response: raw, request_id } = await client.messages .create({ - model: "claude-opus-4-7", + model: "claude-opus-4-8", max_tokens: 1024, messages: [{ role: "user", content: "Hello, Claude" }] }) .withResponse(); console.log("AWS request ID:", raw.headers.get("x-amzn-requestid")); - console.log("Anthropic request ID:", message._request_id); + console.log("Anthropic request ID:", request_id); ``` @@ -137,44 +239,162 @@ For Claude Platform on AWS request-ID examples in other languages, see [Request ## Long requests - Consider using the [streaming Messages API](/docs/en/build-with-claude/streaming) or [Message Batches API](/docs/en/api/creating-message-batches) for long running requests, especially those over 10 minutes. + Consider using the [streaming Messages API](/docs/en/build-with-claude/streaming) or [Message Batches API](/docs/en/api/messages/batches/create) for long-running requests, especially those over 10 minutes. -Avoid setting a large `max_tokens` value without using the [streaming Messages API](/docs/en/build-with-claude/streaming) -or [Message Batches API](/docs/en/api/creating-message-batches): +Avoid setting a large `max_tokens` value without using the [streaming Messages API](/docs/en/build-with-claude/streaming) or [Message Batches API](/docs/en/api/messages/batches/create): -- Some networks may drop idle connections after a variable period of time, which -can cause the request to fail or timeout without receiving a response from Anthropic. -- Networks differ in reliability; the [Message Batches API](/docs/en/api/creating-message-batches) can help you -manage the risk of network issues by allowing you to poll for results rather than requiring an uninterrupted network connection. +* Some networks may drop idle connections after a variable period of time, which can cause the request to fail or time out without receiving a response from Anthropic. +* Networks differ in reliability. The [Message Batches API](/docs/en/api/messages/batches/create) can help you manage the risk of network issues by allowing you to poll for results rather than requiring an uninterrupted network connection. -If you are building a direct API integration, you should be aware that setting a [TCP socket keep-alive](https://tldp.org/HOWTO/TCP-Keepalive-HOWTO/programming.html) can reduce the impact of idle connection timeouts on some networks. +If you are building a direct API integration, setting a [TCP socket keep-alive](https://tldp.org/HOWTO/TCP-Keepalive-HOWTO/programming.html) can reduce the impact of idle connection timeouts on some networks. -The [SDKs](/docs/en/api/client-sdks) validate that your non-streaming Messages API requests are not expected to exceed a 10 minute timeout and -also will set a socket option for TCP keep-alive. +The [SDKs](/docs/en/cli-sdks-libraries/overview) validate that your non-streaming Messages API requests are not expected to exceed a 10-minute timeout. They also set a socket option for TCP keep-alive. -If you don't need to process events incrementally, use `.stream()` with `.get_final_message()` (Python) or `.finalMessage()` (TypeScript) to get the complete `Message` object without writing event-handling code: +If you don't need to process events incrementally, the SDKs can consume the stream for you and return the complete `Message` object, identical to what a non-streaming call returns: - ```python Python - with client.messages.stream( - max_tokens=128000, - messages=[{"role": "user", "content": "Write a detailed analysis..."}], - model="claude-opus-4-7", - ) as stream: - message = stream.get_final_message() - print(message.content) - ``` - - ```typescript TypeScript - const stream = client.messages.stream({ - max_tokens: 128000, - messages: [{ role: "user", content: "Write a detailed analysis..." }], - model: "claude-opus-4-7" - }); - const message = await stream.finalMessage(); - console.log(message.content); - ``` + ```bash cURL + # Raw SSE output requires handling events; there is no single-command way + # to accumulate the final message with curl. Use the SDK examples instead. + ``` + + ```bash CLI + # The CLI streams events; --format jsonl emits one event per line + ant messages create --stream --format jsonl <<'YAML' + model: claude-sonnet-5 + max_tokens: 128000 + messages: + - role: user + content: Write a detailed analysis... + YAML + ``` + + ```python Python + client = anthropic.Anthropic() + + with client.messages.stream( + max_tokens=128000, + messages=[{"role": "user", "content": "Write a detailed analysis..."}], + model="claude-sonnet-5", + ) as stream: + message = stream.get_final_message() + + print(message.content[0].text) + ``` + + ```typescript TypeScript + const client = new Anthropic(); + + const stream = client.messages.stream({ + max_tokens: 128000, + messages: [{ role: "user", content: "Write a detailed analysis..." }], + model: "claude-sonnet-5" + }); + + const message = await stream.finalMessage(); + const textBlock = message.content.find((block) => block.type === "text"); + if (textBlock && textBlock.type === "text") { + console.log(textBlock.text); + } + ``` + + ```csharp C# + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = Model.ClaudeSonnet5, + MaxTokens = 128000, + Messages = [new() { Role = Role.User, Content = "Write a detailed analysis..." }] + }; + + var message = await client.Messages.CreateStreaming(parameters).Aggregate(); + Console.WriteLine(message); + ``` + + ```go Go + client := anthropic.NewClient() + + stream := client.Messages.NewStreaming(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeSonnet5, + MaxTokens: 128000, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("Write a detailed analysis...")), + }, + }) + + message := anthropic.Message{} + for stream.Next() { + event := stream.Current() + if err := message.Accumulate(event); err != nil { + log.Fatal(err) + } + } + if err := stream.Err(); err != nil { + log.Fatal(err) + } + + fmt.Println(message.Content[0].Text) + ``` + + ```java Java + import com.anthropic.client.AnthropicClient; + import com.anthropic.client.okhttp.AnthropicOkHttpClient; + import com.anthropic.helpers.MessageAccumulator; + import com.anthropic.models.messages.Message; + import com.anthropic.models.messages.MessageCreateParams; + import com.anthropic.models.messages.Model; + + void main() { + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model(Model.CLAUDE_SONNET_5) + .maxTokens(128000L) + .addUserMessage("Write a detailed analysis...") + .build(); + + MessageAccumulator accumulator = MessageAccumulator.create(); + try (var streamResponse = client.messages().createStreaming(params)) { + streamResponse.stream().forEach(accumulator::accumulate); + } + + Message message = accumulator.message(); + message.content().get(0).text().ifPresent(textBlock -> IO.println(textBlock.text())); + } + ``` + + ```php PHP + use Anthropic\Lib\Streaming\MessageAccumulator; + + $client = new Client(); + + $stream = $client->messages->createStream( + model: 'claude-sonnet-5', + maxTokens: 128000, + messages: [['role' => 'user', 'content' => 'Write a detailed analysis...']], + ); + + $accumulator = MessageAccumulator::forMessages(); + foreach ($stream as $event) { + $accumulator->accumulate($event); + } + + echo $accumulator->message()->content[0]->text; + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.stream( + model: "claude-sonnet-5", + max_tokens: 128000, + messages: [{ role: "user", content: "Write a detailed analysis..." }] + ).accumulated_message + + puts message.content.first.text + ``` See [Streaming Messages](/docs/en/build-with-claude/streaming#get-the-final-message-without-handling-events) for more details. @@ -183,7 +403,7 @@ See [Streaming Messages](/docs/en/build-with-claude/streaming#get-the-final-mess ### Prefill not supported -[Claude Mythos Preview](https://anthropic.com/glasswing), Claude Opus 4.7, Claude Opus 4.6, and Claude Sonnet 4.6 do not support prefilling assistant messages. Sending a request with a prefilled last assistant message to any of these models returns a 400 `invalid_request_error`: +Claude Fable 5, [Claude Mythos 5](https://anthropic.com/glasswing), [Claude Mythos Preview](https://anthropic.com/glasswing), Claude Opus 4.8, Claude Opus 4.7, Claude Opus 4.6, and Claude Sonnet 4.6 do not support prefilling assistant messages. Sending a request with a prefilled last assistant message to any of these models returns a 400 `invalid_request_error`: ```json { @@ -195,8 +415,34 @@ See [Streaming Messages](/docs/en/build-with-claude/streaming#get-the-final-mess } ``` -Use [structured outputs](/docs/en/build-with-claude/structured-outputs), system prompt instructions, or [`output_config.format`](/docs/en/build-with-claude/structured-outputs#json-outputs) instead. +Use [structured outputs](/docs/en/build-with-claude/structured-outputs) on models that support it, system prompt instructions, or [`output_config.format`](/docs/en/build-with-claude/structured-outputs#json-outputs) instead. + +### Thinking blocks cannot be modified + +If the most recent assistant message contains `thinking` or `redacted_thinking` blocks that were edited, reordered, filtered out, or reconstructed before being sent back to the API, the request returns a 400 `invalid_request_error`. The error message starts with the position of the offending block (for example, `messages.1.content.0`) and contains: + +```text wrap +`thinking` or `redacted_thinking` blocks in the latest assistant message cannot be modified. These blocks must remain as they were in the original response. +``` + +With tool use, every `thinking` and `redacted_thinking` block from the assistant turn must be passed back exactly as received, including blocks whose `thinking` field is empty. Pass thinking blocks back unchanged, and if your application filters content blocks by type before resending, include both `thinking` and `redacted_thinking`. See [Preserving thinking blocks](/docs/en/build-with-claude/extended-thinking#preserving-thinking-blocks) and [Thinking output on Claude Fable 5 and Claude Mythos 5](/docs/en/build-with-claude/adaptive-thinking#thinking-output-on-claude-fable-5-and-claude-mythos-5). ### Outbound web identity federation disabled (Claude Platform on AWS) -If every request to [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) returns `"Outbound web identity federation is disabled for your account"`, run `aws iam enable-outbound-web-identity-federation` once per AWS account. See [Enable outbound web identity federation](/docs/en/build-with-claude/claude-platform-on-aws#enable-outbound-web-identity-federation) for details. \ No newline at end of file +If every request to [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) returns `"Outbound web identity federation is disabled for your account"`, run `aws iam enable-outbound-web-identity-federation` once per AWS account. See [Enable outbound web identity federation](/docs/en/build-with-claude/claude-platform-on-aws#enable-outbound-web-identity-federation) for details. + +## Next steps + + + + Start a Claude Code routine session on demand by sending an authenticated POST request. + + + + To mitigate misuse and manage capacity on the API, limits are in place on how much an organization can use the Claude API. + + + + Stream Messages API responses incrementally with server-sent events, including text, tool use, and extended thinking deltas. + + diff --git a/content/en/api/ip-addresses.md b/content/en/api/ip-addresses.md index da85d61c0..a1ccfa879 100644 --- a/content/en/api/ip-addresses.md +++ b/content/en/api/ip-addresses.md @@ -5,7 +5,7 @@ Anthropic services use fixed IP addresses for both inbound and outbound connecti --- -**[Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws):** The inbound endpoint (`aws-external-anthropic.{region}.api.aws`) resolves to AWS IP ranges. Outbound tool calls (MCP connector, web search, and web fetch) originate from the Anthropic ranges listed on this page. See the [AWS IP address ranges](https://docs.aws.amazon.com/vpc/latest/userguide/aws-ip-ranges.html) for inbound allowlisting. + **[Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws):** The inbound endpoint (`aws-external-anthropic.{region}.api.aws`) resolves to AWS IP ranges. Outbound tool calls (MCP connector, web search, and web fetch) originate from the Anthropic ranges listed on this page. See the [AWS IP address ranges](https://docs.aws.amazon.com/vpc/latest/userguide/aws-ip-ranges.html) for inbound allowlisting. ## Inbound IP addresses @@ -32,10 +32,10 @@ These are the stable IP addresses that Anthropic uses for outbound requests (for The following IP addresses are no longer in use by Anthropic. If you have previously allowlisted these addresses, you should remove them from your firewall rules. -```text +```text wrap 34.162.46.92/32 34.162.102.82/32 34.162.136.91/32 34.162.142.92/32 34.162.183.95/32 -``` \ No newline at end of file +``` diff --git a/content/en/api/overview.md b/content/en/api/overview.md index 45d67c1cc..0df80ebc5 100644 --- a/content/en/api/overview.md +++ b/content/en/api/overview.md @@ -1,19 +1,21 @@ # API overview +Understand the Claude API's available endpoints, authentication headers, client SDKs, pagination, rate limits, and cloud platform access options. + --- The Claude API is a RESTful API at `https://api.anthropic.com` that provides programmatic access to Claude models and Claude Managed Agents. -**New to Claude?** For direct model access, start with [Get started](/docs/en/get-started) and [Working with Messages](/docs/en/build-with-claude/working-with-messages). For managed agent infrastructure, see the [Claude Managed Agents quickstart](/docs/en/managed-agents/quickstart). + **New to Claude?** For direct model access, start with [Get started](/docs/en/get-started) and [Working with Messages](/docs/en/build-with-claude/working-with-messages). For managed agent infrastructure, see the [Claude Managed Agents quickstart](/docs/en/managed-agents/quickstart). ## Prerequisites To use the Claude API, you'll need: -- A [Claude Console account](https://platform.claude.com) -- An [API key](/settings/keys), or a configured [Workload Identity Federation](/docs/en/manage-claude/workload-identity-federation) rule +* A [Claude Console account](https://platform.claude.com) +* An [API key](/settings/keys), or a configured [Workload Identity Federation](/docs/en/manage-claude/workload-identity-federation) rule For step-by-step setup instructions, see [Get started](/docs/en/get-started). @@ -22,17 +24,19 @@ For step-by-step setup instructions, see [Get started](/docs/en/get-started). The Claude API includes the following APIs: **General Availability:** -- **[Messages API](/docs/en/api/messages/create)**: Send messages to Claude for conversational interactions (`POST /v1/messages`) -- **[Message Batches API](/docs/en/api/creating-message-batches)**: Process large volumes of Messages requests asynchronously with 50% cost reduction (`POST /v1/messages/batches`) -- **[Token Counting API](/docs/en/api/messages-count-tokens)**: Count tokens in a message before sending to manage costs and rate limits (`POST /v1/messages/count_tokens`) -- **[Models API](/docs/en/api/models-list)**: List available Claude models and their details (`GET /v1/models`) + +* **[Messages API](/docs/en/api/messages/create)**: Send messages to Claude for conversational interactions (`POST /v1/messages`) +* **[Message Batches API](/docs/en/api/messages/batches/create)**: Process large volumes of Messages requests asynchronously with 50% cost reduction (`POST /v1/messages/batches`) +* **[Token Counting API](/docs/en/api/messages-count-tokens)**: Count tokens in a message before sending to manage costs and rate limits (`POST /v1/messages/count_tokens`) +* **[Models API](/docs/en/api/models/list)**: List available Claude models and their details (`GET /v1/models`) **Beta:** -- **[Files API](/docs/en/api/files-create)**: Upload and manage files for use across multiple API calls (`POST /v1/files`, `GET /v1/files`) -- **[Skills API](/docs/en/api/skills/create-skill)**: Create and manage custom agent skills (`POST /v1/skills`, `GET /v1/skills`) -- **[Agents API](/docs/en/managed-agents/agent-setup)**: Define reusable, versioned agent configurations for Claude Managed Agents (`POST /v1/agents`, `GET /v1/agents`) -- **[Sessions API](/docs/en/managed-agents/sessions)**: Run stateful agent sessions in managed cloud containers (`POST /v1/sessions`, `GET /v1/sessions/{id}/stream`) -- **[Environments API](/docs/en/managed-agents/environments)**: Configure container templates for agent sessions (`POST /v1/environments`, `GET /v1/environments`) + +* **[Files API](/docs/en/api/beta/files/upload)**: Upload and manage files for use across multiple API calls (`POST /v1/files`, `GET /v1/files`) +* **[Skills API](/docs/en/api/skills/create-skill)**: Create and manage custom agent skills (`POST /v1/skills`, `GET /v1/skills`) +* **[Agents API](/docs/en/managed-agents/agent-setup)**: Define reusable, versioned agent configurations for Claude Managed Agents (`POST /v1/agents`, `GET /v1/agents`) +* **[Sessions API](/docs/en/managed-agents/sessions)**: Run stateful agent sessions in managed cloud sandboxes (`POST /v1/sessions`, `GET /v1/sessions/{id}/stream`) +* **[Environments API](/docs/en/managed-agents/environments)**: Configure sandbox templates for agent sessions (`POST /v1/environments`, `GET /v1/environments`) For the complete API reference with all endpoints, parameters, and response schemas, explore the API reference pages listed in the navigation. To access beta features, see [Beta headers](/docs/en/api/beta-headers). @@ -40,12 +44,12 @@ For the complete API reference with all endpoints, parameters, and response sche For details on both authentication methods and when to use each, see [Authentication](/docs/en/manage-claude/authentication). All requests to the Claude API must include these headers: -| Header | Value | Required | -|--------|-------|----------| -| `x-api-key` | Your API key from Console | One of `x-api-key` or `Authorization` | -| `Authorization` | `Bearer `, where `` is a short-lived access token obtained from `POST /v1/oauth/token` via [Workload Identity Federation](/docs/en/manage-claude/workload-identity-federation) | One of `x-api-key` or `Authorization` | -| `anthropic-version` | API version (e.g., `2023-06-01`) | Yes | -| `content-type` | `application/json` | Yes | +| Header | Value | Required | +| ------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------- | +| `x-api-key` | Your API key from Console | One of `x-api-key` or `Authorization` | +| `Authorization` | `Bearer `, where `` is a short-lived access token obtained from `POST /v1/oauth/token` through [Workload Identity Federation](/docs/en/manage-claude/workload-identity-federation) | One of `x-api-key` or `Authorization` | +| `anthropic-version` | API version (for example, `2023-06-01`) | Yes | +| `content-type` | `application/json` | Yes | If you are using the [Client SDKs](#client-sdks), the SDK will send these headers automatically. For API versioning details, see [API versions](/docs/en/api/versioning). @@ -53,20 +57,21 @@ When accessing Claude through a [cloud platform](#claude-api-vs-cloud-platforms) ### Getting API keys -The API is made available through the web [Console](https://platform.claude.com/). You can use the [Workbench](https://platform.claude.com/workbench) to try out the API in the browser and then generate API keys in [Account Settings](https://platform.claude.com/settings/keys). Use [workspaces](https://platform.claude.com/settings/workspaces) to segment your API keys and [control spend](/docs/en/api/rate-limits) by use case. +The API is made available through the web [Console](https://platform.claude.com/). You can use the [Workbench](https://platform.claude.com/workbench) to try out the API in the browser and then generate API keys in [Account Settings](https://platform.claude.com/settings/keys). You choose each key's [expiration](/docs/en/manage-claude/authentication#key-expiration) when you create it. Use [workspaces](https://platform.claude.com/settings/workspaces) to segment your API keys and [control spend](/docs/en/api/rate-limits) by use case. ## Client SDKs Anthropic provides official SDKs that simplify API integration by handling authentication, request formatting, error handling, and more. **Benefits:** -- Automatic header management (x-api-key, anthropic-version, content-type) -- Type-safe request and response handling -- Built-in retry logic and error handling -- Streaming support -- Request timeouts and connection management -For a list of client SDKs and their respective installation instructions, see [Client SDKs](/docs/en/api/client-sdks). +* Automatic header management (x-api-key, anthropic-version, content-type) +* Type-safe request and response handling +* Built-in retry logic and error handling +* Streaming support +* Request timeouts and connection management + +For a list of client SDKs, see [Client SDKs](/docs/en/cli-sdks-libraries/overview). ## Claude API vs cloud platforms @@ -74,66 +79,87 @@ Claude is available through the direct Claude API and through cloud platforms. C ### Claude API -- **Direct access** to the latest models and features -- **Anthropic billing and support** -- **Best for:** New integrations, full feature access, direct relationship with Anthropic +* **Direct access** to the latest models and features +* **Anthropic billing and support** +* **Best for:** New integrations, full feature access, direct relationship with Anthropic ### Cloud platform APIs Access Claude through AWS, Google Cloud, or Microsoft Azure: -- **Integrated** with cloud provider billing and IAM -- **Feature availability varies by platform:** Anthropic-operated platforms include [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) and [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry); partner-operated platforms include Amazon Bedrock and Vertex AI. See each platform's page for feature availability and timing. -- **Best for:** Existing cloud commitments, specific compliance requirements, consolidated cloud billing -| Platform | Provider | Documentation | -|----------|----------|---------------| -| Claude Platform on AWS | AWS (Anthropic-operated) | [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) | -| Amazon Bedrock | AWS | [Claude in Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) | -| Vertex AI | Google Cloud | [Claude on Vertex AI](/docs/en/build-with-claude/claude-on-vertex-ai) | -| Microsoft Foundry | Microsoft Azure (Anthropic-operated) | [Claude in Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) | +* **Integrated** with cloud provider billing and IAM +* **Feature availability varies by platform:** Anthropic-operated platforms include [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) and [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry); partner-operated platforms include Amazon Bedrock and Google Cloud. See each platform's page for feature availability and timing. +* **Best for:** Existing cloud commitments, specific compliance requirements, consolidated cloud billing + +| Platform | Provider | Documentation | +| ---------------------- | ------------------------------------ | ------------------------------------------------------------------------------------- | +| Agent Platform | Google Cloud | [Claude on Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) | +| Amazon Bedrock | AWS | [Claude in Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) | +| Claude Platform on AWS | AWS (Anthropic-operated) | [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) | +| Microsoft Foundry | Microsoft Azure (Anthropic-operated) | [Claude in Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) | -Claude Managed Agents is available through the direct Claude API and [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws). For feature availability across platforms, see the [Features overview](/docs/en/build-with-claude/overview). + Claude Managed Agents is available through the direct Claude API and [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws). For feature availability across platforms, see the [Features overview](/docs/en/build-with-claude/overview). ## Request and response format ### Request size limits -| Endpoint | Maximum request size | -| --- | --- | -| Messages, Token Counting | 32 MB | -| [Message Batches API](/docs/en/build-with-claude/batch-processing) | 256 MB | -| [Files API](/docs/en/build-with-claude/files) | 500 MB | -| Sessions, Agents, Environments | 32 MB | +| Endpoint | Maximum request size | +| ------------------------------------------------------------------ | -------------------- | +| Messages, Token Counting | 32 MB | +| [Message Batches API](/docs/en/build-with-claude/batch-processing) | 256 MB | +| [Files API](/docs/en/build-with-claude/files) | 500 MB | +| Sessions, Agents, Environments | 32 MB | If you exceed these limits, you'll receive a 413 `request_too_large` error. -Partner-operated platforms have their own request size limits: Vertex AI limits requests to 30 MB, and Bedrock limits requests to 20 MB. Claude Platform on AWS uses the same limits as the direct Claude API. Consult your platform's documentation for current values. + Partner-operated platforms have their own request size limits: Bedrock limits requests to 20 MB, and Google Cloud limits requests to 30 MB. Claude Platform on AWS uses the same limits as the direct Claude API. Consult your platform's documentation for current values. ### Response headers The Claude API includes the following headers in every response: -- `request-id`: A globally unique identifier for the request -- `anthropic-organization-id`: The organization ID associated with the API key used in the request +* `request-id`: A globally unique identifier for the request +* `anthropic-organization-id`: The organization ID associated with the API key used in the request + + + Claude Platform on AWS adds an AWS request ID (`x-amzn-requestid`) alongside the standard `request-id` header. See [Request IDs](/docs/en/build-with-claude/claude-platform-on-aws#request-ids) for the dual-ID handling pattern. + + +## Pagination + +List endpoints return results in pages. Most newer list endpoints use the `page` and `next_page` cursor scheme described in this section. Some use a different scheme; see the note at the end of this section. Use the `limit` query parameter to control the page size and the `page` query parameter to fetch an adjacent page. Each response includes a `data` array alongside cursor fields for navigating between pages. + +| Name | Location | Description | +| ----------- | --------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `limit` | Query parameter | Maximum number of items to return per page. | +| `page` | Query parameter | Opaque cursor from a previous response. Pass a `next_page` or `prev_page` value here to fetch the adjacent page. | +| `order` | Query parameter | Sort direction for the results (`asc` or `desc`), on list endpoints that support sorting. A `page` cursor is only valid with the `order` it was created with. | +| `next_page` | Response field | Cursor for the next page, or `null` if there are no more results. | +| `prev_page` | Response field | Cursor for the previous page on endpoints that support backward pagination (currently `GET /v1/sessions`), or `null` if you are on the first page. Other list endpoints omit the field. | + +To go back a page, pass `prev_page` as the `page` parameter. `prev_page` is `null` when you're on the first page. Not all list endpoints support `prev_page`. Only `GET /v1/sessions` returns `prev_page`; on list endpoints that do not support backward pagination, the field is absent from the response rather than `null`. For a request walkthrough, see [Listing sessions](/docs/en/managed-agents/session-operations#listing-sessions). + +Every SDK provides an auto-paginating iterator that follows `next_page` for you. In Python and TypeScript, you get it by iterating the list result directly. The other SDKs provide the iterator through a separate method. SDK auto-pagination is forward-only; to go back a page, read `prev_page` from the response and pass it back as the `page` parameter yourself. See [client SDKs](/docs/en/cli-sdks-libraries/overview) for language-specific details. -Claude Platform on AWS adds an AWS request ID (`x-amzn-requestid`) alongside the standard `request-id` header. See [Request IDs](/docs/en/build-with-claude/claude-platform-on-aws#request-ids) for the dual-ID handling pattern. + Some list endpoints use a different cursor scheme. The [Message Batches API](/docs/en/build-with-claude/batch-processing), the [Files API](/docs/en/build-with-claude/files), the [Models API](/docs/en/api/models/list), and several [Admin API](/docs/en/manage-claude/admin-api) endpoints take `after_id` and `before_id` query parameters instead of `page`. Their responses return `has_more`, `first_id`, and `last_id` instead of `next_page`. Some endpoints that use the `page` scheme, such as `GET /v1/skills`, also return a `has_more` Boolean alongside `next_page`. See the reference page for each endpoint for its exact pagination fields. ## Rate limits and availability ### Rate limits -The API enforces rate limits and spend limits to prevent misuse and manage capacity. Limits are organized into usage tiers that increase automatically as you use the API. Each tier has: +The API enforces rate limits and spend limits to prevent misuse and manage capacity. Limits are organized into usage tiers; your organization is placed on a tier automatically and can move to a higher tier over time. Each tier has: -- **Spend limits**: Maximum monthly cost for API usage -- **Rate limits**: Maximum number of requests per minute (RPM) and tokens per minute (TPM) +* **Spend limits**: Maximum monthly cost for API usage +* **Rate limits**: Maximum number of requests per minute (RPM) and tokens per minute (TPM) -You can view your organization's current limits in the [Console](/settings/limits). For higher limits or Priority Tier (enhanced service levels with committed spend), contact sales through the Console. +You can view your organization's current limits in the [Console](/settings/limits). For higher limits, use **Request rate limit increase** on the [Limits](/settings/limits) page. For detailed information about limits, tiers, and the token bucket algorithm used for rate limiting, see [Rate limits](/docs/en/api/rate-limits). @@ -147,13 +173,16 @@ The Claude API is available in [many countries and regions](/docs/en/api/support Complete API specification for direct model interactions + Agents, Sessions, and Environments endpoints - - Python, TypeScript, Java, Go, C#, Ruby, and PHP + + + Python, TypeScript, C#, Go, Java, PHP, and Ruby + - Usage tiers, spend limits, and token bucket algorithm + Usage tiers, requesting higher limits, and the token bucket algorithm - \ No newline at end of file + diff --git a/content/en/api/rate-limits.md b/content/en/api/rate-limits.md index 0851a995d..f3aa4c471 100644 --- a/content/en/api/rate-limits.md +++ b/content/en/api/rate-limits.md @@ -5,7 +5,7 @@ To mitigate misuse and manage capacity on the API, limits are in place on how mu --- -**[Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws):** The rate limits on this page apply. Billing and spend limits differ: spend limits are not available, and billing is through AWS Marketplace (not Anthropic credit purchases). Organizations start at Tier 1. Rate limit increases go through your Anthropic account representative; there is no automatic tier advancement, and per-workspace rate limit configuration is not available. [Fast mode](/docs/en/build-with-claude/fast-mode) and [Priority Tier](/docs/en/api/service-tiers) are not available on Claude Platform on AWS. + **[Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws):** The rate limits on this page apply to Claude Platform on AWS, but billing and limit management differ. Billing is through AWS Marketplace (not Anthropic credit purchases). Organizations on Claude Platform on AWS are placed on the Start tier and do not move between usage tiers automatically. To request higher limits, contact your Anthropic account representative or [Anthropic support](https://support.claude.com); the **Request rate limit increase** flow is not available. Spend limits are set in [Settings > Billing](/settings/billing) rather than **Settings > Limits**. Per-workspace rate limit configuration and [fast mode](/docs/en/build-with-claude/fast-mode) are not available on Claude Platform on AWS. For details, see [Rate limits and quotas on Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws#rate-limits-and-quotas). There are two types of limits: @@ -15,278 +15,225 @@ There are two types of limits: The API enforces service-configured limits at the organization level, but you may also set user-configurable limits for your organization's workspaces. -These limits apply to both Standard and Priority Tier usage. For more information about Priority Tier, which offers enhanced service levels in exchange for committed spend, see [Service Tiers](/docs/en/api/service-tiers). - ## About rate limits * Limits are designed to prevent API abuse, while minimizing impact on common customer usage patterns. -* Limits are defined by **usage tier**, where each tier is associated with a different set of spend and rate limits. -* Your organization will increase tiers automatically as you reach certain thresholds while using the API. - Limits are set at the organization level. You can see your organization's limits on the [Limits](/settings/limits) page in the [Claude Console](/). +* Limits are defined by **usage tier**. Your organization is placed on a tier automatically and can move to a higher tier over time as you use the API. +* Limits are set at the organization level. You can see your organization's tier and current limits on the [Limits](/settings/limits) page in the [Claude Console](/). * You might hit rate limits over shorter time intervals. For instance, a rate of 60 requests per minute (RPM) might be enforced as 1 request per second. Short bursts of requests can exceed the limit and trigger rate limit errors. -* The limits outlined below are the standard tier limits. If you're seeking higher, custom limits or Priority Tier for enhanced service levels, contact sales on the [Limits](/settings/limits) page. +* The following limits are the standard limits for each tier. If you need higher limits, see [Requesting higher limits](#requesting-higher-limits). * The API uses the [token bucket algorithm](https://en.wikipedia.org/wiki/Token_bucket) to do rate limiting. This means that your capacity is continuously replenished up to your maximum limit, rather than being reset at fixed intervals. * All limits described here represent maximum allowed usage, not guaranteed minimums. These limits are intended to reduce unintentional overspend and ensure fair distribution of resources among users. ## Spend limits -Each usage tier has a limit on how much you can spend on the API each calendar month. Once you reach the spend limit of your tier, until you qualify for the next tier, you will have to wait until the next month to be able to use the API again. - -To qualify for the next tier, you must meet a deposit requirement. To minimize the risk of overfunding your account, you cannot deposit more than your monthly spend limit. - -### Requirements to advance tier - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - -
Usage TierCredit PurchaseMax Credit PurchaseMonthly Spend Limit
Tier 1\$5\$100\$100
Tier 2\$40\$500\$500
Tier 3\$200\$1,000\$1,000
Tier 4\$400\$200,000\$200,000
Monthly InvoicingN/AN/ANo limit
- -**Credit Purchase** shows the cumulative credit purchases (excluding tax) required to advance to that tier. You advance immediately upon reaching the threshold. - -**Max Credit Purchase** limits the maximum amount you can add to your account in a single transaction to prevent account overfunding. - -**Monthly Spend Limit** is the maximum you can spend on the API each calendar month at that tier. + **[Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws):** Spend limits work differently on Claude Platform on AWS. Set spend limits in [Settings > Billing](/settings/billing) instead of **Settings > Limits**. See [Spend limits on Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws#spend-limits) for how spend caps and self-set spend limits apply to your organization. -## Increasing your spend limits +Each of the Start, Build, and Scale tiers carries a monthly spend cap, which is the maximum your organization can spend on the API each calendar month. Once you reach your tier's spend cap, API usage pauses until the next month unless you request a higher limit. You can view your organization's monthly spend cap on the [Limits](/settings/limits) page. -Your organization has two kinds of spend limits: a customer-set limit you control directly, and a tier-enforced ceiling set by your usage tier. Each has a different process for increasing it. +| Usage tier | Monthly spend cap | +| ---------- | ----------------- | +| Start | $500 USD | +| Build | $1,000 USD | +| Scale | $200,000 USD | -### Customer-set spend limits +Organizations on the Custom tier have no monthly spend cap; limits are arranged with their account team. -You can set a spend limit lower than your tier's ceiling to control costs. To adjust it: +You can also set your own spend limit below your tier's cap to control costs: Go to [Settings > Limits](/settings/limits) in the Claude Console. + In the **Spend limits** section, click **Change Limit** (or **Set spend limit** if no limit is currently set). + - Enter a new value. Your customer-set limit cannot exceed your current tier's limit. + Enter a new value. Your spend limit cannot exceed your current tier's cap. -### Tier-enforced spend limits - -When you need a limit higher than your tier's ceiling (Tier 4's ceiling is $200,000 per month), click **Contact Sales** on the [Limits](/settings/limits) page. This opens the contact form in a new tab, and a member of the sales team will follow up by email when your organization is upgraded. - -Monthly Invoicing removes the monthly spend cap entirely and uses Net-30 payment terms by default. - - -Support can also raise tier-enforced limits. For urgent needs, contact [support](https://support.anthropic.com). - - ## Rate limits -The rate limits for the Messages API are measured in requests per minute (RPM), input tokens per minute (ITPM), and output tokens per minute (OTPM) for each model class. -If you exceed any of the rate limits you will get a [429 error](/docs/en/api/errors) describing which rate limit was exceeded, along with a `retry-after` header indicating how long to wait. +The rate limits for the Messages API are measured in requests per minute (RPM), input tokens per minute (ITPM), and output tokens per minute (OTPM) for each model class. If you exceed any of the rate limits you will get a [429 error](/docs/en/api/errors) describing which rate limit was exceeded, along with a `retry-after` header indicating how long to wait. -You might also encounter 429 errors because of acceleration limits on the API if your organization has a sharp increase in usage. To avoid hitting acceleration limits, ramp up your traffic gradually and maintain consistent usage patterns. + You might also encounter 429 errors because of acceleration limits on the API if your organization has a sharp increase in usage. To avoid hitting acceleration limits, ramp up your traffic gradually and maintain consistent usage patterns. ### Cache-aware ITPM -Many API providers use a combined "tokens per minute" (TPM) limit that may include all tokens, both cached and uncached, input and output. **For most Claude models, only uncached input tokens count towards your ITPM rate limits.** This is a key advantage that makes the rate limits effectively higher than they might initially appear. +Many API providers use a combined "tokens per minute" (TPM) limit that may include all tokens, both cached and uncached, input and output. **For most Claude models, only uncached input tokens count toward your ITPM rate limits.** This is a key advantage that makes the rate limits effectively higher than they might initially appear. ITPM rate limits are estimated at the beginning of each request, and the estimate is adjusted during the request to reflect the actual number of input tokens used. -Here's what counts towards ITPM: -- `input_tokens` (tokens after the last cache breakpoint) ✓ **Count towards ITPM** -- `cache_creation_input_tokens` (tokens being written to cache) ✓ **Count towards ITPM** -- `cache_read_input_tokens` (tokens read from cache) ✗ **Do NOT count towards ITPM** for most models +Here's what counts toward ITPM: + +* `input_tokens` (tokens after the last cache breakpoint) ✓ **Count toward ITPM** +* `cache_creation_input_tokens` (tokens being written to cache) ✓ **Count toward ITPM** +* `cache_read_input_tokens` (tokens read from cache) ✗ **Do NOT count toward ITPM** for most models -The `input_tokens` field only represents tokens that appear **after your last cache breakpoint**, not all input tokens in your request. To calculate total input tokens: + The `input_tokens` field only represents tokens that appear **after your last cache breakpoint**, not all input tokens in your request. To calculate total input tokens: -```text -total_input_tokens = cache_read_input_tokens + cache_creation_input_tokens + input_tokens -``` + ```text wrap + total_input_tokens = cache_read_input_tokens + cache_creation_input_tokens + input_tokens + ``` -This means when you have cached content, `input_tokens` will typically be much smaller than your total input. For example, with a 200k token cached document and a 50 token user question, you'd see `input_tokens: 50` even though the total input is 200,050 tokens. + This means when you have cached content, `input_tokens` will typically be much smaller than your total input. For example, with a 200k token cached document and a 50 token user question, you'd see `input_tokens: 50` even though the total input is 200,050 tokens. -For rate limit purposes on most models, only `input_tokens` + `cache_creation_input_tokens` count toward your ITPM limit, making [prompt caching](/docs/en/build-with-claude/prompt-caching) an effective way to increase your effective throughput. + For rate limit purposes on most models, only `input_tokens` + `cache_creation_input_tokens` count toward your ITPM limit, making [prompt caching](/docs/en/build-with-claude/prompt-caching) an effective way to increase your effective throughput. -**Example**: With a 2,000,000 ITPM limit and an 80% cache hit rate, you could effectively process 10,000,000 total input tokens per minute (2M uncached + 8M cached), because cached tokens don't count towards your rate limit. +**Example:** With a 2,000,000 ITPM limit and an 80% cache hit rate, you could effectively process 10,000,000 total input tokens per minute (2M uncached + 8M cached), because cached tokens don't count toward your rate limit. -Some older models (marked with † in the following rate limit tables) also count `cache_read_input_tokens` toward ITPM rate limits. + Claude Haiku 3.5 (marked with † in the following rate limit tables) also counts `cache_read_input_tokens` toward ITPM rate limits. -For all models without the † marker, cached input tokens do not count towards rate limits and are billed at a reduced rate (10% of base input token price). This means you can achieve significantly higher effective throughput by using [prompt caching](/docs/en/build-with-claude/prompt-caching). + For all models without the † marker, cached input tokens do not count toward rate limits and are billed at a reduced rate (10% of base input token price). This means you can achieve significantly higher effective throughput by using [prompt caching](/docs/en/build-with-claude/prompt-caching). -**Maximize your rate limits with prompt caching** + **Maximize your rate limits with prompt caching** -To get the most out of your rate limits, use [prompt caching](/docs/en/build-with-claude/prompt-caching) for repeated content like: -- System instructions and prompts -- Large context documents -- Tool definitions -- Conversation history + See [prompt caching](/docs/en/build-with-claude/prompt-caching) for guidance on increasing effective throughput by caching repeated content such as: -With effective caching, you can dramatically increase your actual throughput without increasing your rate limits. Monitor your cache hit rate on the [Usage page](/usage) to optimize your caching strategy. + * System instructions and prompts + * Large context documents + * Tool definitions + * Conversation history + + With effective caching, you can dramatically increase your actual throughput without increasing your rate limits. Monitor your cache hit rate on the [Usage page](/usage) to optimize your caching strategy. OTPM rate limits are evaluated in real time as output tokens are produced, counting only the actual tokens generated. The `max_tokens` parameter does not factor into OTPM rate limit calculations, so there is no rate limit downside to setting a higher `max_tokens` value. -Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously. -You can check your current rate limits and behavior in the [Claude Console](/settings/limits), or read the configured limits programmatically with the [Rate Limits API](/docs/en/manage-claude/rate-limits-api). +Rate limits are applied separately for each model; therefore you can use different models up to their respective limits simultaneously. You can check your current rate limits and behavior on the [Limits](/settings/limits) page in the Claude Console, or read the configured limits programmatically with the [Rate Limits API](/docs/en/manage-claude/rate-limits-api). -Rate limits are currently shared across all `inference_geo` values. Requests with `inference_geo: "us"` and `inference_geo: "global"` draw from the same rate limit pool. + Rate limits are currently shared across all `inference_geo` values. Requests with `inference_geo: "us"` and `inference_geo: "global"` draw from the same rate limit pool. - -| Model | Maximum requests per minute (RPM) | Maximum input tokens per minute (ITPM) | Maximum output tokens per minute (OTPM) | -| -------------------------------------------------------------------------------------------- | --------------------------------- | -------------------------------------- | --------------------------------------- | -| Claude Sonnet 4.x** | 50 | 30,000 | 8,000 | -| Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations)) | 50 | 20,000 | 8,000 | -| Claude Haiku 4.5 | 50 | 50,000 | 10,000 | -| Claude Haiku 3.5 ([deprecated](/docs/en/about-claude/model-deprecations)) | 50 | 50,000 | 10,000 | -| Claude Opus 4.x* | 50 | 500,000 | 80,000 | - - - -| Model | Maximum requests per minute (RPM) | Maximum input tokens per minute (ITPM) | Maximum output tokens per minute (OTPM) | -| -------------------------------------------------------------------------------------------- | --------------------------------- | -------------------------------------- | --------------------------------------- | -| Claude Sonnet 4.x** | 1,000 | 450,000 | 90,000 | -| Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations)) | 1,000 | 40,000 | 16,000 | -| Claude Haiku 4.5 | 1,000 | 450,000 | 90,000 | -| Claude Haiku 3.5 ([deprecated](/docs/en/about-claude/model-deprecations)) | 1,000 | 100,000 | 20,000 | -| Claude Opus 4.x* | 1,000 | 2,000,000 | 200,000 | - - - -| Model | Maximum requests per minute (RPM) | Maximum input tokens per minute (ITPM) | Maximum output tokens per minute (OTPM) | -| -------------------------------------------------------------------------------------------- | --------------------------------- | -------------------------------------- | --------------------------------------- | -| Claude Sonnet 4.x** | 2,000 | 800,000 | 160,000 | -| Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations)) | 2,000 | 80,000 | 32,000 | -| Claude Haiku 4.5 | 2,000 | 1,000,000 | 200,000 | -| Claude Haiku 3.5 ([deprecated](/docs/en/about-claude/model-deprecations)) | 2,000 | 200,000 | 40,000 | -| Claude Opus 4.x* | 2,000 | 5,000,000 | 400,000 | - - - -| Model | Maximum requests per minute (RPM) | Maximum input tokens per minute (ITPM) | Maximum output tokens per minute (OTPM) | -| -------------------------------------------------------------------------------------------- | --------------------------------- | -------------------------------------- | --------------------------------------- | -| Claude Sonnet 4.x** | 4,000 | 2,000,000 | 400,000 | -| Claude Sonnet 3.7 ([deprecated](/docs/en/about-claude/model-deprecations)) | 4,000 | 200,000 | 80,000 | -| Claude Haiku 4.5 | 4,000 | 4,000,000 | 800,000 | -| Claude Haiku 3.5 ([deprecated](/docs/en/about-claude/model-deprecations)) | 4,000 | 400,000 | 80,000 | -| Claude Opus 4.x* | 4,000 | 10,000,000 | 800,000 | - - - -If you're seeking higher limits for an Enterprise use case, contact sales through the [Claude Console](/settings/limits). - + + | Model | Maximum requests per minute (RPM) | Maximum input tokens per minute (ITPM) | Maximum output tokens per minute (OTPM) | + | ---------------------------------------------------------------------------------------------------------- | --------------------------------- | -------------------------------------- | --------------------------------------- | + | Claude Fable 5 | 1,000 | 500,000 | 100,000 | + | Claude Opus 4.x\* | 1,000 | 2,000,000 | 400,000 | + | Claude Sonnet 5 | 1,000 | 2,000,000 | 400,000 | + | Claude Sonnet 4.x\*\* | 1,000 | 2,000,000 | 400,000 | + | Claude Haiku 4.5 | 1,000 | 2,000,000 | 400,000 | + | Claude Haiku 3.5 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | 1,000 | 100,000† | 20,000 | + + + + | Model | Maximum requests per minute (RPM) | Maximum input tokens per minute (ITPM) | Maximum output tokens per minute (OTPM) | + | ---------------------------------------------------------------------------------------------------------- | --------------------------------- | -------------------------------------- | --------------------------------------- | + | Claude Fable 5 | 2,000 | 1,500,000 | 300,000 | + | Claude Opus 4.x\* | 5,000 | 5,000,000 | 1,000,000 | + | Claude Sonnet 5 | 5,000 | 5,000,000 | 1,000,000 | + | Claude Sonnet 4.x\*\* | 5,000 | 5,000,000 | 1,000,000 | + | Claude Haiku 4.5 | 5,000 | 5,000,000 | 1,000,000 | + | Claude Haiku 3.5 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | 2,000 | 200,000† | 40,000 | + + + + | Model | Maximum requests per minute (RPM) | Maximum input tokens per minute (ITPM) | Maximum output tokens per minute (OTPM) | + | ---------------------------------------------------------------------------------------------------------- | --------------------------------- | -------------------------------------- | --------------------------------------- | + | Claude Fable 5 | 4,000 | 4,000,000 | 800,000 | + | Claude Opus 4.x\* | 10,000 | 10,000,000 | 2,000,000 | + | Claude Sonnet 5 | 10,000 | 10,000,000 | 2,000,000 | + | Claude Sonnet 4.x\*\* | 10,000 | 10,000,000 | 2,000,000 | + | Claude Haiku 4.5 | 10,000 | 10,000,000 | 2,000,000 | + | Claude Haiku 3.5 ([retired, except on Bedrock and Google Cloud](/docs/en/about-claude/model-deprecations)) | 4,000 | 400,000† | 80,000 | + + + + If you need limits higher than the Scale tier, contact sales through the [Limits](/settings/limits) page in the Claude Console. + -_* - Opus rate limit is a total limit that applies to combined traffic across Opus 4.7, Opus 4.6, Opus 4.5, Opus 4.1, and Opus 4._ +*\* - Opus rate limit is a total limit that applies to combined traffic across Claude Opus 4.8, Opus 4.7, Opus 4.6, and Opus 4.5.* -_** - Sonnet 4.x rate limit is a total limit that applies to combined traffic across Sonnet 4.6, Sonnet 4.5, and Sonnet 4._ +*\*\* - Sonnet 4.x rate limit is a total limit that applies to combined traffic across Sonnet 4.6 and Sonnet 4.5. Claude Sonnet 5 has a separate rate limit and is not part of this combined bucket.* -_† - Limit counts `cache_read_input_tokens` towards ITPM usage._ +*† - Limit counts `cache_read_input_tokens` toward ITPM usage.* ### Message Batches API -The Message Batches API has its own set of rate limits which are shared across all models. These include a requests per minute (RPM) limit to all API endpoints and a limit on the number of batch requests that can be in the processing queue at the same time. A "batch request" here refers to part of a Message Batch. You may create a Message Batch containing thousands of batch requests, each of which count towards this limit. A batch request is considered part of the processing queue when it has yet to be successfully processed by the model. +The Message Batches API has its own set of rate limits which are shared across all models. These include a requests per minute (RPM) limit to all API endpoints and a limit on the number of batch requests that can be in the processing queue at the same time. A "batch request" here refers to part of a Message Batch. You may create a Message Batch containing thousands of batch requests, each of which count toward this limit. A batch request is considered part of the processing queue when it has yet to be successfully processed by the model. - -| Maximum requests per minute (RPM) | Maximum batch requests in processing queue | Maximum batch requests per batch | -| --------------------------------- | ------------------------------------------ | -------------------------------- | -| 50 | 100,000 | 100,000 | - - -| Maximum requests per minute (RPM) | Maximum batch requests in processing queue | Maximum batch requests per batch | -| --------------------------------- | ------------------------------------------ | -------------------------------- | -| 1,000 | 200,000 | 100,000 | - - -| Maximum requests per minute (RPM) | Maximum batch requests in processing queue | Maximum batch requests per batch | -| --------------------------------- | ------------------------------------------ | -------------------------------- | -| 2,000 | 300,000 | 100,000 | - - -| Maximum requests per minute (RPM) | Maximum batch requests in processing queue | Maximum batch requests per batch | -| --------------------------------- | ------------------------------------------ | -------------------------------- | -| 4,000 | 500,000 | 100,000 | - - -If you're seeking higher limits for an Enterprise use case, contact sales through the [Claude Console](/settings/limits). - + + | Maximum requests per minute (RPM) | Maximum batch requests in processing queue | Maximum batch requests per batch | + | --------------------------------- | ------------------------------------------ | -------------------------------- | + | 1,000 | 200,000 | 100,000 | + + + + | Maximum requests per minute (RPM) | Maximum batch requests in processing queue | Maximum batch requests per batch | + | --------------------------------- | ------------------------------------------ | -------------------------------- | + | 2,000 | 300,000 | 100,000 | + + + + | Maximum requests per minute (RPM) | Maximum batch requests in processing queue | Maximum batch requests per batch | + | --------------------------------- | ------------------------------------------ | -------------------------------- | + | 4,000 | 500,000 | 100,000 | + + + + If you need limits higher than the Scale tier, contact sales through the [Limits](/settings/limits) page in the Claude Console. + ### Managed Agents [Claude Managed Agents](/docs/en/managed-agents/overview) endpoints are rate-limited per organization. These limits are separate from the Messages API rate limits above. -| Operation | Limit | -| --- | --- | -| Create endpoints (for example, agents, sessions, and environments) | 300 requests per minute | -| Read endpoints (for example, retrieve, list, and stream) | 600 requests per minute | +| Operation | Limit | +| ------------------------------------------------------------------ | ------------------------- | +| Create endpoints (for example, agents, sessions, and environments) | 300 requests per minute | +| Read endpoints (for example, retrieve, list, and stream) | 1,200 requests per minute | ### Fast mode rate limits -When using [fast mode](/docs/en/build-with-claude/fast-mode) (beta: research preview) with `speed: "fast"` on Opus 4.6, dedicated rate limits apply that are separate from standard Opus rate limits. When fast mode rate limits are exceeded, the API returns a `429` error with a `retry-after` header. +When using [fast mode](/docs/en/build-with-claude/fast-mode) (research preview) with `speed: "fast"` on Claude Opus 4.8 or Opus 4.7, dedicated rate limits apply that are separate from standard Opus rate limits. When fast mode rate limits are exceeded, the API returns a `429` error with a `retry-after` header. Fast mode is not available on Claude Opus 4.6: requests to `claude-opus-4-6` with `speed: "fast"` run at standard speed. See [Fast mode](/docs/en/build-with-claude/fast-mode#supported-models). -The response includes `anthropic-fast-*` headers that indicate your fast mode rate limit status. See [Fast mode](/docs/en/build-with-claude/fast-mode#rate-limits) for details on these headers. +The response includes `anthropic-fast-*` headers that indicate your fast mode rate limit status. See [Fast mode rate limits](/docs/en/build-with-claude/fast-mode#rate-limits) for details on these headers. ### Monitoring your rate limits in the Console You can monitor your rate limit usage on the [Usage](/usage) page of the [Claude Console](/). -In addition to providing token and request charts, the Usage page provides two separate rate limit charts. Use these charts to see what headroom you have to grow, when you may be hitting peak use, better understand what rate limits to request, or how you can improve your caching rates. The charts visualize a number of metrics for a given rate limit (for example, per model): +In addition to providing token and request charts, the Usage page provides two separate rate limit charts. Use these charts to see what headroom you have to grow, identify when you may be hitting peak use, understand what rate limits to request, and learn how to improve your caching rates. The charts visualize a number of metrics for a given rate limit (for example, per model): + +* The **Rate Limit - Input Tokens** chart includes: + + * Hourly maximum uncached input tokens per minute + * Your current input tokens per minute rate limit + * The cache rate for your input tokens (that is, the percentage of input tokens read from the cache) + +* The **Rate Limit - Output Tokens** chart includes: -- The **Rate Limit - Input Tokens** chart includes: - - Hourly maximum uncached input tokens per minute - - Your current input tokens per minute rate limit - - The cache rate for your input tokens (that is, the percentage of input tokens read from the cache) -- The **Rate Limit - Output Tokens** chart includes: - - Hourly maximum output tokens per minute - - Your current output tokens per minute rate limit + * Hourly maximum output tokens per minute + * Your current output tokens per minute rate limit + +## Requesting higher limits + +To request higher rate limits or a higher monthly spend cap, use **Request rate limit increase** on the [Limits](/settings/limits) page. + + + Support can also raise limits. For urgent needs, contact [Anthropic support](https://support.claude.com). + + + + **[Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws):** The **Request rate limit increase** flow is not available. Contact your Anthropic account representative or [Anthropic support](https://support.claude.com), and include the models you need raised, your peak input and output tokens per minute for each model, and roughly what share of your input is cached or repeated context. See [Rate limits and quotas on Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws#rate-limits-and-quotas). + ## Setting lower limits for Workspaces @@ -297,10 +244,11 @@ To protect Workspaces in your Organization from potential overuse, you can set c Example: If your Organization's limit is 40,000 input tokens per minute and 8,000 output tokens per minute, you might limit one Workspace to 30,000 input tokens per minute. This protects other Workspaces from potential overuse and ensures a more equitable distribution of resources across your Organization. The remaining unused tokens per minute (or more, if that Workspace doesn't use the limit) are then available for other Workspaces to use. Note: -- You can't set limits on the default Workspace. -- If not set, Workspace limits match the Organization's limit. -- Workspace limits are set per limiter type (such as requests per minute, input tokens per minute, or output tokens per minute). -- Organization-wide limits always apply, even if Workspace limits add up to more. + +* You can't set limits on the default Workspace. +* If not set, Workspace limits match the Organization's limit. +* Workspace limits are set per limiter type (such as requests per minute, input tokens per minute, or output tokens per minute). +* Organization-wide limits always apply, even if Workspace limits add up to more. To read your current organization and workspace rate limits programmatically, use the [Rate Limits API](/docs/en/manage-claude/rate-limits-api). @@ -310,26 +258,26 @@ The API response includes headers that show you the rate limit enforced, current The following headers are returned: -| Header | Description | -| --------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------- | -| `retry-after` | The number of seconds to wait until you can retry the request. Earlier retries will fail. | -| `anthropic-ratelimit-requests-limit` | The maximum number of requests allowed within any rate limit period. | -| `anthropic-ratelimit-requests-remaining` | The number of requests remaining before being rate limited. | -| `anthropic-ratelimit-requests-reset` | The time when the request rate limit will be fully replenished, provided in RFC 3339 format. | -| `anthropic-ratelimit-tokens-limit` | The maximum number of tokens allowed within any rate limit period. | -| `anthropic-ratelimit-tokens-remaining` | The number of tokens remaining (rounded to the nearest thousand) before being rate limited. | -| `anthropic-ratelimit-tokens-reset` | The time when the token rate limit will be fully replenished, provided in RFC 3339 format. | -| `anthropic-ratelimit-input-tokens-limit` | The maximum number of input tokens allowed within any rate limit period. | -| `anthropic-ratelimit-input-tokens-remaining` | The number of input tokens remaining (rounded to the nearest thousand) before being rate limited. | -| `anthropic-ratelimit-input-tokens-reset` | The time when the input token rate limit will be fully replenished, provided in RFC 3339 format. | -| `anthropic-ratelimit-output-tokens-limit` | The maximum number of output tokens allowed within any rate limit period. | -| `anthropic-ratelimit-output-tokens-remaining` | The number of output tokens remaining (rounded to the nearest thousand) before being rate limited. | -| `anthropic-ratelimit-output-tokens-reset` | The time when the output token rate limit will be fully replenished, provided in RFC 3339 format. | -| `anthropic-priority-input-tokens-limit` | The maximum number of Priority Tier input tokens allowed within any rate limit period. (Priority Tier only) | -| `anthropic-priority-input-tokens-remaining` | The number of Priority Tier input tokens remaining (rounded to the nearest thousand) before being rate limited. (Priority Tier only) | -| `anthropic-priority-input-tokens-reset` | The time when the Priority Tier input token rate limit will be fully replenished, provided in RFC 3339 format. (Priority Tier only) | -| `anthropic-priority-output-tokens-limit` | The maximum number of Priority Tier output tokens allowed within any rate limit period. (Priority Tier only) | -| `anthropic-priority-output-tokens-remaining` | The number of Priority Tier output tokens remaining (rounded to the nearest thousand) before being rate limited. (Priority Tier only) | -| `anthropic-priority-output-tokens-reset` | The time when the Priority Tier output token rate limit will be fully replenished, provided in RFC 3339 format. (Priority Tier only) | - -The `anthropic-ratelimit-tokens-*` headers display the values for the most restrictive limit currently in effect. For instance, if you have exceeded the Workspace per-minute token limit, the headers will contain the Workspace per-minute token rate limit values. If Workspace limits do not apply, the headers will return the total tokens remaining, where total is the sum of input and output tokens. This approach ensures that you have visibility into the most relevant constraint on your current API usage. \ No newline at end of file +| Header | Description | +| --------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------- | +| `retry-after` | The number of seconds to wait until you can retry the request. Earlier retries will fail. | +| `anthropic-ratelimit-requests-limit` | The maximum number of requests allowed within any rate limit period. | +| `anthropic-ratelimit-requests-remaining` | The number of requests remaining before being rate limited. | +| `anthropic-ratelimit-requests-reset` | The time when the request rate limit will be fully replenished, provided in RFC 3339 format. | +| `anthropic-ratelimit-tokens-limit` | The maximum number of tokens allowed within any rate limit period. | +| `anthropic-ratelimit-tokens-remaining` | The number of tokens remaining (rounded to the nearest thousand) before being rate limited. | +| `anthropic-ratelimit-tokens-reset` | The time when the token rate limit will be fully replenished, provided in RFC 3339 format. | +| `anthropic-ratelimit-input-tokens-limit` | The maximum number of input tokens allowed within any rate limit period. | +| `anthropic-ratelimit-input-tokens-remaining` | The number of input tokens remaining (rounded to the nearest thousand) before being rate limited. | +| `anthropic-ratelimit-input-tokens-reset` | The time when the input token rate limit will be fully replenished, provided in RFC 3339 format. | +| `anthropic-ratelimit-output-tokens-limit` | The maximum number of output tokens allowed within any rate limit period. | +| `anthropic-ratelimit-output-tokens-remaining` | The number of output tokens remaining (rounded to the nearest thousand) before being rate limited. | +| `anthropic-ratelimit-output-tokens-reset` | The time when the output token rate limit will be fully replenished, provided in RFC 3339 format. | +| `anthropic-priority-input-tokens-limit` | The maximum number of Priority Tier input tokens allowed within any rate limit period. (Priority Tier only) | +| `anthropic-priority-input-tokens-remaining` | The number of Priority Tier input tokens remaining (rounded to the nearest thousand) before being rate limited. (Priority Tier only) | +| `anthropic-priority-input-tokens-reset` | The time when the Priority Tier input token rate limit will be fully replenished, provided in RFC 3339 format. (Priority Tier only) | +| `anthropic-priority-output-tokens-limit` | The maximum number of Priority Tier output tokens allowed within any rate limit period. (Priority Tier only) | +| `anthropic-priority-output-tokens-remaining` | The number of Priority Tier output tokens remaining (rounded to the nearest thousand) before being rate limited. (Priority Tier only) | +| `anthropic-priority-output-tokens-reset` | The time when the Priority Tier output token rate limit will be fully replenished, provided in RFC 3339 format. (Priority Tier only) | + +The `anthropic-ratelimit-tokens-*` headers display the values for the most restrictive limit currently in effect. For instance, if you have exceeded the Workspace per-minute token limit, the headers will contain the Workspace per-minute token rate limit values. If Workspace limits do not apply, the headers will return the total tokens remaining, where total is the sum of input and output tokens. This approach ensures that you have visibility into the most relevant constraint on your current API usage. diff --git a/content/en/api/service-tiers.md b/content/en/api/service-tiers.md index 1f9fe647e..f830c28c3 100644 --- a/content/en/api/service-tiers.md +++ b/content/en/api/service-tiers.md @@ -4,14 +4,15 @@ Different tiers of service allow you to balance availability, performance, and p --- + + Priority Tier capacity commitments are no longer available for purchase. Organizations with an existing commitment can continue to use Priority Tier through their contract end date, and this page remains available as a reference for them. If you need guaranteed capacity, [contact sales](https://claude.com/contact-sales). + + Anthropic offers three service tiers: -- **Priority Tier:** Best for workflows deployed in production where time, availability, and predictable pricing are important -- **Standard:** Default tier for both piloting and scaling everyday use cases -- **Batch:** Best for asynchronous workflows that can wait or benefit from being outside your normal capacity - -[Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) supports Standard and Batch service tiers. Priority Tier is not available. - +* **Priority Tier:** Available only to organizations with an existing capacity commitment +* **Standard:** Default tier for both piloting and scaling everyday use cases +* **Batch:** Best for asynchronous workflows that can wait or benefit from being outside your normal capacity ## Standard Tier @@ -21,56 +22,170 @@ The standard tier is the default service tier for all API requests. The API prio The API prioritizes requests in this tier over all other requests. This prioritization helps minimize ["server overloaded" errors](/docs/en/api/errors#http-errors), even during peak times. -For more information, see [Get started with Priority Tier](#get-started-with-priority-tier) +For more information, see [Existing Priority Tier commitments](#existing-priority-tier-commitments). ## How requests get assigned tiers When handling a request, Anthropic decides to assign a request to Priority Tier in the following scenarios: -- Your organization has sufficient priority tier capacity **input** tokens per minute -- Your organization has sufficient priority tier capacity **output** tokens per minute + +* Your organization has sufficient Priority Tier capacity **input** tokens per minute +* Your organization has sufficient Priority Tier capacity **output** tokens per minute Anthropic counts usage against Priority Tier capacity as follows: -**Input Tokens** -- Cache reads as 0.1 tokens per token read from the cache -- Cache writes as 1.25 tokens per token written to the cache with a 5 minute TTL -- Cache writes as 2.00 tokens per token written to the cache with a 1 hour TTL -- For [US-only inference](/docs/en/manage-claude/data-residency) (`inference_geo: "us"`) requests on Claude Opus 4.6, Claude Sonnet 4.6, and later models, input tokens are 1.1 tokens per token -- All other input tokens are 1 token per token +**Input tokens** -**Output Tokens** -- For [US-only inference](/docs/en/manage-claude/data-residency) (`inference_geo: "us"`) requests on Claude Opus 4.6, Claude Sonnet 4.6, and later models, output tokens are 1.1 tokens per token -- All other output tokens are 1 token per token +* Cache reads as 0.1 tokens per token read from the cache +* Cache writes as 1.25 tokens per token written to the cache with a 5 minute TTL +* Cache writes as 2.00 tokens per token written to the cache with a 1 hour TTL +* For [US-only inference](/docs/en/manage-claude/data-residency) (`inference_geo: "us"`) requests on Claude Opus 4.6, Claude Sonnet 4.6, and later models, input tokens are 1.1 tokens per token +* All other input tokens are 1 token per token + +**Output tokens** + +* For [US-only inference](/docs/en/manage-claude/data-residency) (`inference_geo: "us"`) requests on Claude Opus 4.6, Claude Sonnet 4.6, and later models, output tokens are 1.1 tokens per token +* All other output tokens are 1 token per token Otherwise, requests proceed at standard tier. -These burndown rates reflect the relative pricing of each token type. For example, US-only inference is priced at 1.1x on Opus 4.6, Sonnet 4.6, and later models, so each token consumed with `inference_geo: "us"` draws down 1.1 tokens from your Priority Tier capacity. + These burndown rates reflect the relative pricing of each token type. For example, US-only inference is priced at 1.1x on Opus 4.6, Sonnet 4.6, and later models, so each token consumed with `inference_geo: "us"` draws down 1.1 tokens from your Priority Tier capacity. -Requests assigned Priority Tier pull from both the Priority Tier capacity and the regular rate limits. -If servicing the request would exceed the rate limits, the request is declined. + Requests assigned Priority Tier pull from both the Priority Tier capacity and the regular rate limits. If servicing the request would exceed the rate limits, the request is declined. ## Using service tiers You can control which service tiers can be used for a request by setting the `service_tier` parameter: -```python Python -message = client.messages.create( - model="claude-opus-4-7", - max_tokens=1024, - messages=[{"role": "user", "content": "Hello, Claude!"}], - service_tier="auto", # Automatically use Priority Tier when available, fallback to standard -) -print(message.usage.service_tier) -``` + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -H "content-type: application/json" \ + -d '{ + "model": "claude-opus-4-8", + "max_tokens": 1024, + "messages": [{"role": "user", "content": "Hello, Claude!"}], + "service_tier": "auto" + }' + ``` + + ```bash CLI + ant messages create \ + --transform usage.service_tier \ + --raw-output <<'YAML' + model: claude-opus-4-8 + max_tokens: 1024 + messages: + - role: user + content: Hello, Claude! + service_tier: auto # Automatically use Priority Tier when available, fallback to standard + YAML + ``` + + ```python Python + client = anthropic.Anthropic() + + message = client.messages.create( + model="claude-opus-4-8", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude!"}], + service_tier="auto", # Automatically use Priority Tier when available, fallback to standard + ) + print(message.usage.service_tier) + ``` + + ```typescript TypeScript + const client = new Anthropic(); + + const message = await client.messages.create({ + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude!" }], + service_tier: "auto" // Automatically use Priority Tier when available, fallback to standard + }); + console.log(message.usage.service_tier); + ``` + + ```csharp C# + AnthropicClient client = new(); + + var message = await client.Messages.Create(new MessageCreateParams + { + Model = Model.ClaudeOpus4_8, + MaxTokens = 1024, + Messages = [new() { Role = Role.User, Content = "Hello, Claude!" }], + ServiceTier = ServiceTier.Auto, // Automatically use Priority Tier when available, fallback to standard + }); + Console.WriteLine(message.Usage.ServiceTier); + ``` + + ```go Go + client := anthropic.NewClient() + + message, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 1024, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("Hello, Claude!")), + }, + // Automatically use Priority Tier when available, fallback to standard + ServiceTier: anthropic.MessageNewParamsServiceTierAuto, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(message.Usage.ServiceTier) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(1024L) + .addUserMessage("Hello, Claude!") + // Automatically use Priority Tier when available, fallback to standard + .serviceTier(MessageCreateParams.ServiceTier.AUTO) + .build(); + + Message message = client.messages().create(params); + IO.println(message.usage().serviceTier().orElseThrow()); + ``` + + ```php PHP + $client = new Client(); + + $message = $client->messages->create( + model: 'claude-opus-4-8', + maxTokens: 1024, + messages: [['role' => 'user', 'content' => 'Hello, Claude!']], + serviceTier: 'auto', // Automatically use Priority Tier when available, fallback to standard + ); + echo $message->usage->serviceTier; + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.create( + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude!" }], + service_tier: :auto # Automatically use Priority Tier when available, fallback to standard + ) + puts(message.usage.service_tier) + ``` + The `service_tier` parameter accepts the following values: -- `"auto"` (default) - Uses the Priority Tier capacity if available, falling back to your other capacity if not -- `"standard_only"` - Only use standard tier capacity, useful if you don't want to use your Priority Tier capacity +* `"auto"` (default) - Uses the Priority Tier capacity if available, falling back to your other capacity if not +* `"standard_only"` - Only use standard tier capacity, useful if you don't want to use your Priority Tier capacity The response `usage` object also includes the service tier assigned to the request: @@ -85,10 +200,12 @@ The response `usage` object also includes the service tier assigned to the reque } } ``` + This allows you to determine which service tier was assigned to the request. When requesting `service_tier="auto"` with a model with a Priority Tier commitment, these response headers provide insights: -```text + +```text wrap anthropic-priority-input-tokens-limit: 10000 anthropic-priority-input-tokens-remaining: 9618 anthropic-priority-input-tokens-reset: 2025-01-12T23:11:59Z @@ -96,35 +213,22 @@ anthropic-priority-output-tokens-limit: 10000 anthropic-priority-output-tokens-remaining: 6000 anthropic-priority-output-tokens-reset: 2025-01-12T23:12:21Z ``` + You can use the presence of these headers to detect if your request was eligible for Priority Tier, even if it was over the limit. -## Get started with Priority Tier +## Existing Priority Tier commitments -You may want to commit to Priority Tier capacity if you are interested in: -- **Higher availability:** Target 99.5% uptime with prioritized computational resources -- **Cost control:** Predictable spend and discounts for longer commitments -- **Flexible overflow:** Automatically falls back to standard tier when you exceed your committed capacity +A Priority Tier commitment consists of: -Committing to Priority Tier involves deciding: -- A number of input tokens per minute -- A number of output tokens per minute -- A commitment duration (1, 3, 6, or 12 months) -- A specific model version +* A number of input tokens per minute +* A number of output tokens per minute +* A commitment duration (1, 3, 6, or 12 months) +* A specific model version - -The ratio of input to output tokens you purchase matters. Sizing your Priority Tier capacity to align with your actual traffic patterns helps you maximize utilization of your purchased tokens. - +Priority Tier targets 99.5% uptime with prioritized computational resources. Requests beyond your committed capacity automatically fall back to standard tier. ### Supported models -Priority Tier is supported on all available Claude models (including Claude Opus 4.7) except [Claude Mythos Preview](https://anthropic.com/glasswing). +Priority Tier is supported on all available Claude models (including Claude Fable 5 and Claude Opus 4.8) except Claude Sonnet 5, [Claude Mythos Preview](https://anthropic.com/glasswing), and Claude Mythos 5. Check the [Models overview](/docs/en/about-claude/models/overview) for more details on available models. - -### How to access Priority Tier - -To begin using Priority Tier: - -1. [Contact sales](https://claude.com/contact-sales/priority-tier) to complete provisioning. -2. (Optional) Update your API requests to set the `service_tier` parameter to `auto`. -3. Monitor your usage through response headers and the Claude Console. \ No newline at end of file diff --git a/content/en/api/supported-regions.md b/content/en/api/supported-regions.md index 2a7000f4c..a28a69c25 100644 --- a/content/en/api/supported-regions.md +++ b/content/en/api/supported-regions.md @@ -178,4 +178,4 @@ Here are the countries, regions, and territories we can currently support access * Vanuatu * Vietnam * Zambia -* Zimbabwe \ No newline at end of file +* Zimbabwe diff --git a/content/en/api/versioning.md b/content/en/api/versioning.md index 7e328a32f..2453e4b0a 100644 --- a/content/en/api/versioning.md +++ b/content/en/api/versioning.md @@ -1,31 +1,35 @@ # Versions -When making API requests, you must send an `anthropic-version` request header. For example, `anthropic-version: 2023-06-01`. If you are using our [client SDKs](/docs/en/api/client-sdks), this is handled for you automatically. +When making API requests, you must send an `anthropic-version` request header. For example, `anthropic-version: 2023-06-01`. If you are using the [client SDKs](/docs/en/cli-sdks-libraries/overview), this is handled for you automatically. --- -For any given version with the Messages API, we will preserve: +For any given version with the Messages API, Anthropic preserves: * Existing input parameters * Existing output parameters -However, we may do the following: +However, Anthropic may do the following: * Add additional optional inputs * Add additional values to the output * Change conditions for specific error types * Add new variants to enum-like output values (for example, streaming event types) -Generally, if you are using the API as documented in this reference, we will not break your usage. +Generally, if you are using the API as documented in this reference, Anthropic will not break your usage. ## Version history -We always recommend using the latest API version whenever possible. Previous versions are considered deprecated and may be unavailable for new users. +Anthropic recommends using the latest API version whenever possible. Previous versions are considered deprecated and may be unavailable for new users. * `2023-06-01` - * New format for [streaming](/docs/en/build-with-claude/streaming) server-sent events (SSE): - * Completions are incremental. For example, `" Hello"`, `" my"`, `" name"`, `" is"`, `" Claude." ` instead of `" Hello"`, `" Hello my"`, `" Hello my name"`, `" Hello my name is"`, `" Hello my name is Claude."`. - * All events are [named events](https://developer.mozilla.org/en-US/Web/API/Server-sent%5Fevents/Using%5Fserver-sent%5Fevents#named%5Fevents), rather than [data-only events](https://developer.mozilla.org/en-US/Web/API/Server-sent%5Fevents/Using%5Fserver-sent%5Fevents#data-only%5Fmessages). - * Removed unnecessary `data: [DONE]` event. - * Removed legacy `exception` and `truncated` values in responses. -* `2023-01-01`: Initial release. \ No newline at end of file + + * New format for [streaming](/docs/en/build-with-claude/streaming) server-sent events (SSE): + + * Completions are incremental. For example, `" Hello"`, `" my"`, `" name"`, `" is"`, `" Claude." `instead of `" Hello"`, `" Hello my"`, `" Hello my name"`, `" Hello my name is"`, `" Hello my name is Claude."`. + * All events are [named events](https://developer.mozilla.org/en-US/Web/API/Server-sent%5Fevents/Using%5Fserver-sent%5Fevents#named%5Fevents), rather than [data-only events](https://developer.mozilla.org/en-US/Web/API/Server-sent%5Fevents/Using%5Fserver-sent%5Fevents#data-only%5Fmessages). + * Removed unnecessary `data: [DONE]` event. + + * Removed legacy `exception` and `truncated` values in responses. + +* `2023-01-01`: Initial release. diff --git a/content/en/build-with-claude/claude-platform-on-aws.md b/content/en/build-with-claude/claude-platform-on-aws.md index 6b65fdd94..27ffcc798 100644 --- a/content/en/build-with-claude/claude-platform-on-aws.md +++ b/content/en/build-with-claude/claude-platform-on-aws.md @@ -791,25 +791,25 @@ Two Claude Console roles are available: **Admin** and **Developer**. The Admin r The **Through AWS gateway** column indicates whether the page reads and writes data through the AWS gateway (and is therefore governed by [IAM actions](/docs/en/api/claude-platform-on-aws-iam-actions)). Pages marked **No** read organization-level metadata directly from Anthropic and bypass IAM action checks. -| Page | Available | Through AWS gateway | Notes | -| --------------------- | ------------- | ------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| **Usage** | Yes | No | View token usage by model, workspace, and dimension. Data can take a few minutes to appear after a request. | -| **Cost** | Yes | No | View cost breakdowns by model and workspace. AWS Cost Explorer shows the aggregated [Claude Consumption Unit (CCU)](#billing) line item. | -| **Limits** | Yes | No | View rate limits (read-only). Tier increases go through your Anthropic account representative; see [Rate limits and quotas](#rate-limits-and-quotas). | -| **Workspaces** | Yes | No | View per-region workspaces (read-only). | -| **Files** | Yes | Yes | View and manage uploaded files. | -| **Skills** | Yes | Yes | View and manage Agent Skills. | -| **Batches** | Yes | Yes | View and manage batch processing jobs. | -| **Agents** | Yes | Yes | View and manage agent definitions. | -| **Sessions** | Yes | Yes | View agent sessions and event history. | -| **Environments** | Yes | Yes | View and manage cloud sandbox configurations for sessions. | -| **Credential vaults** | Yes | Yes | View and manage credential vaults for session authentication. | -| **Memory stores** | Yes | Yes | View and manage persistent agent memory. | -| **Webhooks** | Yes | Yes | View and manage webhook endpoints under **Settings → Webhooks**. | -| **API keys** | No | N/A | Manage API keys in the AWS Console (**Claude Platform on AWS → API keys**). See [API key authentication](#api-key-authentication). | -| **Members** | No | N/A | Not applicable. AWS IAM manages access. | -| **Billing** | Yes (limited) | No | Set an organization monthly spend limit and spend alerts; see [Spend limits](#spend-limits). AWS Marketplace manages invoicing. View cost breakdowns on the Cost page. | -| **Claude Code** | No | N/A | View Claude Code usage on the Usage page. | +| Page | Available | Through AWS gateway | Notes | +| --------------------- | ------------- | ------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | +| **Usage** | Yes | No | View token usage by model, workspace, and dimension. Data can take a few minutes to appear after a request. | +| **Cost** | Yes | No | View cost breakdowns by model and workspace. AWS Cost Explorer shows the aggregated [Claude Consumption Unit (CCU)](#billing) line item. | +| **Limits** | Yes | No | View rate limits (read-only). Tier increases go through your Anthropic account representative; see [Rate limits and quotas](#rate-limits-and-quotas). | +| **Workspaces** | Yes | No | View per-region workspaces (read-only). | +| **Files** | Yes | Yes | View and manage uploaded files. | +| **Skills** | Yes | Yes | View and manage Agent Skills. | +| **Batches** | Yes | Yes | View and manage batch processing jobs. | +| **Agents** | Yes | Yes | View and manage agent definitions. | +| **Sessions** | Yes | Yes | View agent sessions and event history. | +| **Environments** | Yes | Yes | View and manage cloud sandbox configurations for sessions. | +| **Credential vaults** | Yes | Yes | View and manage credential vaults for session authentication. | +| **Memory stores** | Yes | Yes | View and manage persistent agent memory. | +| **Webhooks** | Yes | Yes | View and manage webhook endpoints under **Settings → Webhooks**. | +| **API keys** | No | N/A | Manage API keys in the AWS Console (**Claude Platform on AWS → API keys**). See [API key authentication](#api-key-authentication). | +| **Members** | No | N/A | Not applicable. AWS IAM manages access. | +| **Billing** | Yes (limited) | No | Set an organization monthly spend limit; see [Spend limits](#spend-limits). AWS Marketplace manages invoicing. View cost breakdowns on the Cost page. | +| **Claude Code** | No | N/A | View Claude Code usage on the Usage page. | ### Switching organizations @@ -841,9 +841,8 @@ The Start, Build, and Scale usage tiers each carry a monthly spend cap; see [the You can also set your own monthly spend limit to cap what your organization spends: -* **Organization spend limit:** Go to [Settings > Billing](/settings/billing) in the [Claude Console](#using-the-claude-console) to set a monthly spend limit and optional spend alerts. On Claude Platform on AWS, spend limits are managed on the Billing page rather than the Limits page. +* **Organization spend limit:** Go to [Settings > Billing](/settings/billing) in the [Claude Console](#using-the-claude-console) to set a monthly spend limit. On Claude Platform on AWS, spend limits are managed on the Billing page rather than the Limits page. * **Workspace spend limits:** Set monthly spend limits for individual workspaces from each workspace's limits settings. -* **Spend alerts:** Alerts are sent to the email addresses you list. Role-based recipients, such as all organization admins, are not supported on Claude Platform on AWS. The spend limits you set are soft limits: spend is calculated at list prices and can take about two hours to reflect recent usage. diff --git a/content/en/build-with-claude/prompt-engineering/claude-prompting-best-practices.md b/content/en/build-with-claude/prompt-engineering/claude-prompting-best-practices.md index 8ebe6a006..832ea3b7c 100644 --- a/content/en/build-with-claude/prompt-engineering/claude-prompting-best-practices.md +++ b/content/en/build-with-claude/prompt-engineering/claude-prompting-best-practices.md @@ -4,171 +4,32 @@ Comprehensive guide to prompt engineering techniques for Claude's latest models, --- -This is the single reference for prompt engineering with Claude's latest models, including Claude Opus 4.7, Claude Opus 4.6, Claude Sonnet 4.6, and Claude Haiku 4.5. It covers foundational techniques, output control, tool use, thinking, and agentic systems. Jump to the section that matches your situation. +This is the reference for prompt engineering with Claude's latest models, including Claude Fable 5, Claude Mythos 5, Claude Opus 4.8, Claude Opus 4.7, Claude Opus 4.6, Claude Sonnet 5, Claude Sonnet 4.6, and Claude Haiku 4.5. The page is organized in three parts: + +* **Model-specific guidance** first: where [Claude Fable 5](/docs/en/build-with-claude/prompt-engineering/prompting-claude-fable-5), [Claude Sonnet 5](/docs/en/build-with-claude/prompt-engineering/prompting-claude-sonnet-5), and [Claude Opus 4.8](/docs/en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8) behave differently and what to change. +* **Techniques for all current models** after that: general principles, output and formatting, tool use, thinking, and agentic systems. +* **Migration considerations** last, for prompts moving from earlier generations. - For an overview of model capabilities, see the [models overview](/docs/en/about-claude/models/overview). For details on what's new in Claude Opus 4.7, see [What's new in Claude Opus 4.7](/docs/en/about-claude/models/whats-new-claude-4-7). For migration guidance, see the [Migration guide](/docs/en/about-claude/models/migration-guide). + For an overview of model capabilities, see the [models overview](/docs/en/about-claude/models/overview). For Claude Fable 5 capabilities and API changes, see [Introducing Claude Fable 5 and Claude Mythos 5](/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5). For details on what's new in Claude Sonnet 5, see [What's new in Claude Sonnet 5](/docs/en/about-claude/models/whats-new-sonnet-5). For details on what's new in Claude Opus 4.8, see [What's new in Claude Opus 4.8](/docs/en/about-claude/models/whats-new-claude-4-8). For migration guidance, see the [Migration guide](/docs/en/about-claude/models/migration-guide). -## Prompting Claude Opus 4.7 - -Claude Opus 4.7 is our most capable generally available model, with particular strengths in long-horizon agentic work, knowledge work, vision, and memory tasks. It performs well out of the box on existing Claude Opus 4.6 prompts. The patterns below cover the behaviors that most often require tuning. - - -For API parameter changes when migrating from Claude Opus 4.6 (effort levels, task budgets, thinking configuration, sampling-parameter removal, and tokenization), see the [migration guide](/docs/en/about-claude/models/migration-guide#migrating-to-claude-opus-4-7). - - -### Response length and verbosity - -Claude Opus 4.7 calibrates response length to how complex it judges the task to be, rather than defaulting to a fixed verbosity. This usually means shorter answers on simple lookups and much longer ones on open-ended analysis. - -If your product depends on a certain style or verbosity of output, you may need to tune your prompts. As an example, to decrease verbosity, you might add: - -```text -Provide concise, focused responses. Skip non-essential context, and keep examples minimal. -``` - -If you see specific examples of kinds of verbosity (i.e. over-explaining), you can add additional instructions in your prompt to prevent them. Positive examples showing how Claude can communicate with the appropriate level of concision tend to be more effective than negative examples or instructions that tell the model what not to do. - -### Calibrating effort and thinking depth - -The [effort parameter](/docs/en/build-with-claude/effort) allows you to tune Claude's intelligence vs. token spend, trading off capability for faster speed and lower costs. Start with the new `xhigh` effort level for coding and agentic use cases, and use a minimum of `high` effort for most intelligence-sensitive use cases. Experiment with other effort levels to further tune token usage and intelligence: - -- **`max`:** Max effort can deliver performance gains in some use cases, but may show diminishing returns from increased token usage. This setting can also sometimes be prone to overthinking. We recommend testing max effort for intelligence-demanding tasks. -- **`xhigh` (new):** Extra high effort is the best setting for most coding and agentic use cases. -- **`high`:** This setting balances token usage and intelligence. For most intelligence-sensitive use cases, we recommend a minimum of `high` effort. -- **`medium`:** Good for cost-sensitive use cases that need to reduce token usage while trading off intelligence. -- **`low`:** Reserve for short, scoped tasks and latency-sensitive workloads that are not intelligence-sensitive. - -Meaningfully changing from Claude Opus 4.6, Claude Opus 4.7 respects effort levels strictly, especially at the low end. At `low` and `medium`, the model scopes its work to what was asked rather than going above and beyond. This is good for latency and cost, but on moderately complex tasks running at `low` effort there is some risk of under-thinking. - -If you observe shallow reasoning on complex problems, raise effort to `high` or `xhigh` rather than prompting around it. If you need to keep effort at `low` for latency, add targeted guidance: - -```text -This task involves multi-step reasoning. Think carefully through the problem before responding. -``` - -We expect effort to be more important for this model than for any prior Opus, and recommend experimenting with it actively when you upgrade. - -The triggering behavior for adaptive thinking is steerable. If you find the model thinking more often than you'd like — which can happen with large or complex system prompts — add guidance to steer it. As always, measure the effect of any prompting changes on performance. Example: - -```text -Thinking adds latency and should only be used when it will meaningfully improve answer quality — typically for problems that require multi-step reasoning. When in doubt, respond directly. -``` - -Conversely, if you're running hard workloads at `medium` and seeing under-thinking, the first lever is to raise effort. If you need finer control, prompt for it directly. - - -If you are running Claude Opus 4.7 at `max` or `xhigh` effort, set a large max output token budget so the model has room to think and act across its subagents and tool calls. We recommend starting at 64k tokens and tuning from there. - - -### Tool use triggering - -Claude Opus 4.7 has a tendency to use tools less often than Claude Opus 4.6 and to use reasoning more. This produces better results in most cases. However, increasing the effort setting is a useful lever to increase the level of tool usage, especially in knowledge work. `high` or `xhigh` effort settings show substantially more tool usage in agentic search and coding. For scenarios where you want more tool use, you can also adjust your prompt to explicitly instruct the model about when and how to properly use its tools. For instance, if you find that the model is not using your web search tools, clearly describe why and how it should. - -### User-facing progress updates - -Claude Opus 4.7 provides more regular, higher-quality updates to the user throughout long agentic traces. If you've added scaffolding to force interim status messages ("After every 3 tool calls, summarize progress"), try removing it. If you find that the length or contents of Claude Opus 4.7's user-facing updates are not well-calibrated to your use case, explicitly describe what these updates should look like in the prompt and provide examples. - -### More literal instruction following - -Claude Opus 4.7 interprets prompts more literally and explicitly than Claude Opus 4.6, particularly at lower effort levels. It will not silently generalize an instruction from one item to another, and it will not infer requests you didn't make. The upside of this literalism is precision and less thrash, and it generally performs better for API use cases with carefully tuned prompts, structured extraction, and pipelines where you want predictable behavior. If you need Claude to apply an instruction broadly, state the scope explicitly (for example, "Apply this formatting to every section, not just the first one"). - -### Tone and writing style - -As with any new model, prose style on long-form writing may shift. Claude Opus 4.7 is more direct and opinionated, with less validation-forward phrasing and fewer emoji than Claude Opus 4.6's warmer style. If your product relies on a specific voice, re-evaluate style prompts against the new baseline. - -For instance, if your product voice is warmer or more conversational, add: - -```text -Use a warm, collaborative tone. Acknowledge the user's framing before answering. -``` - -### Controlling subagent spawning - -Claude Opus 4.7 tends to spawn fewer subagents by default. However, this behavior is steerable through prompting; give Claude Opus 4.7 explicit guidance around when subagents are desirable. A toy example for a coding use case: - -```text -Do not spawn a subagent for work you can complete directly in a single response (e.g. refactoring a function you can already see). - -Spawn multiple subagents in the same turn when fanning out across items or reading multiple files. -``` - -### Design and frontend defaults - -Claude Opus 4.7 has stronger design instincts than Claude Opus 4.6, with a consistent default house style: warm cream/off-white backgrounds (~`#F4F1EA`), serif display type (Georgia, Fraunces, Playfair), italic word-accents, and a terracotta/amber accent. This reads well for editorial, hospitality, and portfolio briefs, but will feel off for dashboards, dev tools, fintech, healthcare, or enterprise apps — and it appears in slide decks as well as web UIs. - -This default is persistent. Generic instructions ("don't use cream," "make it clean and minimal") tend to shift the model to a different fixed palette rather than producing variety. Two approaches work reliably: - -**1. Specify a concrete alternative.** The model follows explicit specs precisely: - -```text -Design a desktop landing page for a supplement brand called AEFRM. - -The visual direction should come from a cold monochrome atmosphere using pale silver-gray tones that gradually deepen into blue-gray and near-black, similar to a misted metallic surface. - -The page should feel sharp and controlled, with a strong sense of structure and restraint. - -Use this tonal system across the full page instead of introducing bright accent colors. - -Use the uploaded image on the hero design in black and white. - -The layout should be built with clear horizontal sections and a centered max-width container. Use 4px corner radius consistently across cards, buttons, inputs, and media frames. Margins should feel generous, with enough empty space around each section so the page breathes. - -Typography should use a square, angular sans-serif with wider letter spacing than usual, especially in headings and navigation, so the text feels more engineered and less compressed. Headline text can be large and uppercase, while supporting copy remains short and sparse. The sub texts should be written with Alumni Sans SC in 4-6px like tiny little texts on corners bottom centre like that. - -For the structure, start with a hero section containing a strong product statement, one short supporting paragraph, and a clean product placeholder or packshot frame. Below that, add a benefit grid with three or four blocks, then a formulation or ingredients section, and finally a cta. - -Buttons should be flat and precise, with subtle hover changes using transition: all 160ms ease out where brightness and border contrast shift slightly rather than using dramatic motion. - -Color palette should stay within this range: -#E9ECEC, #C9D2D4, #8C9A9E, #44545B, #11171B. -``` - -**2. Have the model propose options before building.** This breaks the default and gives users control. If you previously relied on `temperature` for design variety, use this approach — it produces meaningfully different directions across runs. Example prompt: - -```text -Before building, propose 4 distinct visual directions tailored to this brief (each as: bg hex / accent hex / typeface — one-line rationale). Ask the user to pick one, then implement only that direction. -``` - -Additionally, Claude Opus 4.7 requires less frontend design prompting than previous models to avoid generic patterns that users call the "AI slop" aesthetic. With earlier models, we recommended a lengthier prompt snippet in our [frontend-design skill](https://github.com/anthropics/claude-code/blob/main/plugins/frontend-design/skills/frontend-design/SKILL.md). However, Claude Opus 4.7 generates distinctive, creative frontends with more minimal prompting guidance. This prompt snippet works well with the above prompting advice for variety: - -```text - -NEVER use generic AI-generated aesthetics like overused font families (Inter, Roboto, Arial, system fonts), cliched color schemes (particularly purple gradients on white or dark backgrounds), predictable layouts and component patterns, and cookie-cutter design that lacks context-specific character. Use unique fonts, cohesive colors and themes, and animations for effects and micro-interactions. - -``` - -### Interactive coding products - -Claude Opus 4.7's token usage and behavior can differ between autonomous, asynchronous coding agents with a single user turn and interactive, synchronous coding agents with multiple user turns. Specifically, it tends to use more tokens in interactive settings, primarily because it reasons more after user turns. This can improve long-horizon coherence, instruction following, and coding capabilities in long, interactive coding sessions, but also comes with more token usage. To maximize both performance and token efficiency in coding products, we recommend using `xhigh` or `high` effort, adding autonomous features like an auto mode, and reducing the number of human interactions required from your users. - -Of course, when limiting the number of required user interactions, it's important to specify the task, intent, and relevant constraints upfront in the first human turn. Providing well-specified, clear, and accurate task descriptions upfront can help maximize autonomy and intelligence while minimizing extra token usage after user turns. We find that because Claude Opus 4.7 is more autonomous than prior models, this usage pattern helps to maximize performance. In contrast, ambiguous or underspecified prompts conveyed progressively over multiple user turns tend to relatively reduce token efficiency and sometimes performance. - -### Code review harnesses - -Claude Opus 4.7 is meaningfully better at finding bugs than prior models, and has both higher recall and precision in our evals — 11pp better recall in one of our hardest bug-finding evals based on real Anthropic PRs. However, if your code-review harness was tuned for an earlier model, you may initially see lower recall. This is likely a harness effect, not a capability regression. When a review prompt says things like "only report high-severity issues," "be conservative," or "don't nitpick," Claude Opus 4.7 may follow that instruction more faithfully than earlier models did — it may investigate the code just as thoroughly, identify the bugs, and then not report findings it judges to be below your stated bar. This can show up as the model doing the same depth of investigation but converting fewer investigations into reported findings, especially on lower-severity bugs. Precision typically rises, but measured recall can fall even though the model's underlying bug-finding ability has improved. +## Claude Fable 5 -Some recommended prompt language: +Prompting guidance for Claude Fable 5 and Claude Mythos 5 has its own page: [Prompting Claude Fable 5](/docs/en/build-with-claude/prompt-engineering/prompting-claude-fable-5). It covers the behavioral differences from Claude Opus 4.8 and the prompt and scaffolding changes worth making, including effort levels, instruction following, long-run progress claims, memory systems, and the `reasoning_extraction` refusal category. -```text -Report every issue you find, including ones you are uncertain about or consider low-severity. Do not filter for importance or confidence at this stage - a separate verification step will do that. Your goal here is coverage: it is better to surface a finding that later gets filtered out than to silently drop a real bug. For each finding, include your confidence level and an estimated severity so a downstream filter can rank them. -``` - -This prompt can be used without having an actual second step, but moving confidence filtering out of the finding step often helps. If your harness has a separate verification, deduplication, or ranking stage, tell the model explicitly that its job at the finding stage is coverage rather than filtering. - -If you do want the model to self-filter in a single pass, be concrete about where the bar is rather than using qualitative terms like "important" — for example, "report any bugs that could cause incorrect behavior, a test failure, or a misleading result; only omit nits like pure style or naming preferences." +## Claude Sonnet 5 -We recommend iterating on prompts against a subset of your evals or test cases to validate recall or F1 score gains. +Prompting guidance for Claude Sonnet 5 has its own page: [Prompting Claude Sonnet 5](/docs/en/build-with-claude/prompt-engineering/prompting-claude-sonnet-5). It covers the behavioral differences from Claude Sonnet 4.6 and the prompt changes worth making, including response length, effort and thinking-depth calibration, tool use triggering, literal instruction following, and design and frontend defaults. -### Computer use +## Prompting Claude Opus 4.8 -[Computer use](/docs/en/agents-and-tools/tool-use/computer-use-tool) capability works across resolutions, up to a new maximum resolution of 2576px / 3.75MP. In our computer use testing, we find that sending images at 1080p provides a good balance of performance and cost. - -For particularly cost-sensitive workloads, we recommend 720p or 1366×768 as lower-cost options with strong performance. We recommend that you conduct your own testing to find the ideal settings for your use case; experimenting with effort settings can also help tune the model's behavior. +Prompting guidance for Claude Opus 4.8 has its own page: [Prompting Claude Opus 4.8](/docs/en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8). It covers response length, effort and thinking-depth calibration, tool use triggering, literal instruction following, subagent control, and design and frontend defaults. ## General principles +The techniques in this section and the sections that follow apply to all current Claude models, including Claude Fable 5 and Claude Mythos 5. + ### Be clear and direct Claude responds well to clear, explicit instructions. Being specific about your desired output can help enhance results. If you want "above and beyond" behavior, explicitly request it rather than relying on the model to infer this from vague prompts. @@ -177,94 +38,217 @@ Think of Claude as a brilliant but new employee who lacks context on your norms **Golden rule:** Show your prompt to a colleague with minimal context on the task and ask them to follow it. If they'd be confused, Claude will be too. -- Be specific about the desired output format and constraints. -- Provide instructions as sequential steps using numbered lists or bullet points when the order or completeness of steps matters. +* Be specific about the desired output format and constraints. +* Provide instructions as sequential steps using numbered lists or bullet points when the order or completeness of steps matters. -
+ + **Less effective:** -**Less effective:** -```text -Create an analytics dashboard -``` + ```text wrap + Create an analytics dashboard + ``` -**More effective:** -```text -Create an analytics dashboard. Include as many relevant features and interactions as possible. Go beyond the basics to create a fully-featured implementation. -``` + **More effective:** -
+ ```text wrap + Create an analytics dashboard. Include as many relevant features and interactions as possible. Go beyond the basics to create a fully-featured implementation. + ``` + ### Add context to improve performance Providing context or motivation behind your instructions, such as explaining to Claude why such behavior is important, can help Claude better understand your goals and deliver more targeted responses. -
+ + **Less effective:** -**Less effective:** -```text -NEVER use ellipses -``` + ```text wrap + NEVER use ellipses + ``` -**More effective:** -```text -Your response will be read aloud by a text-to-speech engine, so never use ellipses since the text-to-speech engine will not know how to pronounce them. -``` + **More effective:** -
+ ```text wrap + Your response will be read aloud by a text-to-speech engine, so never use ellipses since the text-to-speech engine will not know how to pronounce them. + ``` + Claude is smart enough to generalize from the explanation. ### Use examples effectively -Examples are one of the most reliable ways to steer Claude's output format, tone, and structure. A few well-crafted examples (known as few-shot or multishot prompting) can dramatically improve accuracy and consistency. +Examples are one of the most reliable ways to steer Claude's output format, tone, and structure. A few well-crafted examples (known as few-shot or multishot prompting) improve accuracy and consistency. When adding examples, make them: -- **Relevant:** Mirror your actual use case closely. -- **Diverse:** Cover edge cases and vary enough that Claude doesn't pick up unintended patterns. -- **Structured:** Wrap examples in `` tags (multiple examples in `` tags) so Claude can distinguish them from instructions. -Include 3–5 examples for best results. You can also ask Claude to evaluate your examples for relevance and diversity, or to generate additional ones based on your initial set. +* **Relevant:** Mirror your actual use case closely. +* **Diverse:** Cover edge cases and vary enough that Claude doesn't pick up unintended patterns. +* **Structured:** Wrap examples in `` tags (multiple examples in `` tags) so Claude can distinguish them from instructions. + + + Include 3–5 examples for best results. You can also ask Claude to evaluate your examples for relevance and diversity, or to generate additional ones based on your initial set. + ### Structure prompts with XML tags -XML tags help Claude parse complex prompts unambiguously, especially when your prompt mixes instructions, context, examples, and variable inputs. Wrapping each type of content in its own tag (e.g. ``, ``, ``) reduces misinterpretation. +XML tags help Claude parse complex prompts unambiguously, especially when your prompt mixes instructions, context, examples, and variable inputs. Wrapping each type of content in its own tag (for example, ``, ``, ``) reduces misinterpretation. Best practices: -- Use consistent, descriptive tag names across your prompts. -- Nest tags when content has a natural hierarchy (documents inside ``, each inside ``). + +* Use consistent, descriptive tag names across your prompts. +* Nest tags when content has a natural hierarchy (documents inside ``, each inside ``). ### Give Claude a role Setting a role in the system prompt focuses Claude's behavior and tone for your use case. Even a single sentence makes a difference: -```python Python -import anthropic - -client = anthropic.Anthropic() - -message = client.messages.create( - model="claude-opus-4-7", - max_tokens=1024, - system="You are a helpful coding assistant specializing in Python.", - messages=[ + + ```bash cURL + curl https://api.anthropic.com/v1/messages \ + -H "content-type: application/json" \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -d '{ + "model": "claude-opus-4-8", + "max_tokens": 1024, + "system": "You are a helpful coding assistant specializing in Python.", + "messages": [ {"role": "user", "content": "How do I sort a list of dictionaries by key?"} - ], -) -print(message.content) -``` + ] + }' + ``` + + ```bash CLI + ant messages create \ + --model claude-opus-4-8 \ + --max-tokens 1024 \ + --system "You are a helpful coding assistant specializing in Python." \ + --message '{role: user, content: "How do I sort a list of dictionaries by key?"}' + ``` + + ```python Python + client = anthropic.Anthropic() + + message = client.messages.create( + model="claude-opus-4-8", + max_tokens=1024, + system="You are a helpful coding assistant specializing in Python.", + messages=[ + {"role": "user", "content": "How do I sort a list of dictionaries by key?"} + ], + ) + + print(message.content) + ``` + + ```typescript TypeScript + const client = new Anthropic(); + + const message = await client.messages.create({ + model: "claude-opus-4-8", + max_tokens: 1024, + system: "You are a helpful coding assistant specializing in Python.", + messages: [{ role: "user", content: "How do I sort a list of dictionaries by key?" }] + }); + + console.log(message.content); + ``` + + ```csharp C# + AnthropicClient client = new(); + + var parameters = new MessageCreateParams + { + Model = Model.ClaudeOpus4_8, + MaxTokens = 1024, + System = "You are a helpful coding assistant specializing in Python.", + Messages = + [ + new() { Role = Role.User, Content = "How do I sort a list of dictionaries by key?" } + ] + }; + + var message = await client.Messages.Create(parameters); + Console.WriteLine(message); + ``` + + ```go Go + client := anthropic.NewClient() + + message, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 1024, + System: []anthropic.TextBlockParam{ + {Text: "You are a helpful coding assistant specializing in Python."}, + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("How do I sort a list of dictionaries by key?")), + }, + }) + if err != nil { + log.Fatal(err) + } + fmt.Println(message.Content) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + MessageCreateParams params = MessageCreateParams.builder() + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(1024) + .system("You are a helpful coding assistant specializing in Python.") + .addUserMessage("How do I sort a list of dictionaries by key?") + .build(); + + Message message = client.messages().create(params); + System.out.println(message.content()); + ``` + + ```php PHP + $client = new Client(); + + $message = $client->messages->create( + maxTokens: 1024, + messages: [ + ['role' => 'user', 'content' => 'How do I sort a list of dictionaries by key?'] + ], + model: 'claude-opus-4-8', + system: 'You are a helpful coding assistant specializing in Python.', + ); + + echo $message->content[0]->text; + ``` + + ```ruby Ruby + client = Anthropic::Client.new + + message = client.messages.create( + model: "claude-opus-4-8", + max_tokens: 1024, + system: "You are a helpful coding assistant specializing in Python.", + messages: [ + { role: "user", content: "How do I sort a list of dictionaries by key?" } + ] + ) + + puts message.content + ``` + ### Long context prompting When working with large documents or data-rich inputs (20k+ tokens), structure your prompt carefully to get the best results: -- **Put longform data at the top:** Place your long documents and inputs near the top of your prompt, above your query, instructions, and examples. This can significantly improve performance across all models. - - Queries at the end can improve response quality by up to 30% in tests, especially with complex, multi-document inputs. +* **Put longform data at the top:** Place your long documents and inputs near the top of your prompt, above your query, instructions, and examples. This improves performance across all models. -- **Structure document content and metadata with XML tags:** When using multiple documents, wrap each document in `` tags with `` and `` (and other metadata) subtags for clarity. + + Queries at the end can improve response quality by up to 30 percent in tests, especially with complex, multidocument inputs. + -
+* **Structure document content and metadata with XML tags:** When using multiple documents, wrap each document in `` tags with `` and `` (and other metadata) subtags for clarity. + ```xml @@ -283,13 +267,11 @@ When working with large documents or data-rich inputs (20k+ tokens), structure y Analyze the annual report and competitor analysis. Identify strategic advantages and recommend Q3 focus areas. ``` - -
+ -- **Ground responses in quotes:** For long document tasks, ask Claude to quote relevant parts of the documents first before carrying out its task. This helps Claude cut through the noise of the rest of the document's contents. - -
+* **Ground responses in quotes:** For long document tasks, ask Claude to quote relevant parts of the documents first before carrying out its task. This helps Claude focus on the relevant content and ignore the rest of the document. + ```xml You are an AI physician's assistant. Your task is to help doctors diagnose possible patient illnesses. @@ -316,21 +298,21 @@ When working with large documents or data-rich inputs (20k+ tokens), structure y Find quotes from the patient records and appointment history that are relevant to diagnosing the patient's reported symptoms. Place these in tags. Then, based on these quotes, list all information that would help the doctor diagnose the patient's symptoms. Place your diagnostic information in tags. ``` - -
+ ### Model self-knowledge If you would like Claude to identify itself correctly in your application or use specific API strings: -```text Sample prompt for model identity -The assistant is Claude, created by Anthropic. The current model is Claude Opus 4.7. +```text Sample prompt for model identity wrap +The assistant is Claude, created by Anthropic. The current model is Claude Opus 4.8. ``` For LLM-powered apps that need to specify model strings: -```text Sample prompt for model string -When an LLM is needed, please default to Claude Opus 4.7 unless the user requests otherwise. The exact model string for Claude Opus 4.7 is claude-opus-4-7. +```text Sample prompt for model string wrap +When an LLM is needed, please default to Claude Opus 4.8 unless the user requests +otherwise. The exact model string for Claude Opus 4.8 is claude-opus-4-8. ``` ## Output and formatting @@ -339,13 +321,13 @@ When an LLM is needed, please default to Claude Opus 4.7 unless the user request Claude's latest models have a more concise and natural communication style compared to previous models: -- **More direct and grounded:** Provides fact-based progress reports rather than self-celebratory updates -- **More conversational:** Slightly more fluent and colloquial, less machine-like -- **Less verbose:** May skip detailed summaries for efficiency unless prompted otherwise +* **More direct and grounded:** Provides fact-based progress reports rather than self-celebratory updates +* **More conversational:** Slightly more fluent and colloquial, less machine-like +* **Less verbose:** May skip detailed summaries for efficiency unless prompted otherwise This means Claude may skip verbal summaries after tool calls, jumping directly to the next action. If you prefer more visibility into its reasoning: -```text Sample prompt +```text Sample prompt wrap After completing a task that involves tool use, provide a quick summary of the work you've done. ``` @@ -355,12 +337,12 @@ There are a few particularly effective ways to steer output formatting: 1. **Tell Claude what to do instead of what not to do** - - Instead of: "Do not use markdown in your response" - - Try: "Your response should be composed of smoothly flowing prose paragraphs." + * Instead of: "Do not use markdown in your response" + * Try: "Your response should be composed of smoothly flowing prose paragraphs." 2. **Use XML format indicators** - - Try: "Write the prose sections of your response in \ tags." + * Try: "Write the prose sections of your response in \ tags." 3. **Match your prompt style to the desired output** @@ -370,122 +352,132 @@ There are a few particularly effective ways to steer output formatting: For more control over markdown and formatting usage, provide explicit guidance: -```text Sample prompt to minimize markdown +````text Sample prompt to minimize markdown wrap -When writing reports, documents, technical explanations, analyses, or any long-form content, write in clear, flowing prose using complete paragraphs and sentences. Use standard paragraph breaks for organization and reserve markdown primarily for `inline code`, code blocks (```...```), and simple headings (###, and ###). Avoid using **bold** and *italics*. - -DO NOT use ordered lists (1. ...) or unordered lists (*) unless : a) you're presenting truly discrete items where a list format is the best option, or b) the user explicitly requests a list or ranking - -Instead of listing items with bullets or numbers, incorporate them naturally into sentences. This guidance applies especially to technical writing. Using prose instead of excessive formatting will improve user satisfaction. NEVER output a series of overly short bullet points. - -Your goal is readable, flowing text that guides the reader naturally through ideas rather than fragmenting information into isolated points. +When writing reports, documents, technical explanations, analyses, or any long-form +content, write in clear, flowing prose using complete paragraphs and sentences. Use +standard paragraph breaks for organization and reserve markdown primarily for `inline +code`, code blocks (```...```), and simple headings (## and ###). Avoid using **bold** +and *italics*. + +DO NOT use ordered lists (1. ...) or unordered lists (*) unless: a) you're presenting +truly discrete items where a list format is the best option, or b) the user explicitly +requests a list or ranking + +Instead of listing items with bullets or numbers, incorporate them naturally into +sentences. This guidance applies especially to technical writing. Using prose instead of +excessive formatting will improve user satisfaction. NEVER output a series of overly +short bullet points. + +Your goal is readable, flowing text that guides the reader naturally through ideas +rather than fragmenting information into isolated points. -``` +```` ### LaTeX output Claude's latest models default to LaTeX for mathematical expressions, equations, and technical explanations. If you prefer plain text, add the following instructions to your prompt: -```text Sample prompt -Format your response in plain text only. Do not use LaTeX, MathJax, or any markup notation such as \( \), $, or \frac{}{}. Write all math expressions using standard text characters (e.g., "/" for division, "*" for multiplication, and "^" for exponents). +```text Sample prompt wrap +Format your response in plain text only. Do not use LaTeX, MathJax, or any markup +notation such as \( \), $, or \frac{}{}. Write all math expressions using standard text +characters (e.g., "/" for division, "*" for multiplication, and "^" for exponents). ``` ### Document creation -Claude's latest models excel at creating presentations, animations, and visual documents with impressive creative flair and strong instruction following. The models produce polished, usable output on the first try in most cases. +Claude's latest models create presentations, animations, and visual documents with strong instruction following, and usually produce usable output on the first try. For best results with document creation: -```text Sample prompt -Create a professional presentation on [topic]. Include thoughtful design elements, visual hierarchy, and engaging animations where appropriate. +```text Sample prompt wrap +Create a professional presentation on [topic]. Include thoughtful design elements, +visual hierarchy, and engaging animations where appropriate. ``` ### Migrating away from prefilled responses -Starting with Claude 4.6 models and [Claude Mythos Preview](https://anthropic.com/glasswing), prefilled responses on the last assistant turn are no longer supported. Requests with prefilled assistant messages to these models return a 400 error. Model intelligence and instruction following have advanced such that most use cases of prefill no longer require it. Earlier models continue to support prefills, and adding assistant messages elsewhere in the conversation is not affected. +Starting with Claude 4.6 models and [Claude Mythos Preview](https://anthropic.com/glasswing), prefilled responses (providing a partial assistant message for Claude to continue from) on the last assistant turn are no longer supported. Requests with prefilled assistant messages to these models return a 400 error. Model intelligence and instruction following have advanced such that most use cases of prefill no longer require it. Earlier models continue to support prefills, and adding assistant messages elsewhere in the conversation is not affected. Here are common prefill scenarios and how to migrate away from them: -
- -Prefills have been used to force specific output formats like JSON/YAML, classification, and similar patterns where the prefill constrains Claude to a particular structure. - -**Migration:** The [Structured Outputs](/docs/en/build-with-claude/structured-outputs) feature is designed specifically to constrain Claude's responses to follow a given schema. Try simply asking the model to conform to your output structure first, as newer models can reliably match complex schemas when told to, especially if implemented with retries. For classification tasks, use either tools with an enum field containing your valid labels or structured outputs. - -
- -
+ + Prefills have been used to force specific output formats like JSON/YAML, classification, and similar patterns where the prefill constrains Claude to a particular structure. -Prefills like `Here is the requested summary:\n` were used to skip introductory text. + **Migration:** The [Structured Outputs](/docs/en/build-with-claude/structured-outputs) feature is designed specifically to constrain Claude's responses to follow a given schema. Try asking the model to conform to your output structure first, as newer models can reliably match complex schemas when told to, especially if implemented with retries. For classification tasks, use either tools with an enum field containing your valid labels or structured outputs. + -**Migration:** Use direct instructions in the system prompt: "Respond directly without preamble. Do not start with phrases like 'Here is...', 'Based on...', etc." Alternatively, direct the model to output within XML tags, use structured outputs, or use tool calling. If the occasional preamble slips through, strip it in post-processing. + + Prefills like `Here is the requested summary:\n` were used to skip introductory text. -
+ **Migration:** Use direct instructions in the system prompt: "Respond directly without preamble. Do not start with phrases like 'Here is...', 'Based on...', etc." Alternatively, direct the model to output within XML tags, use structured outputs, or use tool calling. If the occasional preamble slips through, strip it in post-processing. + -
+ + Prefills were used to steer around unnecessary refusals. -Prefills were used to steer around unnecessary refusals. + **Migration:** Claude is much better at appropriate refusals now. Clear prompting within the `user` message without prefill should be sufficient. + -**Migration:** Claude is much better at appropriate refusals now. Clear prompting within the `user` message without prefill should be sufficient. + + Prefills were used to continue partial completions, resume interrupted responses, or pick up where a previous generation left off. -
+ **Migration:** Move the continuation to the user message, and include the final text from the interrupted response: "Your previous response was interrupted and ended with \`\[previous\_response]\`. Continue from where you left off." If this is part of error-handling or incomplete-response-handling and there is no UX penalty, retry the request. + -
+ + Prefills were used to periodically ensure refreshed or injected context. -Prefills were used to continue partial completions, resume interrupted responses, or pick up where a previous generation left off. - -**Migration:** Move the continuation to the user message, and include the final text from the interrupted response: "Your previous response was interrupted and ended with \`[previous_response]\`. Continue from where you left off." If this is part of error-handling or incomplete-response-handling and there is no UX penalty, retry the request. - -
- -
- -Prefills were used to periodically ensure refreshed or injected context. - -**Migration:** For very long conversations, inject what were previously prefilled-assistant reminders into the user turn. If context hydration is part of a more complex agentic system, consider hydrating via tools (expose or encourage use of tools containing context based on heuristics such as number of turns) or during context compaction. - -
+ **Migration:** For very long conversations, inject what were previously prefilled-assistant reminders into the user turn. If context hydration is part of a more complex agentic system, consider hydrating through tools (expose or encourage use of tools containing context based on heuristics such as number of turns) or during [context compaction](/docs/en/build-with-claude/compaction). + ## Tool use ### Tool usage -Claude's latest models are trained for precise instruction following and benefit from explicit direction to use specific tools. If you say "can you suggest some changes," Claude will sometimes provide suggestions rather than implementing them, even if making changes might be what you intended. +Claude's latest models are trained for precise instruction following and benefit from explicit direction to use specific tools. If you say "can you suggest some changes," Claude will sometimes provide suggestions rather than implementing them, even if making changes might be what you intended. For how to define tools and troubleshoot tool triggering, see [Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview). For Claude to take action, be more explicit: -
+ + **Less effective (Claude will only suggest):** -**Less effective (Claude will only suggest):** -```text -Can you suggest some changes to improve this function? -``` + ```text wrap + Can you suggest some changes to improve this function? + ``` -**More effective (Claude will make the changes):** -```text -Change this function to improve its performance. -``` + **More effective (Claude will make the changes):** -Or: -```text -Make these edits to the authentication flow. -``` + ```text wrap + Change this function to improve its performance. + ``` + + Or: -
+ ```text wrap + Make these edits to the authentication flow. + ``` + To make Claude more proactive about taking action by default, you can add this to your system prompt: -```text Sample prompt for proactive action +```text Sample prompt for proactive action wrap -By default, implement changes rather than only suggesting them. If the user's intent is unclear, infer the most useful likely action and proceed, using tools to discover any missing details instead of guessing. Try to infer the user's intent about whether a tool call (e.g., file edit or read) is intended or not, and act accordingly. +By default, implement changes rather than only suggesting them. If the user's intent is +unclear, infer the most useful likely action and proceed, using tools to discover any +missing details instead of guessing. Try to infer the user's intent about whether a tool +call (e.g., file edit or read) is intended or not, and act accordingly. ``` -On the other hand, if you want the model to be more hesitant by default, less prone to jumping straight into implementations, and only take action if requested, you can steer this behavior with a prompt like the below: +On the other hand, if you want the model to be more hesitant by default, less prone to jumping straight into implementations, and only take action if requested, you can steer this behavior with a prompt like the following: -```text Sample prompt for conservative action +```text Sample prompt for conservative action wrap -Do not jump into implementation or change files unless clearly instructed to make changes. When the user's intent is ambiguous, default to providing information, doing research, and providing recommendations rather than taking action. Only proceed with edits, modifications, or implementations when the user explicitly requests them. +Do not jump into implementation or change files unless clearly instructed to make +changes. When the user's intent is ambiguous, default to providing information, doing +research, and providing recommendations rather than taking action. Only proceed with +edits, modifications, or implementations when the user explicitly requests them. ``` @@ -493,21 +485,29 @@ Claude Opus 4.5 and Claude Opus 4.6 are also more responsive to the system promp ### Optimize parallel tool calling -Claude's latest models excel at parallel tool execution. These models will: +Claude's latest models run independent tool calls in parallel. These models will: -- Run multiple speculative searches during research -- Read several files at once to build context faster -- Execute bash commands in parallel (which can even bottleneck system performance) +* Run multiple speculative searches during research +* Read several files at once to build context faster +* Run bash commands in parallel (which can even bottleneck system performance) -This behavior is easily steerable. While the model has a high success rate in parallel tool calling without prompting, you can boost this to ~100% or adjust the aggression level: +This behavior is steerable. While the model has a high success rate in parallel tool calling without prompting, you can boost this to \~100% or adjust the aggression level: -```text Sample prompt for maximum parallel efficiency +```text Sample prompt for maximum parallel efficiency wrap -If you intend to call multiple tools and there are no dependencies between the tool calls, make all of the independent tool calls in parallel. Prioritize calling tools simultaneously whenever the actions can be done in parallel rather than sequentially. For example, when reading 3 files, run 3 tool calls in parallel to read all 3 files into context at the same time. Maximize use of parallel tool calls where possible to increase speed and efficiency. However, if some tool calls depend on previous calls to inform dependent values like the parameters, do NOT call these tools in parallel and instead call them sequentially. Never use placeholders or guess missing parameters in tool calls. +If you intend to call multiple tools and there are no dependencies between the tool +calls, make all of the independent tool calls in parallel. Prioritize calling tools +simultaneously whenever the actions can be done in parallel rather than sequentially. +For example, when reading 3 files, run 3 tool calls in parallel to read all 3 files into +context at the same time. Maximize use of parallel tool calls where possible to increase +speed and efficiency. However, if some tool calls depend on previous calls to inform +dependent values like the parameters, do NOT call these tools in parallel and instead +call them sequentially. Never use placeholders or guess missing parameters in tool +calls. ``` -```text Sample prompt to reduce parallel execution +```text Sample prompt to reduce parallel execution wrap Execute operations sequentially with brief pauses between each step to ensure stability. ``` @@ -515,73 +515,266 @@ Execute operations sequentially with brief pauses between each step to ensure st ### Overthinking and excessive thoroughness -Claude Opus 4.6 does significantly more upfront exploration than previous models, especially at higher `effort` settings. This initial work often helps to optimize the final results, but the model may gather extensive context or pursue multiple threads of research without being prompted. If your prompts previously encouraged the model to be more thorough, you should tune that guidance for Claude Opus 4.6: +Claude Opus 4.6 does more upfront exploration than previous models, especially at higher [`effort`](/docs/en/build-with-claude/effort) settings. This initial work often helps to optimize the final results, but the model may gather extensive context or pursue multiple threads of research without being prompted. If your prompts previously encouraged the model to be more thorough, you should tune that guidance for Claude Opus 4.6: -- **Replace blanket defaults with more targeted instructions.** Instead of "Default to using \[tool\]," add guidance like "Use \[tool\] when it would enhance your understanding of the problem." -- **Remove over-prompting.** Tools that undertriggered in previous models are likely to trigger appropriately now. Instructions like "If in doubt, use \[tool\]" will cause overtriggering. -- **Use effort as a fallback.** If Claude continues to be overly aggressive, use a lower setting for `effort`. +* **Replace blanket defaults with more targeted instructions.** Instead of "Default to using \[tool]," add guidance like "Use \[tool] when it would enhance your understanding of the problem." +* **Remove over-prompting.** Tools that undertriggered in previous models are likely to trigger appropriately now. Instructions like "If in doubt, use \[tool]" will cause overtriggering. +* **Use effort as a fallback.** If Claude continues to be overly aggressive, use a lower setting for `effort`. In some cases, Claude Opus 4.6 may think extensively, which can inflate thinking tokens and slow down responses. If this behavior is undesirable, you can add explicit instructions to constrain its reasoning, or you can lower the `effort` setting to reduce overall thinking and token usage. -```text Sample prompt -When you're deciding how to approach a problem, choose an approach and commit to it. Avoid revisiting decisions unless you encounter new information that directly contradicts your reasoning. If you're weighing two approaches, pick one and see it through. You can always course-correct later if the chosen approach fails. +```text Sample prompt wrap +When you're deciding how to approach a problem, choose an approach and commit to it. +Avoid revisiting decisions unless you encounter new information that directly +contradicts your reasoning. If you're weighing two approaches, pick one and see it +through. You can always course-correct later if the chosen approach fails. ``` -If you need a hard ceiling on thinking costs, extended thinking with a `budget_tokens` cap is still functional on Opus 4.6 and Sonnet 4.6 but is deprecated. Prefer lowering the [effort](/docs/en/build-with-claude/effort) setting or using `max_tokens` as a hard limit with [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). +If you need a hard ceiling on thinking costs, extended thinking with a `budget_tokens` cap is still functional on Opus 4.6 and Sonnet 4.6 but is deprecated. On Claude Opus 4.7 and later models, and on Claude Fable 5 and Claude Mythos 5, setting `budget_tokens` returns a 400 error. Prefer lowering the [effort](/docs/en/build-with-claude/effort) setting or using `max_tokens` as a hard limit with [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). ### Leverage thinking & interleaved thinking capabilities -Claude's latest models offer thinking capabilities that can be especially helpful for tasks involving reflection after tool use or complex multi-step reasoning. You can guide its initial or interleaved thinking for better results. +Claude's latest models offer thinking capabilities that can be especially helpful for tasks involving reflection after tool use or complex multistep reasoning. You can guide its initial or interleaved thinking for better results. -Claude Opus 4.6 and Claude Sonnet 4.6 use [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) (`thinking: {type: "adaptive"}`), where Claude dynamically decides when and how much to think. Claude calibrates its thinking based on two factors: the `effort` parameter and query complexity. Higher effort elicits more thinking, and more complex queries do the same. On easier queries that don't require thinking, the model responds directly. In internal evaluations, adaptive thinking reliably drives better performance than extended thinking. Consider moving to adaptive thinking to get the most intelligent responses. +Claude Opus 4.6, Claude Opus 4.7, Claude Opus 4.8, and Claude Sonnet 4.6 use [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) (`thinking: {type: "adaptive"}`), where Claude dynamically decides when and how much to think. On Claude Fable 5 and Claude Mythos 5, thinking is always on and adaptive thinking is the only mode. Claude calibrates its thinking based on two factors: the `effort` parameter and query complexity. Higher effort elicits more thinking, and more complex queries do the same. On easier queries that don't require thinking, the model responds directly. In internal evaluations, adaptive thinking reliably drives better performance than extended thinking. Consider moving to adaptive thinking to get the most intelligent responses. -Use adaptive thinking for workloads that require agentic behavior such as multi-step tool use, complex coding tasks, and long-horizon agent loops. Older models use manual thinking mode with `budget_tokens`. +Use adaptive thinking for workloads that require agentic behavior such as multistep tool use, complex coding tasks, and long-horizon agent loops. Older models use manual [extended thinking](/docs/en/build-with-claude/extended-thinking) with `budget_tokens`; see the [supported models table](/docs/en/build-with-claude/extended-thinking#supported-models) for which mode each model accepts. You can guide Claude's thinking behavior: -```text Example prompt -After receiving tool results, carefully reflect on their quality and determine optimal next steps before proceeding. Use your thinking to plan and iterate based on this new information, and then take the best next action. +```text Example prompt wrap +After receiving tool results, carefully reflect on their quality and determine optimal +next steps before proceeding. Use your thinking to plan and iterate based on this new +information, and then take the best next action. ``` The triggering behavior for adaptive thinking is promptable. If you find the model thinking more often than you'd like, which can happen with large or complex system prompts, add guidance to steer it: -```text Sample prompt -Extended thinking adds latency and should only be used when it will meaningfully improve answer quality - typically for problems that require multi-step reasoning. When in doubt, respond directly. -``` - -If you are migrating from [extended thinking](/docs/en/build-with-claude/extended-thinking) with `budget_tokens`, replace your thinking configuration and move budget control to `effort`: - -**Before (extended thinking, older models):** - -```python Python nocheck -client.messages.create( - model="claude-sonnet-4-5-20250929", - max_tokens=64000, - thinking={"type": "enabled", "budget_tokens": 32000}, - messages=[{"role": "user", "content": "..."}], -) -``` - -**After (adaptive thinking):** - -```python Python nocheck -client.messages.create( - model="claude-opus-4-7", - max_tokens=64000, - thinking={"type": "adaptive"}, - output_config={"effort": "high"}, # or "max", "xhigh", "medium", "low" - messages=[{"role": "user", "content": "..."}], -) +```text Sample prompt wrap +Extended thinking adds latency and should only be used when it will meaningfully improve +answer quality - typically for problems that require multistep reasoning. When in +doubt, respond directly. ``` -If you are not using extended thinking, no changes are required. Thinking is off by default when you omit the `thinking` parameter. +If you are migrating from [extended thinking](/docs/en/build-with-claude/extended-thinking) with `budget_tokens`, replace your thinking configuration and move budget control to `effort`. The following examples show the same request before and after the migration (see [effort](/docs/en/build-with-claude/effort) for the available levels and per-model availability): + + + ```bash cURL + # Before: extended thinking with a manual budget (older models) + curl https://api.anthropic.com/v1/messages \ + -H "content-type: application/json" \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -d '{ + "model": "claude-sonnet-4-5-20250929", + "max_tokens": 16000, + "thinking": {"type": "enabled", "budget_tokens": 10000}, + "messages": [ + {"role": "user", "content": "..."} + ] + }' + + # After: adaptive thinking with effort (current models) + curl https://api.anthropic.com/v1/messages \ + -H "content-type: application/json" \ + -H "x-api-key: $ANTHROPIC_API_KEY" \ + -H "anthropic-version: 2023-06-01" \ + -d '{ + "model": "claude-opus-4-8", + "max_tokens": 16000, + "thinking": {"type": "adaptive"}, + "output_config": {"effort": "high"}, + "messages": [ + {"role": "user", "content": "..."} + ] + }' + ``` + + ```bash CLI + # Before: extended thinking with a manual budget (older models) + ant messages create <<'YAML' + model: claude-sonnet-4-5-20250929 + max_tokens: 16000 + thinking: + type: enabled + budget_tokens: 10000 + messages: + - role: user + content: "..." + YAML + + # After: adaptive thinking with effort (current models) + ant messages create <<'YAML' + model: claude-opus-4-8 + max_tokens: 16000 + thinking: + type: adaptive + output_config: + effort: high + messages: + - role: user + content: "..." + YAML + ``` + + ```python Python + # Before: extended thinking with a manual budget (older models) + client.messages.create( + model="claude-sonnet-4-5-20250929", + max_tokens=16000, + thinking={"type": "enabled", "budget_tokens": 10000}, + messages=[{"role": "user", "content": "..."}], + ) + + # After: adaptive thinking with effort (current models) + client.messages.create( + model="claude-opus-4-8", + max_tokens=16000, + thinking={"type": "adaptive"}, + output_config={"effort": "high"}, + messages=[{"role": "user", "content": "..."}], + ) + ``` + + ```typescript TypeScript + // Before: extended thinking with a manual budget (older models) + await client.messages.create({ + model: "claude-sonnet-4-5-20250929", + max_tokens: 16000, + thinking: { type: "enabled", budget_tokens: 10000 }, + messages: [{ role: "user", content: "..." }] + }); + + // After: adaptive thinking with effort (current models) + await client.messages.create({ + model: "claude-opus-4-8", + max_tokens: 16000, + thinking: { type: "adaptive" }, + output_config: { effort: "high" }, + messages: [{ role: "user", content: "..." }] + }); + ``` + + ```csharp C# + // Before: extended thinking with a manual budget (older models) + await client.Messages.Create(new MessageCreateParams + { + Model = "claude-sonnet-4-5-20250929", + MaxTokens = 16000, + Thinking = new ThinkingConfigEnabled(budgetTokens: 10000), + Messages = [new() { Role = Role.User, Content = "..." }] + }); + + // After: adaptive thinking with effort (current models) + await client.Messages.Create(new MessageCreateParams + { + Model = Model.ClaudeOpus4_8, + MaxTokens = 16000, + Thinking = new ThinkingConfigAdaptive(), + OutputConfig = new OutputConfig { Effort = Effort.High }, + Messages = [new() { Role = Role.User, Content = "..." }] + }); + ``` + + ```go Go + // Before: extended thinking with a manual budget (older models) + client.Messages.New(ctx, anthropic.MessageNewParams{ + Model: "claude-sonnet-4-5-20250929", + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamUnion{ + OfEnabled: &anthropic.ThinkingConfigEnabledParam{BudgetTokens: 10000}, + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("...")), + }, + }) + + // After: adaptive thinking with effort (current models) + client.Messages.New(ctx, anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 16000, + Thinking: anthropic.ThinkingConfigParamUnion{ + OfAdaptive: &anthropic.ThinkingConfigAdaptiveParam{}, + }, + OutputConfig: anthropic.OutputConfigParam{ + Effort: anthropic.OutputConfigEffortHigh, + }, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("...")), + }, + }) + ``` + + ```java Java + // Before: extended thinking with a manual budget (older models) + client.messages().create(MessageCreateParams.builder() + .model("claude-sonnet-4-5-20250929") + .maxTokens(16000L) + .thinking(ThinkingConfigEnabled.builder().budgetTokens(10000L).build()) + .addUserMessage("...") + .build()); + + // After: adaptive thinking with effort (current models) + client.messages().create(MessageCreateParams.builder() + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(16000L) + .thinking(ThinkingConfigAdaptive.builder().build()) + .outputConfig(OutputConfig.builder() + .effort(OutputConfig.Effort.HIGH) + .build()) + .addUserMessage("...") + .build()); + ``` + + ```php PHP + // Before: extended thinking with a manual budget (older models) + $client->messages->create( + model: 'claude-sonnet-4-5-20250929', + maxTokens: 16000, + thinking: ['type' => 'enabled', 'budget_tokens' => 10000], + messages: [['role' => 'user', 'content' => '...']], + ); + + // After: adaptive thinking with effort (current models) + $client->messages->create( + model: 'claude-opus-4-8', + maxTokens: 16000, + thinking: ['type' => 'adaptive'], + outputConfig: ['effort' => 'high'], + messages: [['role' => 'user', 'content' => '...']], + ); + ``` + + ```ruby Ruby + # Before: extended thinking with a manual budget (older models) + client.messages.create( + model: "claude-sonnet-4-5-20250929", + max_tokens: 16000, + thinking: { type: "enabled", budget_tokens: 10000 }, + messages: [{ role: "user", content: "..." }] + ) + + # After: adaptive thinking with effort (current models) + client.messages.create( + model: "claude-opus-4-8", + max_tokens: 16000, + thinking: { type: "adaptive" }, + output_config: { effort: "high" }, + messages: [{ role: "user", content: "..." }] + ) + ``` + + +If you are not using extended thinking, no changes are required. On Claude Opus 4.6 through Claude Opus 4.8 and Claude Sonnet 4.6, thinking is off when you omit the `thinking` parameter. On Claude Fable 5 and Claude Mythos 5, thinking is always on, regardless of whether you set the `thinking` parameter. + +* **Prefer general instructions over prescriptive steps.** A prompt like "think thoroughly" often produces better reasoning than a hand-written step-by-step plan. Claude's reasoning frequently exceeds what a human would prescribe. +* **Multishot examples work with thinking.** Use `` tags inside your few-shot examples to show Claude the reasoning pattern. It will generalize that style to its own extended thinking blocks. +* **Manual chain-of-thought (CoT) prompting as a fallback.** When thinking is off, you can still encourage step-by-step reasoning by asking Claude to think through the problem. Use structured tags like `` and `` to cleanly separate reasoning from the final output. +* **Ask Claude to self-check.** Append something like "Before you finish, verify your answer against \[test criteria]." This catches errors reliably, especially for coding and math. -- **Prefer general instructions over prescriptive steps.** A prompt like "think thoroughly" often produces better reasoning than a hand-written step-by-step plan. Claude's reasoning frequently exceeds what a human would prescribe. -- **Multishot examples work with thinking.** Use `` tags inside your few-shot examples to show Claude the reasoning pattern. It will generalize that style to its own extended thinking blocks. -- **Manual CoT as a fallback.** When thinking is off, you can still encourage step-by-step reasoning by asking Claude to think through the problem. Use structured tags like `` and `` to cleanly separate reasoning from the final output. -- **Ask Claude to self-check.** Append something like "Before you finish, verify your answer against [test criteria]." This catches errors reliably, especially for coding and math. - -When extended thinking is disabled, Claude Opus 4.5 is particularly sensitive to the word "think" and its variants. Consider using alternatives like "consider," "evaluate," or "reason through" in those cases. + + When extended thinking is disabled, Claude Opus 4.5 is particularly sensitive to the word "think" and its variants. Consider using alternatives like "consider," "evaluate," or "reason through" in those cases. + For more information on thinking capabilities, see [Extended thinking](/docs/en/build-with-claude/extended-thinking) and [Adaptive thinking](/docs/en/build-with-claude/adaptive-thinking). @@ -591,141 +784,163 @@ If you are not using extended thinking, no changes are required. Thinking is off ### Long-horizon reasoning and state tracking -Claude's latest models excel at long-horizon reasoning tasks with exceptional state tracking capabilities. Claude maintains orientation across extended sessions by focusing on incremental progress, making steady advances on a few things at a time rather than attempting everything at once. This capability especially emerges over multiple context windows or task iterations, where Claude can work on a complex task, save the state, and continue with a fresh context window. +Claude's latest models handle long-horizon reasoning tasks with strong state tracking. Claude maintains orientation across extended sessions by focusing on incremental progress, making steady advances on a few things at a time rather than attempting everything at once. This capability especially emerges over multiple context windows or task iterations, where Claude can work on a complex task, save the state, and continue with a fresh context window. -#### Context awareness and multi-window workflows +#### Context awareness and multiwindow workflows -Claude 4.6 and Claude 4.5 models feature [context awareness](/docs/en/build-with-claude/context-windows#context-awareness-in-claude-sonnet-4-6-sonnet-4-5-and-haiku-4-5), enabling the model to track its remaining context window (i.e. "token budget") throughout a conversation. This enables Claude to execute tasks and manage context more effectively by understanding how much space it has to work. +Claude Sonnet 5, Claude Sonnet 4.6, Claude Sonnet 4.5, and Claude Haiku 4.5 feature [context awareness](/docs/en/build-with-claude/context-windows#context-awareness), enabling the model to track its remaining context window (that is, its "token budget") throughout a conversation. This enables Claude to execute tasks and manage context more effectively by understanding how much space it has to work. **Managing context limits:** -If you are using Claude in an agent harness that compacts context or allows saving context to external files (like in Claude Code), consider adding this information to your prompt so Claude can behave accordingly. Otherwise, Claude may sometimes naturally try to wrap up work as it approaches the context limit. Below is an example prompt: +If you are using Claude in an agent harness that compacts context or allows saving context to external files (like in Claude Code), consider adding this information to your prompt so Claude can behave accordingly. Otherwise, Claude may sometimes naturally try to wrap up work as it approaches the context limit. The following is an example prompt: -```text Sample prompt -Your context window will be automatically compacted as it approaches its limit, allowing you to continue working indefinitely from where you left off. Therefore, do not stop tasks early due to token budget concerns. As you approach your token budget limit, save your current progress and state to memory before the context window refreshes. Always be as persistent and autonomous as possible and complete tasks fully, even if the end of your budget is approaching. Never artificially stop any task early regardless of the context remaining. +```text Sample prompt wrap +Your context window will be automatically compacted as it approaches its limit, allowing +you to continue working indefinitely from where you left off. Therefore, do not stop +tasks early due to token budget concerns. As you approach your token budget limit, save +your current progress and state to memory before the context window refreshes. Always be +as persistent and autonomous as possible and complete tasks fully, even if the end of +your budget is approaching. Never artificially stop any task early regardless of the +context remaining. ``` -The [memory tool](/docs/en/agents-and-tools/tool-use/memory-tool) pairs naturally with context awareness for seamless context transitions. +The [memory tool](/docs/en/agents-and-tools/tool-use/memory-tool) pairs well with context awareness for managing context transitions. -#### Multi-context window workflows +#### Workflows across multiple context windows For tasks spanning multiple context windows: 1. **Use a different prompt for the very first context window:** Use the first context window to set up a framework (write tests, create setup scripts), then use future context windows to iterate on a todo-list. -2. **Have the model write tests in a structured format:** Ask Claude to create tests before starting work and keep track of them in a structured format (e.g., `tests.json`). This leads to better long-term ability to iterate. Remind Claude of the importance of tests: "It is unacceptable to remove or edit tests because this could lead to missing or buggy functionality." +2. **Have the model write tests in a structured format:** Ask Claude to create tests before starting work and keep track of them in a structured format (for example, `tests.json`). This leads to better long-term ability to iterate. Remind Claude of the importance of tests: "It is unacceptable to remove or edit tests because this could lead to missing or buggy functionality." -3. **Set up quality of life tools:** Encourage Claude to create setup scripts (e.g., `init.sh`) to gracefully start servers, run test suites, and linters. This prevents repeated work when continuing from a fresh context window. +3. **Set up quality of life tools:** Encourage Claude to create setup scripts (for example, `init.sh`) to gracefully start servers, run test suites, and linters. This prevents repeated work when continuing from a fresh context window. -4. **Starting fresh vs compacting:** When a context window is cleared, consider starting with a brand new context window rather than using compaction. Claude's latest models are extremely effective at discovering state from the local filesystem. In some cases, you may want to take advantage of this over compaction. Be prescriptive about how it should start: - - "Call pwd; you can only read and write files in this directory." - - "Review progress.txt, tests.json, and the git logs." - - "Manually run through a fundamental integration test before moving on to implementing new features." +4. **Starting fresh versus compacting:** When a context window is cleared, consider starting with a brand new context window rather than using compaction. Claude's latest models are extremely effective at discovering state from the local filesystem. In some cases, you may want to take advantage of this over compaction. Be prescriptive about how it should start: + + * "Call pwd; you can only read and write files in this directory." + * "Review progress.txt, tests.json, and the git logs." + * "Manually run through a fundamental integration test before moving on to implementing new features." 5. **Provide verification tools:** As the length of autonomous tasks grows, Claude needs to verify correctness without continuous human feedback. Tools like Playwright MCP server or computer use capabilities for testing UIs are helpful. 6. **Encourage complete usage of context:** Prompt Claude to efficiently complete components before moving on: -```text Sample prompt -This is a very long task, so it may be beneficial to plan out your work clearly. It's encouraged to spend your entire output context working on the task - just make sure you don't run out of context with significant uncommitted work. Continue working systematically until you have completed this task. +```text Sample prompt wrap +This is a very long task, so it may be beneficial to plan out your work clearly. It's +encouraged to spend your entire output context working on the task - just make sure you +don't run out of context with significant uncommitted work. Continue working +systematically until you have completed this task. ``` #### State management best practices -- **Use structured formats for state data:** When tracking structured information (like test results or task status), use JSON or other structured formats to help Claude understand schema requirements -- **Use unstructured text for progress notes:** Freeform progress notes work well for tracking general progress and context -- **Use git for state tracking:** Git provides a log of what's been done and checkpoints that can be restored. Claude's latest models perform especially well in using git to track state across multiple sessions. -- **Emphasize incremental progress:** Explicitly ask Claude to keep track of its progress and focus on incremental work - -
- -```json -// Structured state file (tests.json) -{ - "tests": [ - { "id": 1, "name": "authentication_flow", "status": "passing" }, - { "id": 2, "name": "user_management", "status": "failing" }, - { "id": 3, "name": "api_endpoints", "status": "not_started" } - ], - "total": 200, - "passing": 150, - "failing": 25, - "not_started": 25 -} -``` - -```text -// Progress notes (progress.txt) -Session 3 progress: -- Fixed authentication token validation -- Updated user model to handle edge cases -- Next: investigate user_management test failures (test #2) -- Note: Do not remove tests as this could lead to missing functionality -``` - -
+* **Use structured formats for state data:** When tracking structured information (like test results or task status), use JSON or other structured formats to help Claude understand schema requirements. +* **Use unstructured text for progress notes:** Freeform progress notes work well for tracking general progress and context. +* **Use git for state tracking:** Git provides a log of what's been done and checkpoints that can be restored. Claude's latest models perform especially well in using git to track state across multiple sessions. +* **Emphasize incremental progress:** Explicitly ask Claude to keep track of its progress and focus on incremental work. + + + ```json + // Structured state file (tests.json) + { + "tests": [ + { "id": 1, "name": "authentication_flow", "status": "passing" }, + { "id": 2, "name": "user_management", "status": "failing" }, + { "id": 3, "name": "api_endpoints", "status": "not_started" } + ], + "total": 200, + "passing": 150, + "failing": 25, + "not_started": 25 + } + ``` + + ```text wrap + // Progress notes (progress.txt) + Session 3 progress: + - Fixed authentication token validation + - Updated user model to handle edge cases + - Next: investigate user_management test failures (test #2) + - Note: Do not remove tests as this could lead to missing functionality + ``` + ### Balancing autonomy and safety Without guidance, Claude Opus 4.6 may take actions that are difficult to reverse or affect shared systems, such as deleting files, force-pushing, or posting to external services. If you want Claude Opus 4.6 to confirm before taking potentially risky actions, add guidance to your prompt: -```text Sample prompt -Consider the reversibility and potential impact of your actions. You are encouraged to take local, reversible actions like editing files or running tests, but for actions that are hard to reverse, affect shared systems, or could be destructive, ask the user before proceeding. +```text Sample prompt wrap +Consider the reversibility and potential impact of your actions. You are encouraged to +take local, reversible actions like editing files or running tests, but for actions that +are hard to reverse, affect shared systems, or could be destructive, ask the user before +proceeding. Examples of actions that warrant confirmation: - Destructive operations: deleting files or branches, dropping database tables, rm -rf - Hard to reverse operations: git push --force, git reset --hard, amending published commits -- Operations visible to others: pushing code, commenting on PRs/issues, sending messages, modifying shared infrastructure +- Operations visible to others: pushing code, commenting on PRs/issues, sending +messages, modifying shared infrastructure -When encountering obstacles, do not use destructive actions as a shortcut. For example, don't bypass safety checks (e.g. --no-verify) or discard unfamiliar files that may be in-progress work. +When encountering obstacles, do not use destructive actions as a shortcut. For example, +don't bypass safety checks (e.g. --no-verify) or discard unfamiliar files that may be +in-progress work. ``` ### Research and information gathering -Claude's latest models demonstrate exceptional agentic search capabilities and can find and synthesize information from multiple sources effectively. For optimal research results: +Claude's latest models can find and synthesize information from multiple sources effectively. For optimal research results: -1. **Provide clear success criteria:** Define what constitutes a successful answer to your research question +1. **Provide clear success criteria:** Define what constitutes a successful answer to your research question. -2. **Encourage source verification:** Ask Claude to verify information across multiple sources +2. **Encourage source verification:** Ask Claude to verify information across multiple sources. 3. **For complex research tasks, use a structured approach:** -```text Sample prompt for complex research -Search for this information in a structured way. As you gather data, develop several competing hypotheses. Track your confidence levels in your progress notes to improve calibration. Regularly self-critique your approach and plan. Update a hypothesis tree or research notes file to persist information and provide transparency. Break down this complex research task systematically. +```text Sample prompt for complex research wrap +Search for this information in a structured way. As you gather data, develop several +competing hypotheses. Track your confidence levels in your progress notes to improve +calibration. Regularly self-critique your approach and plan. Update a hypothesis tree or +research notes file to persist information and provide transparency. Break down this +complex research task systematically. ``` -This structured approach allows Claude to find and synthesize virtually any piece of information and iteratively critique its findings, no matter the size of the corpus. +This structured approach helps Claude work through large corpora methodically and iteratively critique its findings. ### Subagent orchestration -Claude's latest models demonstrate significantly improved native subagent orchestration capabilities. These models can recognize when tasks would benefit from delegating work to specialized subagents and do so proactively without requiring explicit instruction. +Claude's latest models orchestrate subagents natively. These models can recognize when tasks would benefit from delegating work to specialized subagents and do so proactively without requiring explicit instruction. To take advantage of this behavior: -1. **Ensure well-defined subagent tools:** Have subagent tools available and described in tool definitions -2. **Let Claude orchestrate naturally:** Claude will delegate appropriately without explicit instruction +1. **Ensure well-defined subagent tools:** Have subagent tools available and described in tool definitions. +2. **Let Claude orchestrate naturally:** Claude will delegate appropriately without explicit instruction. 3. **Watch for overuse:** Claude Opus 4.6 has a strong predilection for subagents and may spawn them in situations where a simpler, direct approach would suffice. For example, the model may spawn subagents for code exploration when a direct grep call is faster and sufficient. If you're seeing excessive subagent use, add explicit guidance about when subagents are and aren't warranted: -```text Sample prompt for subagent usage -Use subagents when tasks can run in parallel, require isolated context, or involve independent workstreams that don't need to share state. For simple tasks, sequential operations, single-file edits, or tasks where you need to maintain context across steps, work directly rather than delegating. +```text Sample prompt for subagent usage wrap +Use subagents when tasks can run in parallel, require isolated context, or involve +independent workstreams that don't need to share state. For simple tasks, sequential +operations, single-file edits, or tasks where you need to maintain context across steps, +work directly rather than delegating. ``` ### Chain complex prompts -With adaptive thinking and subagent orchestration, Claude handles most multi-step reasoning internally. Explicit prompt chaining (breaking a task into sequential API calls) is still useful when you need to inspect intermediate outputs or enforce a specific pipeline structure. +With adaptive thinking and subagent orchestration, Claude handles most multistep reasoning internally. Explicit prompt chaining (breaking a task into sequential API calls) is still useful when you need to inspect intermediate outputs or enforce a specific pipeline structure. The most common chaining pattern is **self-correction:** generate a draft → have Claude review it against criteria → have Claude refine based on the review. Each step is a separate API call so you can log, evaluate, or branch at any point. ### Reduce file creation in agentic coding -Claude's latest models may sometimes create new files for testing and iteration purposes, particularly when working with code. This approach allows Claude to use files, especially python scripts, as a 'temporary scratchpad' before saving its final output. Using temporary files can improve outcomes particularly for agentic coding use cases. +Claude's latest models may sometimes create new files for testing and iteration purposes, particularly when working with code. This approach allows Claude to use files, especially Python scripts, as a 'temporary scratchpad' before saving its final output. Using temporary files can improve outcomes particularly for agentic coding use cases. If you'd prefer to minimize net new file creation, you can instruct Claude to clean up after itself: -```text Sample prompt -If you create any temporary new files, scripts, or helper files for iteration, clean up these files by removing them at the end of the task. +```text Sample prompt wrap +If you create any temporary new files, scripts, or helper files for iteration, clean up +these files by removing them at the end of the task. ``` ### Overeagerness @@ -734,37 +949,57 @@ Claude Opus 4.5 and Claude Opus 4.6 have a tendency to overengineer by creating For example: -```text Sample prompt to minimize overengineering -Avoid over-engineering. Only make changes that are directly requested or clearly necessary. Keep solutions simple and focused: +```text Sample prompt to minimize overengineering wrap +Avoid over-engineering. Only make changes that are directly requested or clearly +necessary. Keep solutions simple and focused: -- Scope: Don't add features, refactor code, or make "improvements" beyond what was asked. A bug fix doesn't need surrounding code cleaned up. A simple feature doesn't need extra configurability. +- Scope: Don't add features, refactor code, or make "improvements" beyond what was +asked. A bug fix doesn't need surrounding code cleaned up. A simple feature doesn't need +extra configurability. -- Documentation: Don't add docstrings, comments, or type annotations to code you didn't change. Only add comments where the logic isn't self-evident. +- Documentation: Don't add docstrings, comments, or type annotations to code you didn't +change. Only add comments where the logic isn't self-evident. -- Defensive coding: Don't add error handling, fallbacks, or validation for scenarios that can't happen. Trust internal code and framework guarantees. Only validate at system boundaries (user input, external APIs). +- Defensive coding: Don't add error handling, fallbacks, or validation for scenarios +that can't happen. Trust internal code and framework guarantees. Only validate at system +boundaries (user input, external APIs). -- Abstractions: Don't create helpers, utilities, or abstractions for one-time operations. Don't design for hypothetical future requirements. The right amount of complexity is the minimum needed for the current task. +- Abstractions: Don't create helpers, utilities, or abstractions for one-time +operations. Don't design for hypothetical future requirements. The right amount of +complexity is the minimum needed for the current task. ``` -### Avoid focusing on passing tests and hard-coding +### Avoid focusing on passing tests and hardcoding -Claude can sometimes focus too heavily on making tests pass at the expense of more general solutions, or may use workarounds like helper scripts for complex refactoring instead of using standard tools directly. To prevent this behavior and ensure robust, generalizable solutions: +Claude can sometimes focus too heavily on making tests pass at the expense of more general solutions, or may use workarounds like helper scripts for complex refactoring instead of using standard tools directly. To prevent this behavior and get solutions that generalize: -```text Sample prompt -Please write a high-quality, general-purpose solution using the standard tools available. Do not create helper scripts or workarounds to accomplish the task more efficiently. Implement a solution that works correctly for all valid inputs, not just the test cases. Do not hard-code values or create solutions that only work for specific test inputs. Instead, implement the actual logic that solves the problem generally. +```text Sample prompt wrap +Please write a high-quality, general-purpose solution using the standard tools +available. Do not create helper scripts or workarounds to accomplish the task more +efficiently. Implement a solution that works correctly for all valid inputs, not just +the test cases. Do not hard-code values or create solutions that only work for specific +test inputs. Instead, implement the actual logic that solves the problem generally. -Focus on understanding the problem requirements and implementing the correct algorithm. Tests are there to verify correctness, not to define the solution. Provide a principled implementation that follows best practices and software design principles. +Focus on understanding the problem requirements and implementing the correct algorithm. +Tests are there to verify correctness, not to define the solution. Provide a principled +implementation that follows best practices and software design principles. -If the task is unreasonable or infeasible, or if any of the tests are incorrect, please inform me rather than working around them. The solution should be robust, maintainable, and extendable. +If the task is unreasonable or infeasible, or if any of the tests are incorrect, please +inform me rather than working around them. The solution should be robust, maintainable, +and extendable. ``` ### Minimizing hallucinations in agentic coding Claude's latest models are less prone to hallucinations and give more accurate, grounded, intelligent answers based on the code. To encourage this behavior even more and minimize hallucinations: -```text Sample prompt +```text Sample prompt wrap -Never speculate about code you have not opened. If the user references a specific file, you MUST read the file before answering. Make sure to investigate and read relevant files BEFORE answering questions about the codebase. Never make any claims about code before investigating unless you are certain of the correct answer - give grounded and hallucination-free answers. +Never speculate about code you have not opened. If the user references a specific file, +you MUST read the file before answering. Make sure to investigate and read relevant +files BEFORE answering questions about the codebase. Never make any claims about code +before investigating unless you are certain of the correct answer - give grounded and +hallucination-free answers. ``` @@ -774,27 +1009,40 @@ Never speculate about code you have not opened. If the user references a specifi Claude Opus 4.5 and Claude Opus 4.6 have improved vision capabilities compared to previous Claude models. They perform better on image processing and data extraction tasks, particularly when there are multiple images present in context. These improvements carry over to computer use, where the models can more reliably interpret screenshots and UI elements. You can also use these models to analyze videos by breaking them up into frames. -One technique that has proven effective to further boost performance is to give Claude a crop tool or [skill](/docs/en/agents-and-tools/agent-skills/overview). Testing has shown consistent uplift on image evaluations when Claude is able to "zoom" in on relevant regions of an image. Anthropic has created a [cookbook for the crop tool](https://platform.claude.com/cookbook/multimodal-crop-tool). +One technique that has proven effective to further boost performance is to give Claude a crop tool or [agent skill](/docs/en/agents-and-tools/agent-skills/overview). Testing has shown consistent uplift on image evaluations when Claude is able to "zoom" in on relevant regions of an image. Anthropic has created a [recipe for the crop tool](https://platform.claude.com/cookbook/multimodal-crop-tool). ### Frontend design -Claude Opus 4.5 and Claude Opus 4.6 excel at building complex, real-world web applications with strong frontend design. However, without guidance, models can default to generic patterns that create what users call the "AI slop" aesthetic. To create distinctive, creative frontends that surprise and delight: +Claude Opus 4.5 and Claude Opus 4.6 build complex, real-world web applications with strong frontend design. However, without guidance, models can default to generic patterns that create what users call the "AI slop" aesthetic. To create distinctive, creative frontends that surprise and delight: -For a detailed guide on improving frontend design, see the blog post on [improving frontend design through skills](https://www.claude.com/blog/improving-frontend-design-through-skills). + For a detailed guide on improving frontend design, see the blog post on [improving frontend design through skills](https://www.claude.com/blog/improving-frontend-design-through-skills). +For frontend design work outside the API, [Claude Design](https://support.claude.com/en/articles/14604416-get-started-with-claude-design) provides a canvas and design tools where Claude generates and iterates on designs interactively. + Here's a system prompt snippet you can use to encourage better frontend design: -```text Sample prompt for frontend aesthetics +```text Sample prompt for frontend aesthetics wrap -You tend to converge toward generic, "on distribution" outputs. In frontend design, this creates what users call the "AI slop" aesthetic. Avoid this: make creative, distinctive frontends that surprise and delight. +You tend to converge toward generic, "on distribution" outputs. In frontend design, this +creates what users call the "AI slop" aesthetic. Avoid this: make creative, distinctive +frontends that surprise and delight. Focus on: -- Typography: Choose fonts that are beautiful, unique, and interesting. Avoid generic fonts like Arial and Inter; opt instead for distinctive choices that elevate the frontend's aesthetics. -- Color & Theme: Commit to a cohesive aesthetic. Use CSS variables for consistency. Dominant colors with sharp accents outperform timid, evenly-distributed palettes. Draw from IDE themes and cultural aesthetics for inspiration. -- Motion: Use animations for effects and micro-interactions. Prioritize CSS-only solutions for HTML. Use Motion library for React when available. Focus on high-impact moments: one well-orchestrated page load with staggered reveals (animation-delay) creates more delight than scattered micro-interactions. -- Backgrounds: Create atmosphere and depth rather than defaulting to solid colors. Layer CSS gradients, use geometric patterns, or add contextual effects that match the overall aesthetic. +- Typography: Choose fonts that are beautiful, unique, and interesting. Avoid generic +fonts like Arial and Inter; opt instead for distinctive choices that elevate the +frontend's aesthetics. +- Color & Theme: Commit to a cohesive aesthetic. Use CSS variables for consistency. +Dominant colors with sharp accents outperform timid, evenly-distributed palettes. Draw +from IDE themes and cultural aesthetics for inspiration. +- Motion: Use animations for effects and micro-interactions. Prioritize CSS-only +solutions for HTML. Use Motion library for React when available. Focus on high-impact +moments: one well-orchestrated page load with staggered reveals (animation-delay) +creates more delight than scattered micro-interactions. +- Backgrounds: Create atmosphere and depth rather than defaulting to solid colors. Layer +CSS gradients, use geometric patterns, or add contextual effects that match the overall +aesthetic. Avoid generic AI-generated aesthetics: - Overused font families (Inter, Roboto, Arial, system fonts) @@ -802,7 +1050,10 @@ Avoid generic AI-generated aesthetics: - Predictable layouts and component patterns - Cookie-cutter design that lacks context-specific character -Interpret creatively and make unexpected choices that feel genuinely designed for the context. Vary between light and dark themes, different fonts, different aesthetics. You still tend to converge on common choices (Space Grotesk, for example) across generations. Avoid this: it is critical that you think outside the box! +Interpret creatively and make unexpected choices that feel genuinely designed for the +context. Vary between light and dark themes, different fonts, different aesthetics. You +still tend to converge on common choices (Space Grotesk, for example) across +generations. Avoid this: it is critical that you think outside the box! ``` @@ -810,7 +1061,7 @@ You can also refer to the [full skill definition](https://github.com/anthropics/ ## Migration considerations -When migrating to Claude 4.6 models from earlier generations: +When migrating to current Claude models from earlier generations: 1. **Be specific about desired behavior:** Consider describing exactly what you'd like to see in the output. @@ -822,83 +1073,26 @@ When migrating to Claude 4.6 models from earlier generations: 5. **Migrate away from prefilled responses:** Prefilled responses on the last assistant turn are no longer supported starting with Claude 4.6 models. See [Migrating away from prefilled responses](#migrating-away-from-prefilled-responses) for detailed guidance on alternatives. -6. **Tune anti-laziness prompting:** If your prompts previously encouraged the model to be more thorough or use tools more aggressively, dial back that guidance. Claude 4.6 models are significantly more proactive and may overtrigger on instructions that were needed for previous models. +6. **Tune anti-laziness prompting:** If your prompts previously encouraged the model to be more thorough or use tools more aggressively, dial back that guidance. Claude 4.6 models are more proactive and may overtrigger on instructions that were needed for previous models. For detailed migration steps, see the [Migration guide](/docs/en/about-claude/models/migration-guide). -### Migrating from Claude Sonnet 4.5 to Claude Sonnet 4.6 - -Claude Sonnet 4.6 defaults to an effort level of `high`, in contrast to Claude Sonnet 4.5 which had no effort parameter. Consider adjusting the effort parameter as you migrate from Claude Sonnet 4.5 to Claude Sonnet 4.6. If not explicitly set, you may experience higher latency with the default effort level. - -**Recommended effort settings:** -- **Medium** for most applications -- **Low** for high-volume or latency-sensitive workloads -- Set a large max output token budget (64k tokens recommended) at medium or high effort to give the model room to think and act - -**When to use Opus 4.7 instead:** For the hardest, longest-horizon problems (large-scale code migrations, deep research, extended autonomous work), Opus 4.7 remains the right choice. Sonnet 4.6 is optimized for workloads where fast turnaround and cost efficiency matter most. - -#### If you're not using extended thinking - -If you're not using extended thinking on Claude Sonnet 4.5, you can continue without it on Claude Sonnet 4.6. You should explicitly set effort to the level appropriate for your use case. At `low` effort with thinking disabled, you can expect similar or better performance relative to Claude Sonnet 4.5 with no extended thinking. - -```python Python -client.messages.create( - model="claude-sonnet-4-6", - max_tokens=8192, - thinking={"type": "disabled"}, - output_config={"effort": "low"}, - messages=[{"role": "user", "content": "..."}], -) -``` - -#### If you're using extended thinking +### Migrating to Claude Sonnet 5 from Claude Sonnet 4.5 or earlier -If you're using extended thinking with `budget_tokens` on Claude Sonnet 4.5, it is still functional on Claude Sonnet 4.6 but is deprecated. Migrate to [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) with the [effort parameter](/docs/en/build-with-claude/effort). +See [Migrating to Claude Sonnet 5 from Claude Sonnet 4.5 or earlier](/docs/en/about-claude/models/migration-guide#migrating-from-sonnet-45) in the migration guide, which covers the effort default change and the removal of manual extended thinking (`budget_tokens`). -##### Migrating to adaptive thinking +## Next steps -Adaptive thinking is particularly well suited to the following workload patterns: + + + Behavioral differences and prompting patterns for Claude Fable 5 and Claude Mythos 5, covering effort, instruction following, long runs, memory, and scaffolding changes. + -- **Autonomous multi-step agents:** coding agents that turn requirements into working software, data analysis pipelines, and bug finding where the model runs independently across many steps. Adaptive thinking lets the model calibrate its reasoning per step, staying on path over longer trajectories. For these workloads, start at `high` effort. If latency or token usage is a concern, scale down to `medium`. -- **Computer use agents:** Claude Sonnet 4.6 achieved best-in-class accuracy on computer use evaluations using adaptive mode. -- **Bimodal workloads:** a mix of easy and hard tasks where adaptive skips thinking on simple queries and reasons deeply on complex ones. - -When using adaptive thinking, evaluate `medium` and `high` effort on your tasks. The right level depends on your workload's tradeoff between quality, latency, and token usage. - -```python Python nocheck -client.messages.create( - model="claude-sonnet-4-6", - max_tokens=64000, - thinking={"type": "adaptive"}, - output_config={"effort": "high"}, - messages=[{"role": "user", "content": "..."}], -) -``` - -##### Keeping budget_tokens during migration - -If you need to keep `budget_tokens` temporarily while migrating, a budget around 16k tokens provides headroom for harder problems without risk of runaway token usage. This configuration is deprecated and will be removed in a future model release. - -**For coding use cases** (agentic coding, tool-heavy workflows, code generation), start with `medium` effort: - -```python Python nocheck -client.messages.create( - model="claude-sonnet-4-6", - max_tokens=16384, - thinking={"type": "enabled", "budget_tokens": 16384}, - output_config={"effort": "medium"}, - messages=[{"role": "user", "content": "..."}], -) -``` + + Behavioral differences and prompting patterns for Claude Sonnet 5, covering effort, adaptive thinking defaults, tool use, and migration from Claude Sonnet 4.6. + -**For chat and non-coding use cases** (chat, content generation, search, classification), start with `low` effort: - -```python Python nocheck -client.messages.create( - model="claude-sonnet-4-6", - max_tokens=8192, - thinking={"type": "enabled", "budget_tokens": 16384}, - output_config={"effort": "low"}, - messages=[{"role": "user", "content": "..."}], -) -``` \ No newline at end of file + + When to use prompt engineering and how to plan your approach before tuning prompts. + + diff --git a/content/en/build-with-claude/prompt-engineering/overview.md b/content/en/build-with-claude/prompt-engineering/overview.md index ac5f76368..7ea21cff0 100644 --- a/content/en/build-with-claude/prompt-engineering/overview.md +++ b/content/en/build-with-claude/prompt-engineering/overview.md @@ -1,20 +1,24 @@ # Prompt engineering overview +Learn when prompt engineering is the right solution, and find Claude prompting techniques, Console prompting tools, and interactive tutorials. + --- ## Before prompt engineering This guide assumes that you have: + 1. A clear definition of the success criteria for your use case 2. Some ways to empirically test against those criteria 3. A first draft prompt you want to improve -If not, we highly suggest you spend time establishing that first. Check out [Define success criteria and build evaluations](/docs/en/test-and-evaluate/develop-tests) for tips and guidance. +If not, spend time establishing that first. Check out [Define success criteria and build evaluations](/docs/en/test-and-evaluate/develop-tests) for tips and guidance. Don't have a first draft prompt? Try the prompt generator in the Claude Console! + For model-specific tuning guidance for Claude's latest models, start here. @@ -24,28 +28,30 @@ If not, we highly suggest you spend time establishing that first. Check out [Def ## When to prompt engineer - This guide focuses on success criteria that are controllable through prompt engineering. - Not every success criteria or failing eval is best solved by prompt engineering. For example, latency and cost can be sometimes more easily improved by selecting a different model. +This guide focuses on success criteria that are controllable through prompt engineering. Not every success criteria or failing eval is best solved by prompt engineering. For example, you can sometimes improve latency and cost more easily by selecting a different model. *** ## How to prompt engineer -All prompting techniques — from clarity and examples to XML structuring, role prompting, thinking, and prompt chaining — are covered in [Prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices). That's the living reference; start there. +All prompting techniques (from clarity and examples to XML structuring, role prompting, thinking, and prompt chaining) are covered in [Prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices). That's the living reference; start there. -The [Claude Console](/dashboard) also offers [prompting tools](/docs/en/build-with-claude/prompt-engineering/prompting-tools)—prompt generator, templates and variables, and prompt improver—to help you build and refine prompts quickly. +For general prompt engineering craft beyond Claude-specific techniques, see the blog post on [best practices for prompt engineering](https://claude.com/blog/best-practices-for-prompt-engineering). + +The [Claude Console](/dashboard) also offers [prompting tools](/docs/en/build-with-claude/prompt-engineering/prompting-tools) (prompt generator, templates and variables, and prompt improver) to help you build and refine prompts quickly. *** ## Prompt engineering tutorial -If you're an interactive learner, you can dive into our interactive tutorials instead! +If you're an interactive learner, you can start with the interactive tutorials instead! - An example-filled tutorial that covers the prompt engineering concepts found in our docs. + An example-filled tutorial that covers the prompt engineering concepts found in the docs. + - A lighter weight version of our prompt engineering tutorial via an interactive spreadsheet. + A lighter-weight version of the prompt engineering tutorial, as an interactive spreadsheet. - \ No newline at end of file + diff --git a/content/en/build-with-claude/prompt-engineering/prompting-claude-fable-5.md b/content/en/build-with-claude/prompt-engineering/prompting-claude-fable-5.md new file mode 100644 index 000000000..9d0ab72ee --- /dev/null +++ b/content/en/build-with-claude/prompt-engineering/prompting-claude-fable-5.md @@ -0,0 +1,176 @@ +# Prompting Claude Fable 5 + +Behavioral differences and prompting patterns for Claude Fable 5 and Claude Mythos 5, covering effort, instruction following, long runs, memory, and scaffolding changes. + +--- + +This guide covers the prompting and scaffolding patterns specific to Claude Fable 5 and Claude Mythos 5. For the model's capabilities, API changes, pricing, and availability, see [Introducing Claude Fable 5 and Claude Mythos 5](/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5). For techniques that apply across all current Claude models, see [Prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices). + +Claude Fable 5 takes on problems that were previously too complex, long-running, or ambiguous for prior models, and is particularly effective at end-to-end work that takes a person hours, days, or weeks to complete. The teams seeing the best outcomes apply Claude Fable 5 to their hardest unsolved problems; testing it only on simpler workloads tends to undersell its capability range. It also performs reliably on more straightforward tasks. + +Claude Fable 5 has several behavioral differences from Claude Opus 4.8 that may require prompt or scaffolding updates. Capability improvements at this level are also a good prompt to re-evaluate which instructions, tools, and guardrails are still needed. The patterns below cover the behaviors that most often require tuning. + + + For API parameter changes specific to Claude Fable 5 and Claude Mythos 5 (adaptive thinking only, summarized-only thinking output, no extended thinking budgets, the `refusal` stop reason and fallback handling), see [Introducing Claude Fable 5 and Claude Mythos 5](/docs/en/about-claude/models/introducing-claude-fable-5-and-claude-mythos-5). + + Claude Fable 5 runs safety classifiers that target offensive cybersecurity techniques (such as building exploits, malware, or attack tooling), biology and life sciences content (such as lab methods or molecular mechanisms), and extraction of the model's summarized thinking. Benign cybersecurity work and beneficial life sciences tasks may also trigger these safeguards. To re-route declined requests automatically, configure [server-side or client-side fallback](/docs/en/build-with-claude/refusals-and-fallback) to Claude Opus 4.8. + + +## Capability improvements + +Compared with Claude Opus 4.8, Claude Fable 5 shows improvement in: + +* **Long-horizon autonomy.** Claude Fable 5 sustains productive output over extended periods, completing multiday, goal-directed runs with strong instruction retention across long, complex tasks. +* **First-shot correctness on complex, well-specified problems.** Early testers reported single-pass implementations of systems that previously took days of iteration. +* **Vision.** Claude Fable 5 interprets dense technical images, web applications, and detailed screenshots with substantially higher accuracy, often while using fewer output tokens, and is trained to use bash and crop tools to handle flipped, blurry, or noisy images. +* **Enterprise workflows.** Claude Fable 5 follows instructions, stays in scope, and produces professional-grade output on financial analysis, spreadsheets, slides, and documents. +* **Code review and debugging.** Bug-finding recall (outside the cybersecurity domains the safety classifiers cover) is noticeably higher than Claude Opus 4.8, including search across codebases and repository history. +* **Navigating ambiguity.** Claude Fable 5 performs well when given complex, multithreaded requests and asked to determine next steps. +* **Delegation and collaboration.** Claude Fable 5 is significantly more dependable at dispatching and sustaining parallel subagents, and reliably manages ongoing communication with long-running subagents and peer agents. + +Beyond these specific improvements, Claude Fable 5 is generally more capable than prior models on almost all tasks. Claude Fable 5 is not intended for offensive cybersecurity or biology and life sciences work; requests in those domains can return [`stop_reason: "refusal"`](/docs/en/build-with-claude/refusals-and-fallback). + +## Longer turns by default + +Individual requests on hard tasks can run for many minutes at higher [effort](/docs/en/build-with-claude/effort) settings, especially when the task requires gathering context, building, and self-verifying, and autonomous runs can extend for hours. This is one of the largest shifts teams encounter when adjusting to Claude Fable 5. Adjust client timeouts, streaming, and user-facing progress indicators before migrating, and consider restructuring harnesses to check on runs asynchronously, for example through scheduled jobs, rather than blocking. To keep Claude Fable 5 from overplanning when a task is ambiguous: + +```text wrap +When you have enough information to act, act. Do not re-derive facts already established in the conversation, re-litigate a decision the user has already made, or narrate options you will not pursue in user-facing messages. If you are weighing a choice, give a recommendation, not an exhaustive survey. This does not apply to thinking blocks. +``` + +## Consider all effort levels + +[Effort](/docs/en/build-with-claude/effort) is the primary control for the trade-off between intelligence, latency, and cost on Claude Fable 5. Use `high` as the default for most tasks, with `xhigh` for the most capability-sensitive workloads and `medium` or `low` for routine work. Lower effort settings on Claude Fable 5 still perform well and often exceed `xhigh` performance on prior models. Reduce effort if a task completes but takes longer than necessary, or if you want a quicker, more interactive working style. + +On routine work at higher effort, Claude Fable 5 can gather context and deliberate beyond what the task needs. At the same time, higher effort often produces excellent verification behavior, sophisticated reasoning, and the most rigorous output. To prevent unrequested tidying or refactoring at higher effort: + +```text wrap +Don't add features, refactor, or introduce abstractions beyond what the task requires. A bug fix doesn't need surrounding cleanup and a one-shot operation usually doesn't need a helper. Don't design for hypothetical future requirements: do the simplest thing that works well. Avoid premature abstraction and half-finished implementations. Don't add error handling, fallbacks, or validation for scenarios that cannot happen. Trust internal code and framework guarantees. Only validate at system boundaries (user input, external APIs). Don't use feature flags or backwards-compatibility shims when you can just change the code. +``` + +## Strong instruction following + +Instruction-following is improved enough that you can steer most behaviors with a brief instruction rather than enumerating each behavior by name. For example, when un-steered, Claude Fable 5 can elaborate beyond what the task needs, especially at higher effort settings: surveying options it won't pursue, explaining root causes at length, producing heavily-structured PR descriptions, or writing comments that narrate what the next line does. A short brevity instruction is as effective as listing each pattern: + +```text wrap +Lead with the outcome. Your first sentence after finishing should answer "what happened" or "what did you find": the thing the user would ask for if they said "just give me the TLDR." Supporting detail and reasoning come after. Being readable and being concise are different things, and readability matters more. + +The way to keep output short is to be selective about what you include (drop details that don't change what the reader would do next), not to compress the writing into fragments, abbreviations, arrow chains like A → B → fails, or jargon. +``` + +The same applies to checkpoint behavior in long-running workflows. To have Claude Fable 5 stop only where it genuinely needs you, there is no need to enumerate every case: + +```text wrap +Pause for the user only when the work genuinely requires them: a destructive or irreversible action, a real scope change, or input that only they can provide. If you hit one of these, ask and end the turn, rather than ending on a promise. +``` + +## Ground progress claims during long runs + +On long autonomous runs, instruct Claude Fable 5 to audit progress against actual tool results. In Anthropic's testing, this nearly eliminated fabricated status reports even on tasks designed to elicit them: + +```text wrap +Before reporting progress, audit each claim against a tool result from this session. Only report work you can point to evidence for; if something is not yet verified, say so explicitly. Report outcomes faithfully: if tests fail, say so with the output; if a step was skipped, say that; when something is done and verified, state it plainly without hedging. +``` + +## State the boundaries + +Claude Fable 5 can occasionally take unrequested actions (drafting an email when none was asked for, creating defensive git-branch backups). Define explicit constraints on what Claude Fable 5 should and should not do: + +```text wrap +When the user is describing a problem, asking a question, or thinking out loud rather than requesting a change, the deliverable is your assessment. Report your findings and stop. Don't apply a fix until they ask for one. Before running a command that changes system state (restarts, deletes, config edits), check that the evidence actually supports that specific action. A signal that pattern-matches to a known failure may have a different cause. +``` + +## Parallel subagents + +Claude Fable 5 dispatches parallel subagents more readily than prior models. Use subagents frequently, provide explicit guidance about when delegation is appropriate, and prefer asynchronous communication between orchestrator and subagents over blocking until each subagent returns. Long-lived subagents that keep their context across subtasks save time and cost through cache reads and avoid bottlenecking on the slowest subagent. + +```text wrap +Delegate independent subtasks to subagents and keep working while they run. Intervene if a subagent goes off track or is missing relevant context. +``` + +## Construct a memory system + +Claude Fable 5 performs particularly well when it can record lessons from previous runs and reference them. Provide a place to write notes, as simple as a Markdown file: + +```text wrap +Store one lesson per file with a one-line summary at the top. Record corrections and confirmed approaches alike, including why they mattered. Don't save what the repo or chat history already records; update an existing note rather than creating a duplicate; delete notes that turn out to be wrong. +``` + +To bootstrap the memory system from existing history, have Claude Fable 5 review past sessions: + +```text wrap +Reflect on the previous sessions we've had together. Use subagents to identify core themes and lessons, and store them in [X]. Make sure you know to reference [X] for future use. +``` + +## Rare cases of early stopping + +Deep into a long session, Claude Fable 5 can occasionally end a turn with a text-only statement of intent ("I'll now run X") without issuing the corresponding tool call, or pause to ask permission when it already has enough to proceed. A "continue" or "go ahead and do it end to end" suffices. To define when pausing is appropriate, pair this with the checkpoint instruction in [Strong instruction following](#strong-instruction-following). For autonomous pipelines, add a system reminder: + +```text wrap +You are operating autonomously. The user is not watching in real time and cannot answer questions mid-task, so asking "Want me to…?" or "Shall I…?" will block the work. For reversible actions that follow from the original request, proceed without asking. Offering follow-ups after the task is done is fine; asking permission after already discussing with the user before doing the work is not. Before ending your turn, check your last paragraph. If it is a plan, an analysis, a question, a list of next steps, or a promise about work you have not done ("I'll…", "let me know when…"), do that work now with tool calls. End your turn only when the task is complete or you are blocked on input only the user can provide. +``` + +## Rare cases of context-budget concern + +In very long sessions, Claude Fable 5 can occasionally suggest a new session, offer to summarize and hand off, or trim its own work. This is most often triggered when the harness shows a remaining-token countdown to the model. Avoid surfacing explicit context-budget counts where possible. If the harness must show them, a reassurance helps: + +```text wrap +You have ample context remaining. Do not stop, summarize, or suggest a new session on account of context limits. Continue the work. +``` + +## Give the reason, not only the request + +Claude Fable 5 tends to perform better when it understands the intent behind a request: context lets it connect the task to relevant information rather than inferring intent on its own. Provide context about why you're asking, especially for long-running agents drawing on multiple workstreams: + +```text wrap +I'm working on [the larger task] for [who it's for]. They need [what the output enables]. With that in mind: [request]. +``` + +## Readability when communicating with the user + +In extended or agentic conversations (many tool calls, large working context), Claude Fable 5 can produce text that's hard to follow: dense arrow-chain shorthand, deep implementation detail, references to thinking the user never saw, or overly technical phrasing. A communication-style addendum mitigates this: + +```text wrap +Terse shorthand is fine between tool calls (that's you thinking out loud, and brevity there is good). Your final summary is different: it's for a reader who didn't see any of that. + +If you've been working for a while without the user watching (overnight, across many tool calls, since they last spoke), your final message is their first look at any of it. Write it as a re-grounding, not a continuation of your working thread: the outcome first, then the one or two things you need from them, each explained as if new. The vocabulary you built up while working is yours, not theirs; leave it behind unless you re-introduce it. + +When you write the summary at the end, drop the working shorthand. Write complete sentences. Spell out terms. Don't use arrow chains, hyphen-stacked compounds, or labels you made up earlier. When you mention files, commits, flags, or other identifiers, give each one its own plain-language clause. Open with the outcome: one sentence on what happened or what you found. Then the supporting detail. If you have to choose between short and clear, choose clear. +``` + +## Create a send-to-user tool + +When running long, asynchronous agents, give the agent a way to surface a message the user must see exactly as written, without ending its turn: a deliverable (a generated code snippet or a drafted message), a progress update with specific numbers, or a direct reply to a question the user asked mid-loop. The tool's input is the message to display; when Claude calls it, render the input directly in your UI and return a simple acknowledgement as the tool result. Tool inputs are never summarized, so the content arrives intact. + +```json +{ + "name": "send_to_user", + "description": "Display a message directly to the user. Use this for progress updates, partial results, or content the user must see exactly as written before the task finishes.", + "input_schema": { + "type": "object", + "properties": { + "message": { + "type": "string", + "description": "The content to display to the user." + } + }, + "required": ["message"] + } +} +``` + +Add this tool whenever your UX depends on delivering content or direct user interactions verbatim mid-task. For agents that only narrate routine progress, the model's own summaries are typically adequate. Defining the tool is not sufficient on its own; without an instruction in the system prompt, Claude Fable 5 rarely calls it. Pair the tool with elicitation language such as: + +```text wrap +Between tool calls, when you have content the user must read verbatim (a partial deliverable, a direct answer to their question), call the send_to_user tool with that content. Use send_to_user only for user-facing content, not for narration or reasoning. +``` + +Do not route narration or internal reasoning through `send_to_user`; over-calling it for non-user-facing content defeats the purpose. + +## Recommended scaffolding changes + +* **Start at the top of your difficulty range.** Pick a task harder than what you'd assign to prior models, and have Claude Fable 5 scope it, ask clarifying questions, and execute. +* **Make self-verification explicit in long-run prompts.** Separate, fresh-context verifier subagents tend to outperform self-critique. For long-running tasks, instruct: `Establish a method for checking your own work at an interval of [X] as you build. Run this every [X interval], verifying your work with subagents against the specification.` +* **Refactor existing prompts and skills.** Skills developed for prior models are often too prescriptive for Claude Fable 5 and can degrade output quality. Review and consider removing older instructions if default performance is better. Claude Fable 5 also does a good job of updating skills on the fly based on what it learns from the task at hand. +* **Don't instruct Claude to reproduce its reasoning in the response.** Prompts, skills, or harness instructions that tell the model to echo, transcribe, or explain its internal reasoning as response text can trigger the [`reasoning_extraction` refusal category](/docs/en/build-with-claude/refusals-and-fallback#refusal-response) on Claude Fable 5, causing elevated fallbacks to Claude Opus 4.8. Audit existing skills and system prompts for reflection or show-your-thinking instructions when migrating. If your application needs reasoning visibility, read the structured `thinking` blocks from [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) instead, and use a [send-to-user tool](#create-a-send-to-user-tool) to surface progress during long runs. +* **Create a send-to-user tool.** For long, asynchronous agents, a client-side tool delivers messages to the user verbatim without ending the turn. See [Create a send-to-user tool](#create-a-send-to-user-tool). diff --git a/content/en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8.md b/content/en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8.md new file mode 100644 index 000000000..cbd49348a --- /dev/null +++ b/content/en/build-with-claude/prompt-engineering/prompting-claude-opus-4-8.md @@ -0,0 +1,162 @@ +# Prompting Claude Opus 4.8 + +Behavioral differences and prompting patterns for Claude Opus 4.8, covering verbosity, effort calibration, tool use, subagents, and frontend defaults. + +--- + +This guide covers the prompting patterns specific to Claude Opus 4.8. For the model's capabilities and API changes, see [What's new in Claude Opus 4.8](/docs/en/about-claude/models/whats-new-claude-4-8). For techniques that apply across all current Claude models, see [Prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices). + +Claude Opus 4.8 has particular strengths in long-horizon agentic work, knowledge work, vision, and memory tasks. It performs well out of the box on existing Claude Opus 4.7 prompts. The following patterns cover the behaviors that most often require tuning. + + + For API parameter changes when migrating from Claude Opus 4.7 (sampling parameters, effort default, 1M context window default, mid-conversation system messages, and refusal stop details), see the [migration guide](/docs/en/about-claude/models/migration-guide#migrating-from-claude-opus-47). + + +## Response length and verbosity + +Claude Opus 4.8 calibrates response length to how complex it judges the task to be, rather than defaulting to a fixed verbosity. This usually means shorter answers on simple lookups and much longer ones on open-ended analysis. + +If your product depends on a certain style or verbosity of output, you may need to tune your prompts. As an example, to decrease verbosity, you might add: + +```text wrap +Provide concise, focused responses. Skip non-essential context, and keep examples minimal. +``` + +If you see specific examples of kinds of verbosity (such as over-explaining), you can add additional instructions in your prompt to prevent them. Positive examples showing how Claude can communicate with the appropriate level of concision tend to be more effective than negative examples or instructions that tell the model what not to do. + +## Calibrating effort and thinking depth + +The [effort parameter](/docs/en/build-with-claude/effort) allows you to tune Claude's intelligence versus token spend, trading off capability for faster speed and lower costs. Start with the `xhigh` effort level for coding and agentic use cases, and use a minimum of `high` effort for most intelligence-sensitive use cases. Experiment with other effort levels to further tune token usage and intelligence: + +* **`max`:** Max effort can deliver performance gains in some use cases, but may show diminishing returns from increased token usage. This setting can also sometimes be prone to overthinking. Test max effort for intelligence-demanding tasks. +* **`xhigh`:** Extra high effort is the best setting for most coding and agentic use cases. +* **`high`:** This setting balances token usage and intelligence. For most intelligence-sensitive use cases, use a minimum of `high` effort. +* **`medium`:** Good for cost-sensitive use cases that need to reduce token usage while trading off intelligence. +* **`low`:** Reserve for short, scoped tasks and latency-sensitive workloads that are not intelligence-sensitive. + +Claude Opus 4.8 respects effort levels strictly, especially at the low end. At `low` and `medium`, the model scopes its work to what was asked rather than going above and beyond. This is good for latency and cost, but on moderately complex tasks running at `low` effort there is some risk of under-thinking. + +If you observe shallow reasoning on complex problems, raise effort to `high` or `xhigh` rather than prompting around it. If you need to keep effort at `low` for latency, add targeted guidance: + +```text wrap +This task involves multistep reasoning. Think carefully through the problem before responding. +``` + +Effort is likely to be more important for this model than for any prior Opus, so experiment with it actively when you upgrade. + +On Claude Opus 4.8, thinking is off unless you explicitly set `thinking: {type: "adaptive"}`. The triggering behavior for adaptive thinking is steerable. If you find the model thinking more often than you'd like, which can happen with large or complex system prompts, add guidance to steer it. As always, measure the effect of any prompting changes on performance. Example: + +```text wrap +Thinking adds latency and should only be used when it will meaningfully improve answer quality — typically for problems that require multistep reasoning. When in doubt, respond directly. +``` + +Conversely, if you're running hard workloads at `medium` and seeing under-thinking, the first lever is to raise effort. If you need finer control, prompt for it directly. + + + If you are running Claude Opus 4.8 at `max` or `xhigh` effort, set a large max output token budget so the model has room to think and act across its subagents and tool calls. Start at 64k tokens and tune from there. + + +## Tool use triggering + +Claude Opus 4.8 has a tendency to favor reasoning over tool calls. This produces better results in most cases. However, increasing the effort setting is a useful lever to increase the level of tool usage, especially in knowledge work. `high` or `xhigh` effort settings show substantially more tool usage in agentic search and coding. For scenarios where you want more tool use, you can also adjust your prompt to explicitly instruct the model about when and how to properly use its tools. For instance, if you find that the model is not using your web search tools, clearly describe why and how it should. + +## User-facing progress updates + +Claude Opus 4.8 provides more regular, higher-quality updates to the user throughout long agentic traces. If you've added scaffolding to force interim status messages ("After every 3 tool calls, summarize progress"), try removing it. If you find that the length or contents of Claude Opus 4.8's user-facing updates are not well-calibrated to your use case, explicitly describe what these updates should look like in the prompt and provide examples. + +## More literal instruction following + +Claude Opus 4.8 interprets prompts literally and explicitly, particularly at lower effort levels. It does not silently generalize an instruction from one item to another, and it does not infer requests you didn't make. The upside of this literalism is precision and less thrash, and it generally performs better for API use cases with carefully tuned prompts, structured extraction, and pipelines where you want predictable behavior. If you need Claude to apply an instruction broadly, state the scope explicitly (for example, "Apply this formatting to every section, not just the first one"). + +## Tone and writing style + +As with any new model, prose style on long-form writing may shift. Claude Opus 4.8 tends toward a direct, opinionated style with minimal validation-forward phrasing and sparing emoji use. If your product relies on a specific voice, re-evaluate style prompts against the new baseline. + +For instance, if your product voice is warmer or more conversational, add: + +```text wrap +Use a warm, collaborative tone. Acknowledge the user's framing before answering. +``` + +## Controlling subagent spawning + +Claude Opus 4.8 tends to spawn fewer subagents by default. However, this behavior is steerable through prompting; give Claude Opus 4.8 explicit guidance around when subagents are desirable. A toy example for a coding use case: + +```text wrap +Do not spawn a subagent for work you can complete directly in a single response (e.g. refactoring a function you can already see). + +Spawn multiple subagents in the same turn when fanning out across items or reading multiple files. +``` + +## Design and frontend defaults + +Claude Opus 4.8 has strong design instincts, with a consistent default house style: warm cream/off-white backgrounds (\~`#F4F1EA`), serif display type (Georgia, Fraunces, Playfair), italic word-accents, and a terracotta/amber accent. This reads well for editorial, hospitality, and portfolio briefs, but will feel off for dashboards, dev tools, fintech, healthcare, or enterprise apps. The default appears in slide decks and web UIs. + +This default is persistent. Generic instructions ("don't use cream," "make it clean and minimal") tend to shift the model to a different fixed palette rather than producing variety. Two approaches work reliably: + +**1. Specify a concrete alternative.** The model follows explicit specs precisely: + +```text wrap +Design a desktop landing page for a supplement brand called AEFRM. + +The visual direction should come from a cold monochrome atmosphere using pale silver-gray tones that gradually deepen into blue-gray and near-black, similar to a misted metallic surface. + +The page should feel sharp and controlled, with a strong sense of structure and restraint. + +Use this tonal system across the full page instead of introducing bright accent colors. + +Use the uploaded image on the hero design in black and white. + +The layout should be built with clear horizontal sections and a centered max-width container. Use 4px corner radius consistently across cards, buttons, inputs, and media frames. Margins should feel generous, with enough empty space around each section so the page breathes. + +Typography should use a square, angular sans-serif with wider letter spacing than usual, especially in headings and navigation, so the text feels more engineered and less compressed. Headline text can be large and uppercase, while supporting copy remains short and sparse. The sub texts should be written with Alumni Sans SC in 4-6px like tiny little texts on corners bottom centre like that. + +For the structure, start with a hero section containing a strong product statement, one short supporting paragraph, and a clean product placeholder or packshot frame. Below that, add a benefit grid with three or four blocks, then a formulation or ingredients section, and finally a cta. + +Buttons should be flat and precise, with subtle hover changes using transition: all 160ms ease out where brightness and border contrast shift slightly rather than using dramatic motion. + +Color palette should stay within this range: +#E9ECEC, #C9D2D4, #8C9A9E, #44545B, #11171B. +``` + +**2. Have the model propose options before building.** This breaks the default and gives users control. If you previously relied on `temperature` for design variety, use this approach; it produces meaningfully different directions across runs. Example prompt: + +```text wrap +Before building, propose 4 distinct visual directions tailored to this brief (each as: bg hex / accent hex / typeface — one-line rationale). Ask the user to pick one, then implement only that direction. +``` + +Additionally, Claude Opus 4.8 requires less frontend design prompting than previous models to avoid generic patterns that users call the "AI slop" aesthetic. With earlier models, Anthropic recommended a lengthier prompt snippet in the [frontend-design skill](https://github.com/anthropics/claude-code/blob/main/plugins/frontend-design/skills/frontend-design/SKILL.md). However, Claude Opus 4.8 generates distinctive, creative frontends with more minimal prompting guidance. This prompt snippet works well with the preceding prompting advice for variety: + +```text wrap + +NEVER use generic AI-generated aesthetics like overused font families (Inter, Roboto, Arial, system fonts), cliched color schemes (particularly purple gradients on white or dark backgrounds), predictable layouts and component patterns, and cookie-cutter design that lacks context-specific character. Use unique fonts, cohesive colors and themes, and animations for effects and micro-interactions. + +``` + +## Interactive coding products + +Claude Opus 4.8's token usage and behavior can differ between autonomous, asynchronous coding agents with a single user turn and interactive, synchronous coding agents with multiple user turns. Specifically, it tends to use more tokens in interactive settings, primarily because it reasons more after user turns. This can improve long-horizon coherence, instruction following, and coding capabilities in long, interactive coding sessions, but also comes with more token usage. To maximize both performance and token efficiency in coding products, use `xhigh` or `high` effort, add autonomous features like an auto mode, and reduce the number of human interactions required from your users. + +Of course, when limiting the number of required user interactions, it's important to specify the task, intent, and relevant constraints upfront in the first human turn. Providing well-specified, clear, and accurate task descriptions upfront can help maximize autonomy and intelligence while minimizing extra token usage after user turns. Because Claude Opus 4.8 is more autonomous than prior models, this usage pattern helps to maximize performance. In contrast, ambiguous or underspecified prompts conveyed progressively over multiple user turns tend to relatively reduce token efficiency and sometimes performance. + +## Code review harnesses + +Claude Opus 4.8 is meaningfully better at finding bugs than prior models, and has both higher recall and precision in internal evals. However, if your code-review harness was tuned for an earlier model, you may initially see lower recall. This is likely a harness effect, not a capability regression. When a review prompt says things like "only report high-severity issues," "be conservative," or "don't nitpick," Claude Opus 4.8 may follow that instruction more faithfully than earlier models did: it may investigate the code just as thoroughly, identify the bugs, and then not report findings it judges to be below your stated bar. This can show up as the model doing the same depth of investigation but converting fewer investigations into reported findings, especially on lower-severity bugs. Precision typically rises, but measured recall can fall even though the model's underlying bug-finding ability has improved. + +Some recommended prompt language: + +```text wrap +Report every issue you find, including ones you are uncertain about or consider low-severity. Do not filter for importance or confidence at this stage - a separate verification step will do that. Your goal here is coverage: it is better to surface a finding that later gets filtered out than to silently drop a real bug. For each finding, include your confidence level and an estimated severity so a downstream filter can rank them. +``` + +This prompt can be used without having an actual second step, but moving confidence filtering out of the finding step often helps. If your harness has a separate verification, deduplication, or ranking stage, tell the model explicitly that its job at the finding stage is coverage rather than filtering. + +If you do want the model to self-filter in a single pass, be concrete about where the bar is rather than using qualitative terms like "important": for example, "report any bugs that could cause incorrect behavior, a test failure, or a misleading result; only omit nits like pure style or naming preferences." + +Iterate on prompts against a subset of your evals or test cases to validate recall or F1 score gains. + +## Computer use + +[Computer use](/docs/en/agents-and-tools/tool-use/computer-use-tool) capability works across resolutions, up to a maximum resolution of 2576px / 3.75MP. Internal computer use testing shows that sending images at 1080p provides a good balance of performance and cost. + +For particularly cost-sensitive workloads, 720p or 1366×768 are lower-cost options with strong performance. Conduct your own testing to find the ideal settings for your use case; experimenting with effort settings can also help tune the model's behavior. diff --git a/content/en/build-with-claude/prompt-engineering/prompting-claude-sonnet-5.md b/content/en/build-with-claude/prompt-engineering/prompting-claude-sonnet-5.md new file mode 100644 index 000000000..b5d7f353f --- /dev/null +++ b/content/en/build-with-claude/prompt-engineering/prompting-claude-sonnet-5.md @@ -0,0 +1,158 @@ +# Prompting Claude Sonnet 5 + +Behavioral differences and prompting patterns for Claude Sonnet 5, covering effort, adaptive thinking defaults, tool use, and migration from Claude Sonnet 4.6. + +--- + +This guide covers the prompting patterns specific to Claude Sonnet 5. For the model's capabilities and API changes, see [What's new in Claude Sonnet 5](/docs/en/about-claude/models/whats-new-sonnet-5). For techniques that apply across all current Claude models, see [Prompting best practices](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices). + +Claude Sonnet 5 has particular strengths in coding and agentic tasks. It performs well out of the box on existing Claude Sonnet 4.6 prompts. The patterns in this guide cover the behaviors that most often require tuning. + + + For API parameter changes when migrating from Claude Sonnet 4.6 (adaptive thinking on by default, sampling parameters not accepted, manual extended thinking removed, and the new tokenizer), see the [migration guide](/docs/en/about-claude/models/migration-guide#migrating-from-claude-sonnet-4-6-to-claude-sonnet-5). + + +## Response length and verbosity + +Claude Sonnet 5 calibrates response length to the complexity of the task rather than defaulting to a fixed verbosity. This usually means shorter answers on simple lookups and longer ones on open-ended analysis. + +If your product depends on a certain style or verbosity of output, you may need to tune your prompts. As an example, to decrease verbosity, you might add: + +```text wrap +Provide concise, focused responses. Skip non-essential context, and keep examples minimal. +``` + +If you see specific kinds of verbosity (such as over-explaining), you can add additional instructions in your prompt to prevent them. Positive examples showing how Claude can communicate with the appropriate level of concision tend to be more effective than negative examples or instructions that tell the model what not to do. + +## Calibrating effort and thinking depth + +The [effort parameter](/docs/en/build-with-claude/effort) allows you to tune Claude's intelligence versus token spend, trading off capability for faster speed and lower costs. On Claude Sonnet 5, effort defaults to `high`, the same as on Claude Sonnet 4.6. For the hardest coding and agentic tasks, raise effort to `xhigh`. Experiment with other effort levels to further tune token usage and intelligence: + +* **`max`:** Absolute maximum capability with no constraints on token spending. +* **`xhigh`:** Extra high effort is the recommended setting for the hardest coding and agentic use cases. +* **`high`:** The default. This setting balances token usage and intelligence for most use cases. +* **`medium`:** Good for cost-sensitive use cases that need to reduce token usage while trading off intelligence. +* **`low`:** Reserve for short, scoped tasks and latency-sensitive workloads that are not intelligence-sensitive. + +As a rough cross-model mapping when migrating: Claude Sonnet 5 at medium is comparable in intelligence to Claude Sonnet 4.6 at high, and Claude Sonnet 5 at high is comparable to Claude Sonnet 4.6 at max. When benchmarking, match by observed thinking length rather than effort name. + +Claude Sonnet 5 respects effort levels strictly, especially at the low end. At `low` and `medium`, the model scopes its work to what was asked rather than going above and beyond. This is good for latency and cost, but on moderately complex tasks running at `low` effort there is some risk of under-thinking. + +If you observe shallow reasoning on complex problems, raise effort to `high` or `xhigh` rather than prompting around it. If you need to keep effort at `low` for latency, add targeted guidance: + +```text wrap +This task involves multistep reasoning. Think carefully through the problem before responding. +``` + +On Claude Sonnet 5, [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) is on by default. Requests without a `thinking` field run with adaptive thinking. This is a change from Claude Sonnet 4.6, where the same requests ran without thinking. To turn thinking off entirely, pass `thinking: {type: "disabled"}`. Because `max_tokens` is a hard limit on total output (thinking plus response text), revisit it for workloads that ran without thinking on Claude Sonnet 4.6. If you were previously using thinking off with Claude Sonnet 4.6, try thinking on with lower effort levels for Claude Sonnet 5. + +The triggering behavior for adaptive thinking is steerable. If you find the model emitting thinking blocks more often than you'd like, which can happen with large or complex system prompts, add guidance to steer it. As always, measure the effect of any prompting changes on performance. Example: + +```text wrap +Thinking adds latency and should only be used when it will meaningfully improve answer quality, typically for problems that require multistep reasoning. When in doubt, respond directly. +``` + +Conversely, if you're running hard workloads at `medium` and seeing under-thinking, the first lever is to raise effort. If you need finer control, prompt for it directly. + +Manual extended thinking (`thinking: {type: "enabled", budget_tokens: N}`) is not supported on Claude Sonnet 5 and returns a 400 error. It was deprecated on Claude Sonnet 4.6 and is now removed. Use adaptive thinking with the effort parameter instead. + + + If you are running Claude Sonnet 5 at `high`, `xhigh`, or `max` effort, leave headroom in `max_tokens` so the model has room for thinking and tool calls. On long tasks, adaptive thinking can use a large share of the budget; if the budget is tight, you may see a response that is almost entirely thinking followed by a truncated answer and `stop_reason: "max_tokens"`. Raising `max_tokens` or dropping to `medium` effort resolves this. Because Claude Sonnet 5 uses a [new tokenizer](/docs/en/about-claude/models/whats-new-sonnet-5#new-tokenizer) that produces approximately 30% more tokens for the same text, `max_tokens` limits tuned for Claude Sonnet 4.6 may truncate equivalent output. The exact increase depends on the content and workload shape. + + +## Tool use triggering + +Claude Sonnet 5 is more agentic than Claude Sonnet 4.6 by default and will reach for tools and run self-verification loops more readily. With thinking disabled, the model is less likely to reach for tools or consider searching; if you rely on tool calls with thinking off, add an explicit nudge in the system prompt. Effort is also a lever for tool usage: `high` or `xhigh` effort settings show substantially more tool usage in agentic search and coding. For scenarios where you want more tool use, you can also adjust your prompt to explicitly instruct the model about when and how to properly use its tools. For instance, if you find that the model is not using your web search tools, clearly describe why and how it should. + +## User-facing progress updates + +Claude Sonnet 5 provides regular, higher-quality updates to the user throughout long agentic traces. If you've added scaffolding to force interim status messages ("After every 3 tool calls, summarize progress"), try removing it. If you find that the length or contents of Claude Sonnet 5's user-facing updates are not well-calibrated to your use case, explicitly describe what these updates should look like in the prompt and provide examples. + +## More literal instruction following + +Claude Sonnet 5 interprets prompts literally and explicitly, particularly at lower effort levels. It does not silently generalize an instruction from one item to another, and it does not infer requests you didn't make. The upside of this literalism is precision, and it generally performs better for API use cases with carefully tuned prompts, structured extraction, and pipelines where you want predictable behavior. If you need Claude to apply an instruction broadly, state the scope explicitly (for example, "Apply this formatting to every section, not just the first one"). + +## Tone and writing style + +As with any new model, prose style on long-form writing may shift. If your product relies on a specific voice, re-evaluate style prompts against the new baseline. + +For instance, if your product voice is warmer or more conversational, add: + +```text wrap +Use a warm, collaborative tone. Acknowledge the user's framing before answering. +``` + +If you previously relied on `temperature` for stylistic variety, note that setting `temperature`, `top_p`, or `top_k` to a non-default value returns a 400 error on Claude Sonnet 5. This constraint is new for Sonnet-class models. Remove these parameters when migrating, and use system-prompt instructions to guide tone and variety instead. + +## Design and frontend defaults + +Claude Sonnet 5 may settle into a consistent default visual style on open-ended frontend and design briefs. A default house style can read well for some briefs but feel off for dashboards, dev tools, fintech, healthcare, or enterprise apps. + +Generic instructions ("don't use that color," "make it clean and minimal") tend to shift the model to a different fixed palette rather than producing variety. Two approaches work reliably: + +**1. Specify a concrete alternative.** The model follows explicit specs precisely: + +```text wrap +Design a desktop landing page for a supplement brand called AEFRM. + +The visual direction should come from a cold monochrome atmosphere using pale silver-gray tones that gradually deepen into blue-gray and near-black, similar to a misted metallic surface. + +The page should feel sharp and controlled, with a strong sense of structure and restraint. + +Use this tonal system across the full page instead of introducing bright accent colors. + +Use the uploaded image on the hero design in black and white. + +The layout should be built with clear horizontal sections and a centered max-width container. Use 4px corner radius consistently across cards, buttons, inputs, and media frames. Margins should feel generous, with enough empty space around each section so the page breathes. + +Typography should use a square, angular sans-serif with wider letter spacing than usual, especially in headings and navigation, so the text feels more engineered and less compressed. Headline text can be large and uppercase, while supporting copy remains short and sparse. The sub texts should be written with Alumni Sans SC in 4-6px like tiny little texts on corners bottom centre like that. + +For the structure, start with a hero section containing a strong product statement, one short supporting paragraph, and a clean product placeholder or packshot frame. Below that, add a benefit grid with three or four blocks, then a formulation or ingredients section, and finally a cta. + +Buttons should be flat and precise, with subtle hover changes using transition: all 160ms ease out where brightness and border contrast shift slightly rather than using dramatic motion. + +Color palette should stay within this range: +#E9ECEC, #C9D2D4, #8C9A9E, #44545B, #11171B. +``` + +**2. Have the model propose options before building.** This breaks the default and gives users control. Because `temperature` is not accepted on Claude Sonnet 5, this approach is the recommended way to produce meaningfully different design directions across runs. Example prompt: + +```text wrap +Before building, propose 4 distinct visual directions tailored to this brief (each as: bg hex / accent hex / typeface, plus a one-line rationale). Ask the user to pick one, then implement only that direction. +``` + +To steer away from generic patterns that users call the "AI slop" aesthetic, you can include a short directive in your system prompt. The [frontend-design skill](https://github.com/anthropics/claude-code/blob/main/plugins/frontend-design/skills/frontend-design/SKILL.md) provides a fuller treatment, but this snippet works well alongside the preceding variety approaches: + +```text wrap + +NEVER use generic AI-generated aesthetics like overused font families (Inter, Roboto, Arial, system fonts), cliched color schemes (particularly purple gradients on white or dark backgrounds), predictable layouts and component patterns, and cookie-cutter design that lacks context-specific character. Use unique fonts, cohesive colors and themes, and animations for effects and micro-interactions. + +``` + +## Interactive coding products + +Token usage and behavior can differ between autonomous, asynchronous coding agents with a single user turn and interactive, synchronous coding agents with multiple user turns. To maximize both performance and token efficiency in coding products, use `xhigh` or `high` effort, add autonomous features like an auto mode, and reduce the number of human interactions required from your users. + +When limiting the number of required user interactions, it's important to specify the task, intent, and relevant constraints upfront in the first human turn. Providing well-specified, clear, and accurate task descriptions upfront can help maximize autonomy and intelligence while minimizing extra token usage after user turns. In contrast, ambiguous or underspecified prompts conveyed progressively over multiple user turns tend to relatively reduce token efficiency and sometimes performance. + +## Code review harnesses + +If your code-review harness was tuned for an earlier model, you may initially see lower recall on Claude Sonnet 5. This is likely a harness effect, not a capability regression. When a review prompt says things like "only report high-severity issues," "be conservative," or "don't nitpick," Claude Sonnet 5 may follow that instruction more faithfully than earlier models did: it may investigate the code just as thoroughly, identify the bugs, and then not report findings it judges to be below your stated bar. This can show up as the model doing the same depth of investigation but converting fewer investigations into reported findings, especially on lower-severity bugs. Precision typically rises, but measured recall can fall even though the model's underlying bug-finding ability has improved. + +Some recommended prompt language: + +```text wrap +Report every issue you find, including ones you are uncertain about or consider low-severity. Do not filter for importance or confidence at this stage - a separate verification step will do that. Your goal here is coverage: it is better to surface a finding that later gets filtered out than to silently drop a real bug. For each finding, include your confidence level and an estimated severity so a downstream filter can rank them. +``` + +This prompt can be used without having an actual second step, but moving confidence filtering out of the finding step often helps. If your harness has a separate verification, deduplication, or ranking stage, tell the model explicitly that its job at the finding stage is coverage rather than filtering. + +If you do want the model to self-filter in a single pass, be concrete about where the bar is rather than using qualitative terms like "important": for example, "report any bugs that could cause incorrect behavior, a test failure, or a misleading result; only omit nits like pure style or naming preferences." + +Iterate on prompts against a subset of your evals or test cases to validate recall or F1 score gains. + +## Computer use + +Claude Sonnet 5 supports the `computer_20251124` tool version. [Computer use](/docs/en/agents-and-tools/tool-use/computer-use-tool) capability works across resolutions, up to a maximum resolution of 2576px / 3.75MP. Internal computer use testing shows that sending images at 1080p provides a good balance of performance and cost. + +For particularly cost-sensitive workloads, 720p or 1366×768 are lower-cost options with strong performance. Conduct your own testing to find the ideal settings for your use case; experimenting with effort settings can also help tune the model's behavior. diff --git a/content/en/build-with-claude/prompt-engineering/prompting-tools.md b/content/en/build-with-claude/prompt-engineering/prompting-tools.md index 84aa9a2a3..a7f787f79 100644 --- a/content/en/build-with-claude/prompt-engineering/prompting-tools.md +++ b/content/en/build-with-claude/prompt-engineering/prompting-tools.md @@ -1,200 +1,220 @@ # Console prompting tools +Draft, template, and refine prompts in the Claude Console with the prompt generator, prompt templates and variables, and the prompt improver. + --- The Claude Console offers a suite of tools to help you build and refine prompts. This page walks through them in the order you'll typically use them: generating a first draft, adding templates and variables, then improving an existing prompt. ---- +*** ## Prompt generator -The prompt generator is compatible with all Claude models, including those with extended thinking capabilities. For prompting tips specific to extended thinking models, see the [extended thinking prompting tips](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#leverage-thinking-and-interleaved-thinking-capabilities). + The prompt generator is compatible with all Claude models, including those with extended thinking capabilities. For prompting tips specific to extended thinking models, see the [extended thinking prompting tips](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#leverage-thinking-and-interleaved-thinking-capabilities). -Sometimes, the hardest part of using an AI model is figuring out how to prompt it effectively. The prompt generator guides Claude to create high-quality prompt templates tailored to your specific tasks, following many of our prompt engineering best practices. +Sometimes, the hardest part of using an AI model is figuring out how to prompt it effectively. The prompt generator guides Claude to create high-quality prompt templates tailored to your specific tasks, following many of Anthropic's prompt engineering best practices. -The prompt generator is particularly useful for solving the "blank page problem"—it gives you a jumping-off point for further testing and iteration. +The prompt generator is particularly useful for solving the "blank page problem": it gives you a starting point for further testing and iteration. -Try the prompt generator now directly on the [Console](/dashboard). + + Try the prompt generator now directly on the -If you're interested in analyzing the underlying prompt and architecture, check out our [prompt generator Google Colab notebook](https://anthropic.com/metaprompt-notebook/). To run the Colab notebook, you'll need an [API key](/settings/keys). + [Console](/dashboard) ---- + . + + +If you're interested in analyzing the underlying prompt and architecture, see the [prompt generator Google Colab notebook](https://colab.research.google.com/github/anthropics/claude-cookbooks/blob/main/misc/metaprompt.ipynb). To run the Colab notebook, you'll need an [API key](/settings/keys). + +*** ## Prompt templates and variables -When deploying an LLM-based application with Claude, your API calls will typically consist of two types of content: -- **Fixed content:** Static instructions or context that remain constant across multiple interactions -- **Variable content:** Dynamic elements that change with each request or conversation, such as: - - User inputs - - Retrieved content for Retrieval-Augmented Generation (RAG) - - Conversation context such as user account history - - System-generated data such as tool use results fed in from other independent calls to Claude +When deploying an LLM-based application with Claude, your API calls typically consist of two types of content: + +* **Fixed content:** Static instructions or context that remain constant across multiple interactions -A **prompt template** combines these fixed and variable parts, using placeholders for the dynamic content. In the [Claude Console](/), these placeholders are denoted with **\{\{double brackets\}\}**, making them easily identifiable and allowing for quick testing of different values. +* **Variable content:** Dynamic elements that change with each request or conversation, such as: -You should use prompt templates and variables when you expect any part of your prompt to be repeated in another call to Claude (via the API or the [Claude Console](/). [claude.ai](https://claude.ai/) currently does not support prompt templates or variables). + * User inputs + * Retrieved content for Retrieval-Augmented Generation (RAG) + * Conversation context such as user account history + * System-generated data such as tool use results fed in from other independent calls to Claude + +A **prompt template** combines these fixed and variable parts, using placeholders for the dynamic content. In the [Claude Console](/), these placeholders are denoted with **\{\{double brackets}}**, making them easily identifiable and allowing for quick testing of different values. + +You should use prompt templates and variables when you expect any part of your prompt to be repeated in another call to Claude (through the API or the [Claude Console](/). [claude.ai](https://claude.ai/) currently does not support prompt templates or variables). Prompt templates offer several benefits: -- **Consistency:** Ensure a consistent structure for your prompts across multiple interactions -- **Efficiency:** Easily swap out variable content without rewriting the entire prompt -- **Testability:** Quickly test different inputs and edge cases by changing only the variable portion -- **Scalability:** Simplify prompt management as your application grows in complexity -- **Version control:** Easily track changes to your prompt structure over time by keeping tabs only on the core part of your prompt, separate from dynamic inputs + +* **Consistency:** Ensure a consistent structure for your prompts across multiple interactions +* **Efficiency:** Easily swap out variable content without rewriting the entire prompt +* **Testability:** Quickly test different inputs and edge cases by changing only the variable portion +* **Scalability:** Simplify prompt management as your application grows in complexity +* **Version control:** Easily track changes to your prompt structure over time by monitoring only the core part of your prompt, separate from dynamic inputs The Console uses prompt templates and variables to power its tooling: -- **Prompt generator:** Decides what variables your prompt needs and includes them in the template it outputs -- **Prompt improver:** Takes your existing template, including all variables, and maintains them in the improved template it outputs -- **[Evaluation tool](/docs/en/test-and-evaluate/eval-tool):** Allows you to easily test, scale, and track versions of your prompts by separating the variable and fixed portions of your prompt template + +* **Prompt generator:** Determines what variables your prompt needs and includes them in the template it outputs +* **Prompt improver:** Takes your existing template, including all variables, and maintains them in the improved template it outputs +* **[Evaluation tool](/docs/en/test-and-evaluate/eval-tool):** Allows you to easily test, scale, and track versions of your prompts by separating the variable and fixed portions of your prompt template ### Example prompt template -Consider a simple application that translates English text to Spanish. The translated text would be variable since it changes between users or calls to Claude. You might use this prompt template: +Consider a simple application that translates English text to Spanish. The translated text would be variable because it changes between users or calls to Claude. You might use this prompt template: -```text +```text wrap Translate this text from English to Spanish: {{text}} ``` -To level up your prompt variables, wrap them in [XML tags](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#structure-prompts-with-xml-tags) for clearer structure. + + To level up your prompt variables, wrap them in ---- + [XML tags](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#structure-prompts-with-xml-tags) + + for clearer structure. + + +*** ## Prompt improver -The prompt improver is compatible with all Claude models, including those with extended thinking capabilities. For prompting tips specific to extended thinking models, see the [extended thinking prompting tips](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#leverage-thinking-and-interleaved-thinking-capabilities). + The prompt improver is compatible with all Claude models, including those with extended thinking capabilities. For prompting tips specific to extended thinking models, see the [extended thinking prompting tips](/docs/en/build-with-claude/prompt-engineering/claude-prompting-best-practices#leverage-thinking-and-interleaved-thinking-capabilities). The prompt improver helps you quickly iterate and improve your prompts through automated analysis and enhancement. It excels at making prompts more robust for complex tasks that require high accuracy. - ![Image](/docs/images/prompt_improver.png) + ![Claude Console prompt improver showing the four-step improvement modal](/docs/images/prompt_improver.png) ### Before you begin You'll need: -- A prompt template (see [Prompt templates and variables](#prompt-templates-and-variables) above) -- Feedback on current issues with Claude's outputs (optional but recommended) -- Example inputs and ideal outputs (optional but recommended) + +* A prompt template (see [Prompt templates and variables](#prompt-templates-and-variables)) +* Feedback on current issues with Claude's outputs (optional but recommended) +* Example inputs and ideal outputs (optional but recommended) ### How the prompt improver works -The prompt improver enhances your prompts in 4 steps: +The prompt improver enhances your prompts in four steps: -1. **Example identification**: Locates and extracts examples from your prompt template -2. **Initial draft**: Creates a structured template with clear sections and XML tags -3. **Chain of thought refinement**: Adds and refines detailed reasoning instructions -4. **Example enhancement**: Updates examples to demonstrate the new reasoning process +1. **Example identification:** Locates and extracts examples from your prompt template. +2. **Initial draft:** Creates a structured template with clear sections and XML tags. +3. **Chain of thought refinement:** Adds and refines detailed reasoning instructions. +4. **Example enhancement:** Updates examples to demonstrate the new reasoning process. You can watch these steps happen in real-time in the improvement modal. ### What you get The prompt improver generates templates with: -- Detailed chain-of-thought instructions that guide Claude's reasoning process and typically improve its performance -- Clear organization using XML tags to separate different components -- Standardized example formatting that demonstrates step-by-step reasoning from input to output -- Strategic prefills that guide Claude's initial responses + +* Detailed chain-of-thought instructions that guide Claude's reasoning process and typically improve its performance +* Clear organization using XML tags to separate different components +* Standardized example formatting that demonstrates step-by-step reasoning from input to output +* Strategic prefills that guide Claude's initial responses -While examples appear separately in the Workbench UI, they're included at the start of the first user message in the actual API call. View the raw format by clicking "**\<\/\> Get Code**" or insert examples as raw text via the Examples box. + While examples appear separately in the Workbench UI, they're included at the start of the first user message in the actual API call. View the raw format by clicking "**\ Get Code**" or insert examples as raw text through the Examples box. ### How to use the prompt improver -1. Submit your prompt template -2. Add any feedback about issues with Claude's current outputs (e.g., "summaries are too basic for expert audiences") -3. Include example inputs and ideal outputs -4. Review the improved prompt +1. Submit your prompt template. +2. Add any feedback about issues with Claude's current outputs (for example, "summaries are too basic for expert audiences.") +3. Include example inputs and ideal outputs. +4. Review the improved prompt. ### Generate test examples Don't have examples yet? Use the [Test Case Generator](/docs/en/test-and-evaluate/eval-tool#creating-test-cases) to: -1. Generate sample inputs -2. Get Claude's responses -3. Edit the responses to match your ideal outputs -4. Add the polished examples to your prompt + +1. Generate sample inputs. +2. Get Claude's responses. +3. Edit the responses to match your ideal outputs. +4. Add the polished examples to your prompt. ### When to use the prompt improver The prompt improver works best for: -- Complex tasks requiring detailed reasoning -- Situations where accuracy is more important than speed -- Problems where Claude's current outputs need significant improvement + +* Complex tasks requiring detailed reasoning +* Situations where accuracy is more important than speed +* Problems where Claude's current outputs need significant improvement -For latency or cost-sensitive applications, consider using simpler prompts. The prompt improver creates templates that produce longer, more thorough, but slower responses. + For latency or cost-sensitive applications, consider using simpler prompts. The prompt improver creates templates that produce longer, more thorough, but slower responses. ### Example improvement Here's how the prompt improver enhances a basic classification prompt: -
- -```text -From the following list of Wikipedia article titles, identify which article this sentence came from. -Respond with just the article title and nothing else. - -Article titles: -{{titles}} + + ```text wrap + From the following list of Wikipedia article titles, identify which article this sentence came from. + Respond with just the article title and nothing else. -Sentence to classify: -{{sentence}} -``` - -
+ Article titles: + {{titles}} -
+ Sentence to classify: + {{sentence}} + ``` + -```text -You are an intelligent text classification system specialized in matching sentences to Wikipedia article titles. Your task is to identify which Wikipedia article a given sentence most likely belongs to, based on a provided list of article titles. + + ```text wrap + You are an intelligent text classification system specialized in matching sentences to Wikipedia article titles. Your task is to identify which Wikipedia article a given sentence most likely belongs to, based on a provided list of article titles. -First, review the following list of Wikipedia article titles: - -{{titles}} - + First, review the following list of Wikipedia article titles: + + {{titles}} + -Now, consider this sentence that needs to be classified: - -{{sentence}} - + Now, consider this sentence that needs to be classified: + + {{sentence}} + -Your goal is to determine which article title from the provided list best matches the given sentence. Follow these steps: + Your goal is to determine which article title from the provided list best matches the given sentence. Follow these steps: -1. List the key concepts from the sentence -2. Compare each key concept with the article titles -3. Rank the top 3 most relevant titles and explain why they are relevant -4. Select the most appropriate article title that best encompasses or relates to the sentence's content + 1. List the key concepts from the sentence + 2. Compare each key concept with the article titles + 3. Rank the top 3 most relevant titles and explain why they are relevant + 4. Select the most appropriate article title that best encompasses or relates to the sentence's content -Wrap your analysis in tags. Include the following: -- List of key concepts from the sentence -- Comparison of each key concept with the article titles -- Ranking of top 3 most relevant titles with explanations -- Your final choice and reasoning + Wrap your analysis in tags. Include the following: + - List of key concepts from the sentence + - Comparison of each key concept with the article titles + - Ranking of top 3 most relevant titles with explanations + - Your final choice and reasoning -After your analysis, provide your final answer: the single most appropriate Wikipedia article title from the list. - -Output only the chosen article title, without any additional text or explanation. -``` + After your analysis, provide your final answer: the single most appropriate Wikipedia article title from the list. -
+ Output only the chosen article title, without any additional text or explanation. + ``` + Notice how the improved prompt: -- Adds clear step-by-step reasoning instructions -- Uses XML tags to organize content -- Provides explicit output formatting requirements -- Guides Claude through the analysis process + +* Adds clear step-by-step reasoning instructions +* Uses XML tags to organize content +* Provides explicit output formatting requirements +* Guides Claude through the analysis process ### Troubleshooting Common issues and solutions: -- **Examples not appearing in output**: Check that examples are properly formatted with XML tags and appear at the start of the first user message -- **Chain of thought too verbose**: Add specific instructions about desired output length and level of detail -- **Reasoning steps don't match your needs**: Modify the steps section to match your specific use case +* **Examples not appearing in output:** Check that examples are properly formatted with XML tags and appear at the start of the first user message. +* **Chain of thought too verbose:** Add specific instructions about desired output length and level of detail. +* **Reasoning steps don't match your needs:** Modify the steps section to match your specific use case. *** @@ -204,10 +224,12 @@ Common issues and solutions: Learn core techniques with worked examples. + Use the evaluation tool to test your improved prompts. + - An example-filled tutorial that covers the prompt engineering concepts found in our docs. + An example-filled tutorial that covers the prompt engineering concepts found in the docs. - \ No newline at end of file + diff --git a/content/en/claude_api_primer.md b/content/en/claude_api_primer.md new file mode 100644 index 000000000..61c70eeff --- /dev/null +++ b/content/en/claude_api_primer.md @@ -0,0 +1,808 @@ +# API usage primer for Claude + +This guide is designed to give Claude the basics of using the Claude API. It gives explanation and examples of model IDs/the basic messages API, tool use, streaming, extended thinking, and nothing else. + +--- + +# API usage primer for Claude + +> This guide is designed to give Claude the basics of using the Claude API. It gives explanation and examples of model IDs/the basic messages API, tool use, streaming, extended thinking, and nothing else. + +## Models + +```text wrap +Smartest model: Claude Opus 4.8: claude-opus-4-8 +Smart model: Claude Sonnet 5: claude-sonnet-5 +For fast, cost-effective tasks: Claude Haiku 4.5: claude-haiku-4-5-20251001 +``` + +## Calling the API + +### Basic request and response + + + ```bash CLI + ant messages create \ + --model claude-opus-4-8 \ + --max-tokens 1024 \ + --message '{"role": "user", "content": "Hello, Claude"}' + ``` + + ```python Python + import anthropic + import os + + message = anthropic.Anthropic( + api_key=os.environ.get("ANTHROPIC_API_KEY") + ).messages.create( + model="claude-opus-4-8", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude"}], + ) + print(message) + ``` + + +```json Output +{ + "id": "msg_01XFDUDYJgAACzvnptvVoYEL", + "type": "message", + "role": "assistant", + "content": [ + { + "type": "text", + "text": "Hello!" + } + ], + "model": "claude-opus-4-8", + "stop_reason": "end_turn", + "stop_sequence": null, + "usage": { + "input_tokens": 12, + "output_tokens": 6 + } +} +``` + +### Multiple conversational turns + +The Messages API is stateless, which means that you always send the full conversational history to the API. You can use this pattern to build up a conversation over time. Earlier conversational turns don't necessarily need to actually originate from Claude. You can use synthetic `assistant` messages. + + + ```bash CLI + ant messages create <<'YAML' + model: claude-opus-4-8 + max_tokens: 1024 + messages: + - role: user + content: Hello, Claude + - role: assistant + content: Hello! + - role: user + content: Can you describe LLMs to me? + YAML + ``` + + ```python Python + import anthropic + + message = anthropic.Anthropic().messages.create( + model="claude-opus-4-8", + max_tokens=1024, + messages=[ + {"role": "user", "content": "Hello, Claude"}, + {"role": "assistant", "content": "Hello!"}, + {"role": "user", "content": "Can you describe LLMs to me?"}, + ], + ) + print(message) + ``` + + +### Prefilling Claude's response + +You can pre-fill part of Claude's response in the last position of the input messages list. This can be used to shape Claude's response. The following example uses `"max_tokens": 1` to get a single multiple choice answer from Claude. + + + ```bash CLI + ant messages create <<'YAML' + model: claude-sonnet-4-5 + max_tokens: 1 + messages: + - role: user + content: "What is latin for Ant? (A) Apoidea, (B) Rhopalocera, (C) Formicidae" + - role: assistant + content: "The answer is (" + YAML + ``` + + ```python Python + import anthropic + + message = anthropic.Anthropic().messages.create( + model="claude-sonnet-4-5", + max_tokens=1, + messages=[ + { + "role": "user", + "content": "What is latin for Ant? (A) Apoidea, (B) Rhopalocera, (C) Formicidae", + }, + {"role": "assistant", "content": "The answer is ("}, + ], + ) + print(message.content[0].text) + ``` + + +### Vision + +Claude can read both text and images in requests. Both `base64` and `url` source types are supported for images, along with the `image/jpeg`, `image/png`, `image/gif`, and `image/webp` media types. + + + ```bash CLI + IMAGE_URL="https://upload.wikimedia.org/wikipedia/commons/a/a7" + IMAGE_URL="$IMAGE_URL/Camponotus_flavomarginatus_ant.jpg" + + # Option 1: Base64-encoded image (@ prefix auto-encodes binary files as base64) + curl -sSo ant.jpg "$IMAGE_URL" + + ant messages create <<'YAML' + model: claude-opus-4-8 + max_tokens: 1024 + messages: + - role: user + content: + - type: image + source: + type: base64 + media_type: image/jpeg + data: "@./ant.jpg" + - type: text + text: What is in the above image? + YAML + + # Option 2: URL-referenced image + ant messages create < + +## Extended thinking + +Extended thinking can sometimes help Claude with very hard tasks. On models before Claude Opus 4.7, temperature must be set to 1 when extended thinking is enabled. + +Extended thinking is supported in the following models: + +* Claude Opus 4.8 (claude-opus-4-8, adaptive thinking only) +* Claude Opus 4.7 (`claude-opus-4-7`) +* Claude Opus 4.6 (`claude-opus-4-6`) +* Claude Opus 4.5 (`claude-opus-4-5-20251101`) +* Claude Sonnet 4.6 (`claude-sonnet-4-6`) +* Claude Sonnet 4.5 (`claude-sonnet-4-5-20250929`) +* Claude Haiku 4.5 (`claude-haiku-4-5-20251001`) + + + On Claude Opus 4.8 and Claude Opus 4.7, manual extended thinking (`type: enabled` with a `budget_tokens` value) is not supported and returns a 400 error. Use [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) (`type: adaptive`) instead. + + +### How extended thinking works + +When extended thinking is turned on, Claude creates `thinking` content blocks where it outputs its internal reasoning. The API response includes `thinking` content blocks, followed by `text` content blocks. + + + ```bash CLI + ant messages create \ + --transform content --format yaml <<'YAML' + model: claude-opus-4-8 + max_tokens: 16000 + thinking: + type: adaptive + display: summarized + messages: + - role: user + content: Are there an infinite number of prime numbers such that n mod 4 == 3? + YAML + ``` + + ```python Python + import anthropic + + client = anthropic.Anthropic() + + response = client.messages.create( + model="claude-opus-4-8", + max_tokens=16000, + thinking={"type": "adaptive", "display": "summarized"}, + messages=[ + { + "role": "user", + "content": "Are there an infinite number of prime numbers such that n mod 4 == 3?", + } + ], + ) + + # The response will contain summarized thinking blocks and text blocks + for block in response.content: + if block.type == "thinking": + print(f"\nThinking summary: {block.thinking}") + elif block.type == "text": + print(f"\nResponse: {block.text}") + ``` + + +When using manual extended thinking (`type: enabled`), the `budget_tokens` parameter determines the maximum number of tokens Claude is allowed to use for its internal reasoning process. In Claude 4 and later models, this limit applies to full thinking tokens, and not to the summarized output. Larger budgets can improve response quality by enabling more thorough analysis for complex problems. Unless you are using [interleaved thinking](#interleaved-thinking), `budget_tokens` must be less than `max_tokens` so that Claude has space to write its response after thinking is complete. + +## Extended thinking with tool use + +Extended thinking can be used alongside tool use, allowing Claude to reason through tool selection and results processing. + +Important limitations: + +1. **Tool choice limitation:** Only supports `tool_choice: {"type": "auto"}` (default) or `tool_choice: {"type": "none"}`. +2. **Preserving thinking blocks:** During tool use, you must pass `thinking` blocks back to the API for the last assistant message. + +### Preserving thinking blocks + + + ```bash CLI + # First request: capture the assistant content array (thinking + tool_use + # blocks, signatures intact) as compact JSON. + ASSISTANT_CONTENT=$(ant messages create \ + --transform content --format jsonl <<'YAML' + model: claude-opus-4-8 + max_tokens: 16000 + thinking: + type: adaptive + display: summarized + tools: + - name: get_weather + description: Get the current weather for a location. + input_schema: + type: object + properties: + location: + type: string + description: The city name. + required: [location] + messages: + - role: user + content: "What's the weather in Paris?" + YAML + ) + + TOOL_USE_ID=$(printf '%s' "$ASSISTANT_CONTENT" \ + | jq -r '.[] | select(.type == "tool_use") | .id') + + # Second request: pass the captured blocks back unchanged as the assistant + # message. The thinking block must accompany the tool_use block. + ant messages create < + +### Interleaved thinking + +Extended thinking with tool use in Claude 4 models supports interleaved thinking, which enables Claude to think between tool calls. To enable on Claude 4, 4.5, and Sonnet 4.6 models, add the beta header `interleaved-thinking-2025-05-14` to your API request. + + + ```bash CLI + ant beta:messages create --beta interleaved-thinking-2025-05-14 <<'YAML' + model: claude-sonnet-4-6 + max_tokens: 16000 + thinking: + type: enabled + budget_tokens: 10000 + tools: + - name: calculator + description: Perform arithmetic calculations. + input_schema: + type: object + properties: + expression: + type: string + description: The math expression to evaluate. + required: + - expression + - name: database_query + description: Query the product database. + input_schema: + type: object + properties: + query: + type: string + description: The database query. + required: + - query + messages: + - role: user + content: "What's the total revenue if we sold 150 units of product A at $50 each?" + YAML + ``` + + ```python Python + import anthropic + + client = anthropic.Anthropic() + + calculator_tool = { + "name": "calculator", + "description": "Perform arithmetic calculations.", + "input_schema": { + "type": "object", + "properties": { + "expression": { + "type": "string", + "description": "The math expression to evaluate.", + } + }, + "required": ["expression"], + }, + } + + database_tool = { + "name": "database_query", + "description": "Query the product database.", + "input_schema": { + "type": "object", + "properties": { + "query": {"type": "string", "description": "The database query."} + }, + "required": ["query"], + }, + } + + response = client.beta.messages.create( + model="claude-sonnet-4-6", + max_tokens=16000, + thinking={"type": "enabled", "budget_tokens": 10000}, + tools=[calculator_tool, database_tool], + messages=[ + { + "role": "user", + "content": "What's the total revenue if we sold 150 units of product A at $50 each?", + } + ], + betas=["interleaved-thinking-2025-05-14"], + ) + + for block in response.content: + if block.type == "thinking": + print(f"Thinking: {block.thinking}") + elif block.type == "tool_use": + print(f"Tool call: {block.name}({block.input})") + elif block.type == "text": + print(f"Response: {block.text}") + ``` + + +With interleaved thinking and ONLY with interleaved thinking (not regular extended thinking), the `budget_tokens` can exceed the `max_tokens` parameter, as `budget_tokens` in this case represents the total budget across all thinking blocks within one assistant turn. + + + For Claude Opus 4.8, Claude Opus 4.7, and Claude Opus 4.6, interleaved thinking is automatically enabled when using [adaptive thinking](/docs/en/build-with-claude/adaptive-thinking) (`thinking: {type: "adaptive"}`). No beta header is needed. Sonnet 4.6 supports both the `interleaved-thinking-2025-05-14` beta header with manual extended thinking and adaptive thinking. + + +## Tool use + +### Specifying client tools + +Client tools are specified in the `tools` top-level parameter of the API request. Each tool definition includes: + +| Parameter | Description | +| -------------- | --------------------------------------------------------------------------------------------------- | +| `name` | The name of the tool. Must match the regex `^[a-zA-Z0-9_-]{1,64}$`. | +| `description` | A detailed plaintext description of what the tool does, when it should be used, and how it behaves. | +| `input_schema` | A [JSON Schema](https://json-schema.org/) object defining the expected parameters for the tool. | + +```json +{ + "name": "get_weather", + "description": "Get the current weather in a given location", + "input_schema": { + "type": "object", + "properties": { + "location": { + "type": "string", + "description": "The city and state, e.g. San Francisco, CA" + }, + "unit": { + "type": "string", + "enum": ["celsius", "fahrenheit"], + "description": "The unit of temperature, either 'celsius' or 'fahrenheit'" + } + }, + "required": ["location"] + } +} +``` + +### Best practices for tool definitions + +**Provide extremely detailed descriptions.** This is by far the most important factor in tool performance. Your descriptions should explain every detail about the tool, including: + +* What the tool does +* When it should be used (and when it shouldn't) +* What each parameter means and how it affects the tool's behavior +* Any important caveats or limitations + +**Consider using `input_examples` for complex tools.** For tools with nested objects, optional parameters, or format-sensitive inputs, you can provide concrete examples using the `input_examples` field (beta). This helps Claude understand expected input patterns. See [Providing tool use examples](/docs/en/agents-and-tools/tool-use/define-tools#providing-tool-use-examples) for details. + +Example of a good tool description: + +```json +{ + "name": "get_stock_price", + "description": "Retrieves the current stock price for a given ticker symbol. The ticker symbol must be a valid symbol for a publicly traded company on a major US stock exchange like NYSE or NASDAQ. The tool will return the latest trade price in USD. It should be used when the user asks about the current or most recent price of a specific stock. It will not provide any other information about the stock or company.", + "input_schema": { + "type": "object", + "properties": { + "ticker": { + "type": "string", + "description": "The stock ticker symbol, e.g. AAPL for Apple Inc." + } + }, + "required": ["ticker"] + } +} +``` + +## Controlling Claude's output + +### Forcing tool use + +You can force Claude to use a specific tool by specifying the tool in the `tool_choice` field: + +```python +tool_choice = {"type": "tool", "name": "get_weather"} +``` + +When working with the `tool_choice` parameter, there are four possible options: + +* `auto` allows Claude to decide whether to call any provided tools or not (default). +* `any` tells Claude that it must use one of the provided tools. +* `tool` forces Claude to always use a particular tool. +* `none` prevents Claude from using any tools. + +### JSON output + +Tools do not necessarily need to be client functions. You can use tools anytime you want the model to return JSON output that follows a provided schema. + +### Chain of thought + +When using tools, Claude often shows its "chain of thought," that is, the step-by-step reasoning it uses to break down the problem and decide which tools to use. + +```json +{ + "role": "assistant", + "content": [ + { + "type": "text", + "text": "To answer this question, I will: 1. Use the get_weather tool to get the current weather in San Francisco. 2. Use the get_time tool to get the current time in the America/Los_Angeles timezone, which covers San Francisco, CA." + }, + { + "type": "tool_use", + "id": "toolu_01A09q90qw90lq917835lq9", + "name": "get_weather", + "input": { "location": "San Francisco, CA" } + } + ] +} +``` + +### Parallel tool use + +By default, Claude may use multiple tools to answer a user query. You can disable this behavior by setting `disable_parallel_tool_use=true`. + +## Handling tool use and tool result content blocks + +### Handling results from client tools + +The response has a `stop_reason` of `tool_use` and one or more `tool_use` content blocks that include: + +* `id`: A unique identifier for this particular tool use block. +* `name`: The name of the tool being used. +* `input`: An object containing the input being passed to the tool. + +When you receive a tool use response, you should: + +1. Extract the `name`, `id`, and `input` from the `tool_use` block. +2. Run the actual tool in your code base corresponding to that tool name. +3. Continue the conversation by sending a new message with a `tool_result`: + +```json +{ + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "toolu_01A09q90qw90lq917835lq9", + "content": "15 degrees" + } + ] +} +``` + +### Handling the `max_tokens` stop reason + +If Claude's response is cut off because it hits the `max_tokens` limit during tool use, retry the request with a higher `max_tokens` value. + +### Handling the `pause_turn` stop reason + +When using server tools such as web search, the API may return a `pause_turn` stop reason. Continue the conversation by passing the paused response back as-is in a subsequent request. + +## Troubleshooting errors + +### Tool execution error + +If the tool itself throws an error during execution, return the error message with `"is_error": true`: + +```json +{ + "role": "user", + "content": [ + { + "type": "tool_result", + "tool_use_id": "toolu_01A09q90qw90lq917835lq9", + "content": "ConnectionError: the weather service API is not available (HTTP 500)", + "is_error": true + } + ] +} +``` + +### Invalid tool name + +If Claude's attempted use of a tool is invalid (for example, missing required parameters), try the request again with more-detailed `description` values in your tool definitions. + +## Streaming messages + +When creating a Message, you can set `"stream": true` to incrementally stream the response using server-sent events (SSE). + +### Streaming with SDKs + + + ```bash CLI + ant messages create --stream --format jsonl \ + --model claude-opus-4-8 \ + --max-tokens 1024 \ + --message '{role: user, content: "Hello"}' \ + | jq -rj 'select(.delta.type? == "text_delta") | .delta.text' + ``` + + ```python Python + import anthropic + + client = anthropic.Anthropic() + + with client.messages.stream( + max_tokens=1024, + messages=[{"role": "user", "content": "Hello"}], + model="claude-opus-4-8", + ) as stream: + for text in stream.text_stream: + print(text, end="", flush=True) + ``` + + +### Event types + +Each server-sent event includes a named event type and associated JSON data. Each stream uses the following event flow: + +1. `message_start`: contains a `Message` object with empty `content`. +2. A series of content blocks, each with `content_block_start`, one or more `content_block_delta` events, and `content_block_stop`. +3. One or more `message_delta` events, indicating top-level changes to the final `Message` object. +4. A final `message_stop` event. + +**Warning:** The token counts shown in the `usage` field of the `message_delta` event are *cumulative*. + +### Content block delta types + +#### Text delta + +```json +{ + "type": "content_block_delta", + "index": 0, + "delta": { "type": "text_delta", "text": "Hello frien" } +} +``` + +#### Input JSON delta + +For `tool_use` content blocks, deltas are *partial JSON strings*: + +```json +{"type": "content_block_delta","index": 1,"delta": {"type": "input_json_delta","partial_json": "{\"location\": \"San Fra"}}} +``` + +#### Thinking delta + +When using extended thinking with streaming: + +```json +{ + "type": "content_block_delta", + "index": 0, + "delta": { + "type": "thinking_delta", + "thinking": "Let me solve this step by step..." + } +} +``` + +### Basic streaming request example + +```sse +event: message_start +data: {"type": "message_start", "message": {"id": "msg_1nZdL29xx5MUA1yADyHTEsnR8uuvGzszyY", "type": "message", "role": "assistant", "content": [], "model": "claude-opus-4-8", "stop_reason": null, "stop_sequence": null, "usage": {"input_tokens": 25, "output_tokens": 1}}} + +event: content_block_start +data: {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}} + +event: content_block_delta +data: {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "Hello"}} + +event: content_block_delta +data: {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "!"}} + +event: content_block_stop +data: {"type": "content_block_stop", "index": 0} + +event: message_delta +data: {"type": "message_delta", "delta": {"stop_reason": "end_turn", "stop_sequence":null}, "usage": {"output_tokens": 15}} + +event: message_stop +data: {"type": "message_stop"} +``` diff --git a/content/en/cli-sdks-libraries/cli/authentication.md b/content/en/cli-sdks-libraries/cli/authentication.md new file mode 100644 index 000000000..3f7825880 --- /dev/null +++ b/content/en/cli-sdks-libraries/cli/authentication.md @@ -0,0 +1,151 @@ +# CLI authentication options + +Authenticate the ant CLI with interactive login, API keys, named profiles, and Workload Identity Federation. + +--- + +The `ant` CLI supports several credential sources. The [Quickstart](/docs/en/cli-sdks-libraries/cli/quickstart#authentication) covers the one-command happy path (`ant auth login`). This page covers every option in full. + +## Interactive login + +`ant auth login` lets you call the API without creating or managing an API key. It opens a browser-based OAuth flow against the Claude Console and stores the resulting credentials under `$ANTHROPIC_CONFIG_DIR` (see [Configuration directory](/docs/en/manage-claude/wif-reference#configuration-directory) for the OS-specific default). On a remote host or in any environment without a local browser, pass `--no-browser` to print the authorize URL and paste the returned code back into the terminal. + +```bash CLI +ant auth login + +# On a remote host without a browser: +ant auth login --no-browser + +# Bind to a specific workspace and skip the browser picker: +ant auth login --workspace-id wrkspc_01... + +# If the named profile you pass with --profile doesn't exist, +# a new named profile will be created with that name. +ant auth login --profile +``` + +During the browser flow, you select an organization and then a [workspace](/docs/en/manage-claude/workspaces). The issued token is [scoped to that workspace](/docs/en/manage-claude/workspaces#api-keys-and-resource-scoping), so the CLI can only see resources that belong to it. Pass `--workspace-id` to bind directly and skip the picker. To work in more than one workspace, see [Switch between workspaces](#switch-between-workspaces). + +Interactive login is intended for local development and scripting on your own machine. For non-interactive workloads such as CI, servers, and containers, use [Workload Identity Federation](/docs/en/manage-claude/workload-identity-federation) instead. + +Login writes credentials to `credentials/.json`. The first login for a profile also creates `configs/.json` and sets it as the active profile. To remove stored credentials, run `ant auth logout`, or `ant auth logout --all` to clear every profile. + +## Admin access + +By default, `ant auth login` requests a workspace-scoped token. To manage the resources documented on the [Admin API](/docs/en/manage-claude/admin-api) page, request the `org:admin` scope under a dedicated profile: + +```bash CLI +ant auth login --profile admin --scope "org:admin" + +# Print a bearer token for Authorization headers: +ant auth print-credentials --profile admin --access-token +``` + +The `org:admin` scope is granted only to organization members with the admin, owner, or primary owner role. The issued token has organization-wide access, and any workspace binding on the profile does not constrain it. Keep the admin profile separate from your day-to-day profile so routine commands never run with elevated access. + +## API key + +The CLI also reads your API key from the `ANTHROPIC_API_KEY` environment variable. Get a key from the [Claude Console](https://platform.claude.com/settings/keys). + + + + ```bash + echo 'export ANTHROPIC_API_KEY=sk-ant-api03-...' >> ~/.zshrc + source ~/.zshrc + ``` + + + + ```bash + echo 'export ANTHROPIC_API_KEY=sk-ant-api03-...' >> ~/.bashrc + source ~/.bashrc + ``` + + + + ```powershell + setx ANTHROPIC_API_KEY "sk-ant-api03-..." + ``` + + Open a new terminal for the change to take effect. + + + +To override the key for a single invocation, pass `--api-key`. To point at a different API host, set `ANTHROPIC_BASE_URL` or pass `--base-url`. + +## Check authentication status + +`ant auth status` prints the credential source the CLI selected (API key environment variable, OAuth login, federation, or profile), the active profile, the workspace the active token is bound to, and the configuration directory paths. Use it to diagnose why a workload picked the wrong credential or workspace. + +```bash CLI +ant auth status +``` + +```text +Active profile: default +Config dir: ~/.config/anthropic +Profile config: ~/.config/anthropic/configs/default.json +Credentials: ~/.config/anthropic/credentials/default.json + +Credentials + (active) * Profile (user_oauth) [via active_config] sk-ant-oat01-EXA... +... + +Workspace + (active) * Workspace wrkspc_01... (Engineering) +``` + +Read the `(active)` rows to see which credential source and workspace won. The command reports status rather than performing a health check, so don't script against the exit status. For the full ordering of credential sources, see [Credential precedence](/docs/en/manage-claude/wif-reference#credential-precedence). + +## Switch between workspaces + +An interactive-login token is bound to a single workspace. To use the CLI against more than one workspace, log in to each under its own named profile, then switch between them: + +```bash CLI +# 1. Create the profile (interactive; pick the other workspace in the +# browser, or pass --workspace-id to skip the picker): +# ant auth login --profile other-ws + +# 2. Make it the default for subsequent commands: +ant profile activate other-ws + +# 3. Or select it for a single command without changing the default: +ant --profile other-ws models list +ANTHROPIC_PROFILE=other-ws ant models list +``` + +Run [`ant auth status`](#check-authentication-status) to confirm which profile and workspace are active. + + + Profiles are only consulted when no API key is set. If `ANTHROPIC_API_KEY` is present in your environment, it overrides every profile and these commands all use whatever workspace that key is scoped to. Unset it before switching profiles. + + +## Manage profiles + +The `ant profile` subcommands inspect and edit profile state directly: + +```bash CLI +ant profile list +ant profile get --profile other-ws +ant profile set workspace_id wrkspc_01... --profile other-ws +``` + +The writable keys for `ant profile set` are `workspace_id`, `base_url`, `organization_id`, `scope`, `client_id`, and `console_url`. Setting `workspace_id` records the target workspace in the profile config but does not rebind credentials that were already issued; run `ant auth login` again under that profile to mint a token for the new workspace. + +For the profile file schema and the federation block, see [Profile configuration file](/docs/en/manage-claude/wif-reference#profile-configuration-file). For Workload Identity Federation, see the [Authentication overview](/docs/en/manage-claude/authentication) and the [WIF reference](/docs/en/manage-claude/wif-reference). + +## Next steps + + + + Command structure, output formats, GJSON transforms, and request bodies + + + + Version-control API resources, scripting patterns, and use from Claude Code + + + + Non-interactive authentication for CI, servers, and containers + + diff --git a/content/en/cli-sdks-libraries/cli/quickstart.md b/content/en/cli-sdks-libraries/cli/quickstart.md new file mode 100644 index 000000000..be82b179b --- /dev/null +++ b/content/en/cli-sdks-libraries/cli/quickstart.md @@ -0,0 +1,155 @@ +# CLI quickstart + +Install the ant command-line tool, authenticate, and send your first request to the Claude API. + +--- + +The `ant` CLI provides access to the Claude API from your terminal. Every API resource is exposed as a subcommand, with output formatting, response filtering, and YAML or JSON file input. + + + [](/docs/videos/ant-cli-demo.webm) + + +Compared to `curl`, `ant` builds request bodies from typed flags or piped YAML instead of hand-written JSON, and inlines file contents into string fields with an `@path` reference. It extracts response fields with a built-in `--transform` query, so you don't need a separate tool such as `jq`, and it paginates list endpoints automatically. + + + For endpoint-specific parameters and response schemas, see the [API reference](/docs/en/api/cli/messages/create). This page gets you to a working command. For everything else the CLI does, see [Using the CLI](/docs/en/cli-sdks-libraries/cli/using) and [CLI scripting and automation](/docs/en/cli-sdks-libraries/cli/scripting). + + +## Installation + + + + ```bash + brew install anthropics/tap/ant + ``` + + + + For Linux environments, download the release binary directly. + + ```bash + VERSION=1.17.0 + OS=$(uname -s | tr '[:upper:]' '[:lower:]') + case $(uname -m) in + x86_64) ARCH=amd64 ;; + aarch64) ARCH=arm64 ;; + esac + curl -fsSL "https://github.com/anthropics/anthropic-cli/releases/download/v${VERSION}/ant_${VERSION}_${OS}_${ARCH}.tar.gz" \ + | sudo tar -xz -C /usr/local/bin ant + ``` + + You can find all releases on the [GitHub releases page](https://github.com/anthropics/anthropic-cli/releases). + + + + You can also install the CLI from source using `go install`. Requires Go 1.22 or later. + + ```bash + go install github.com/anthropics/anthropic-cli/cmd/ant@latest + ``` + + The binary is placed in `$(go env GOPATH)/bin`. Add it to your `PATH` if it isn't already: + + ```bash + export PATH="$PATH:$(go env GOPATH)/bin" + ``` + + + +Check the installation: + +```bash +ant --version +``` + +## Authentication + +`ant auth login` opens a browser-based OAuth flow against the Claude Console and stores the resulting credentials locally, so you can call the API without creating or managing an API key. + +```bash CLI +ant auth login +``` + + + For other ways to authenticate (API key environment variable, headless hosts, multiple workspaces, named profiles, and Workload Identity Federation), see [CLI authentication options](/docs/en/cli-sdks-libraries/cli/authentication). + + +## Send your first request + +With the binary installed and authenticated, call the [Messages API](/docs/en/api/cli/messages/create): + +```bash +ant messages create \ + --model claude-opus-4-8 \ + --max-tokens 1024 \ + --message '{role: user, content: "Hello, Claude"}' +``` + +```text Output wrap +{ + "model": "claude-opus-4-8", + "id": "msg_01YMmR5XodC5nTqMxLZMKaq6", + "type": "message", + "role": "assistant", + "content": [ + { + "type": "text", + "text": "Hello! How are you doing today? Is there something I can help you with?" + } + ], + "stop_reason": "end_turn", + "usage": { "input_tokens": 27, "output_tokens": 20 /*, ... */ } +} +``` + +The response is the full API object, pretty-printed because stdout is a terminal. + +## Shell completion + +The CLI ships completion scripts for bash, zsh, fish, and PowerShell. Generate and install one for your shell: + + + + ```bash + ant @completion zsh > "${fpath[1]}/_ant" + # Restart your shell or run: autoload -U compinit && compinit + ``` + + + + ```bash + ant @completion bash > /etc/bash_completion.d/ant + ``` + + + + ```bash + ant @completion fish > ~/.config/fish/completions/ant.fish + ``` + + + + ```powershell + ant @completion powershell | Out-String | Invoke-Expression + # To persist across sessions: + # ant @completion powershell >> $PROFILE + ``` + + + +## Next steps + + + + API keys, headless hosts, multiple workspaces, and named profiles + + + + Command structure, output formats, GJSON transforms, and request bodies + + + + Version-control API resources, scripting patterns, and use from Claude Code + + diff --git a/content/en/cli-sdks-libraries/cli/scripting.md b/content/en/cli-sdks-libraries/cli/scripting.md new file mode 100644 index 000000000..461f3caee --- /dev/null +++ b/content/en/cli-sdks-libraries/cli/scripting.md @@ -0,0 +1,181 @@ +# CLI scripting and automation + +Version-control API resources as YAML, chain ant CLI commands in scripts, and operate on resources from Claude Code. + +--- + +This page covers task-oriented workflows built on the `ant` CLI. For the underlying flags and output options, see [Using the CLI](/docs/en/cli-sdks-libraries/cli/using). + +## Version-controlling API resources + +You can use the CLI to version control API resources such as skills, agents, environments, or deployments as YAML files in your repository and keep them in sync with the Claude API. + + + For more information on these resources, see [Managed Agents](/docs/en/managed-agents/overview). + + + + + Write the agent definition to `summarizer.agent.yaml`: + + ```yaml summarizer.agent.yaml + name: Summarizer + model: claude-opus-4-8 + system: | + You are a helpful assistant that writes concise summaries. + tools: + - type: agent_toolset_20260401 + ``` + + + + ```bash + ant beta:agents create < summarizer.agent.yaml + ``` + + ```json Output + { + "id": "agent_011CYm1BLqPXpQRk5khsSXrs", + "version": 1, + "name": "Summarizer", + "model": "claude-opus-4-8" + /* ... */ + } + ``` + + Note the `id` from the response. You'll pass it to the session create command in a later step. + + + Check `summarizer.agent.yaml` into your repository and keep it in sync with the API in your CI pipeline. The update command needs the agent ID and current version as flags: + + ```bash CLI + ant beta:agents update --agent-id agent_011CYm1BLqPXpQRk5khsSXrs --version 1 < summarizer.agent.yaml + ``` + + + + + A session runs in an [environment](/docs/en/api/cli/beta/environments), which defines the sandbox it executes in. Write the environment definition to `summarizer.environment.yaml`: + + ```yaml summarizer.environment.yaml + name: summarizer-env + config: + type: cloud + networking: + type: unrestricted + ``` + + + + ```bash + ant beta:environments create < summarizer.environment.yaml + ``` + + ```json Output + { + "id": "env_01595EKxaaTTGwwY3kyXdtbs", + "name": "summarizer-env" + /* ... */ + } + ``` + + Note the `id` from the response. You'll pass it to the session create command in a later step. + + + Check `summarizer.environment.yaml` into your repository and keep it in sync with the API in your CI pipeline. The update command needs the environment ID as a flag: + + ```bash CLI + ant beta:environments update --environment-id env_01595EKxaaTTGwwY3kyXdtbs < summarizer.environment.yaml + ``` + + + + + Paste the agent `id` and environment `id` from the previous outputs into the session create command: + + ```bash + ant beta:sessions create \ + --agent agent_011CYm1BLqPXpQRk5khsSXrs \ + --environment-id env_01595EKxaaTTGwwY3kyXdtbs \ + --title "Summarization task" + ``` + + ```json Output + { + "id": "session_01JZCh78XvmxJjiXVy3oSi7K", + "status": "running" + /* ... */ + } + ``` + + + + Copy the session `id` from the previous output into `--session-id`: + + ```bash + ant beta:sessions:events send \ + --session-id session_01JZCh78XvmxJjiXVy3oSi7K \ + --event '{type: user.message, content: [{type: text, text: "Summarize the benefits of type safety in one sentence."}]}' + ``` + + + + `--transform` runs against each listed event, so this prints the text of every message in order. `--format auto` overrides the interactive explorer that list commands open by default in a terminal: + + ```bash + ant beta:sessions:events list \ + --session-id session_01JZCh78XvmxJjiXVy3oSi7K \ + --transform 'content.0.text' --format auto --raw-output + ``` + + ```text Output wrap + Summarize the benefits of type safety in one sentence. + Type safety catches errors at compile time rather than runtime, reducing bugs, improving code clarity, enabling better tooling support, and making codebases easier to maintain and refactor with confidence. + ``` + + + To watch a session as it runs, use `ant beta:sessions:events stream --session-id session_01JZCh78XvmxJjiXVy3oSi7K`. Events are written to stdout as they arrive. + + + + +## Scripting patterns + +The CLI is designed to compose with standard shell tooling. + +### Chain list output into a second command + +`--transform id --raw-output` on a list endpoint emits one bare ID per line, so standard tools such as `head` and `xargs` apply directly. Capture the first result, then pass it to a follow-up command: + +```bash +FIRST_AGENT=$(ant beta:agents list \ + --transform id --raw-output | head -1) + +ant beta:agents:versions list \ + --agent-id "$FIRST_AGENT" \ + --transform "{version,created_at}" --format jsonl +``` + +### Inspect errors + +The `--transform-error` and `--format-error` flags apply the same filtering to error responses. `--raw-output` does not apply to errors, so use `--format-error yaml` for an unquoted scalar. Extract only the error message: + +```bash +ant beta:agents retrieve --agent-id bogus \ + --transform-error error.message --format-error yaml 2>&1 +``` + +```text Output wrap +GET "https://api.anthropic.com/v1/agents/bogus?beta=true": 404 Not Found +Agent not found. +``` + +## Use the CLI from Claude Code + +[Claude Code](https://code.claude.com/docs/en/overview) can use the `ant` CLI out of the box. With the CLI installed and authenticated, you can ask Claude Code to operate on your API resources directly. For example: + +* "List my recent agent sessions and summarize which ones errored." +* "Upload every PDF in `./reports` to the Files API and print the resulting IDs." +* "Pull the events for session `session_01...` and tell me where the agent got stuck." + +Claude Code shells out to `ant`, parses the structured output, and reasons over the results (no custom integration code required). diff --git a/content/en/cli-sdks-libraries/cli/using.md b/content/en/cli-sdks-libraries/cli/using.md new file mode 100644 index 000000000..6b53dcf5a --- /dev/null +++ b/content/en/cli-sdks-libraries/cli/using.md @@ -0,0 +1,216 @@ +# Using the CLI + +Command structure, output formats, GJSON transforms, request bodies, and debugging for the ant CLI. + +--- + +This page covers the `ant` CLI's input and output mechanics that apply across every endpoint. For installing and authenticating, see the [Quickstart](/docs/en/cli-sdks-libraries/cli/quickstart). For chaining commands and version-controlling resources, see [CLI scripting and automation](/docs/en/cli-sdks-libraries/cli/scripting). + +## Command structure + +Commands follow a `resource action` pattern. Nested resources use colons: + +```text wrap +ant [:] [flags] +``` + +Run `ant --help` for the full resource list, or append `--help` to any subcommand for its flags. + +Resources in beta (including agents, sessions, deployments, environments, and skills) live under the `beta:` prefix. Commands in this namespace automatically send the appropriate `anthropic-beta` header for that resource, so you don't need to pass it yourself. Use `--beta
` only to override the default (for example, to opt into a different schema version). + +```bash +ant models list +ant messages create --model claude-opus-4-8 --max-tokens 1024 ... +ant beta:agents retrieve --agent-id agent_01... +ant beta:sessions:events list --session-id session_01... +``` + +### Global flags + +| Flag | Description | +| ------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `--profile` | Named profile to use for this invocation (equivalent to setting `ANTHROPIC_PROFILE`). See [Switch between workspaces](/docs/en/cli-sdks-libraries/cli/authentication#switch-between-workspaces). | +| `--format` | Output format: `auto`, `json`, `jsonl`, `yaml`, `pretty`, `raw`, `explore` | +| `--transform` | Filter or reshape the response with a [GJSON path](#transform-output-with-gjson) | +| `-r`, `--raw-output` | Print string results without surrounding quotes, like `jq -r` | +| `--base-url` | Override the API base URL | +| `--debug` | Print full HTTP request and response to stderr | +| `--format-error`, `--transform-error` | Same as `--format` and `--transform` but applied to [error responses](/docs/en/cli-sdks-libraries/cli/scripting#inspect-errors) | + +## Output formats + +`auto` pretty-prints JSON and is the default for commands that create or modify resources. List and retrieve commands default to the [interactive explorer](#interactive-explorer) when writing to a terminal, and to pretty-printed JSON when piped. Override either default with `--format`: + +```bash +ant models retrieve --model-id claude-opus-4-8 --format yaml +``` + +```yaml Output +type: model +id: claude-opus-4-8 +display_name: Claude Opus 4.8 +created_at: "2026-02-04T00:00:00Z" +... +``` + +List endpoints auto-paginate. In the default formats each item is written separately (one compact JSON object per line in `jsonl` mode, a stream of YAML documents in `yaml` mode), which streams cleanly into `head`, `grep`, and `--transform` filters. + +### Interactive explorer + +The explorer is a fold-and-search TUI for browsing large responses. Arrow keys expand and collapse nodes, `/` searches, `q` exits. List and retrieve commands open it by default when connected to a terminal. Pass `--format explore` to open it explicitly: + +```bash +ant models list --format explore +``` + +## Transform output with GJSON + +Use `--transform` to reshape responses before printing. The expression is a [GJSON path](https://github.com/tidwall/gjson/blob/master/SYNTAX.md). For list endpoints the transform runs against each item individually, not the envelope: + +```bash +ant beta:agents list \ + --transform "{id,name,model}" \ + --format jsonl +``` + +```jsonl Output +{"id": "agent_011CYm1BLqPX...", "name": "Docs CLI Test Agent", "model": "claude-opus-4-8"} +{"id": "agent_011CYkVwfaEt...", "name": "Coffee Making Assistant", "model": "claude-opus-4-8"} +{"id": "agent_011CYixHhtUP...", "name": "Coding Assistant", "model": "claude-opus-4-8"} +``` + +### Extract a scalar + +To capture a single field as an unquoted string (for example, the ID of a newly created resource), pair `--transform` with `--raw-output`. The result prints without JSON quotes and is ready to assign to a shell variable: + +```bash +AGENT_ID=$(ant beta:agents create \ + --name "My Agent" \ + --model '{id: claude-opus-4-8}' \ + --transform id --raw-output) + +printf '%s\n' "$AGENT_ID" +``` + +```text Output wrap +agent_011CYm1BLqPXpQRk5khsSXrs +``` + + + `--raw-output` is distinct from `--format raw`. `--raw-output` strips JSON quotes from string results, like `jq -r`. `--format raw` prints the response body's raw JSON bytes without auto-paginating; on list endpoints it applies `--transform` to the pagination envelope rather than to each item. + + +## Passing request bodies + +The right input mechanism depends on the shape of the data: use **flags** for scalar fields and short structured values, pipe a **stdin** document for nested or multiline bodies, and use **`@file` references** to pull file contents into any string or binary field. + +### Flags + +Scalar fields map directly to flags. Structured fields accept a relaxed YAML-like syntax (unquoted keys, optional quotes around strings) or strict JSON: + +```bash +ant beta:sessions create \ + --agent '{type: agent, id: agent_011CYm1BLqPXpQRk5khsSXrs, version: 1}' \ + --environment-id env_01595EKxaaTTGwwY3kyXdtbs \ + --title "CLI docs test session" +``` + +Repeatable flags build arrays. Each `--tool` or `--event` appends one element: + +```bash +ant beta:agents create \ + --name "Research Agent" \ + --model '{id: claude-opus-4-8}' \ + --tool '{type: agent_toolset_20260401}' \ + --tool '{type: custom, name: search_docs, input_schema: {type: object, properties: {query: {type: string}}}}' +``` + +### Stdin + +Pipe a JSON or YAML document to stdin to supply the full request body. Fields from stdin are merged with flags, with flags taking precedence. Here `version` is the optimistic-locking token returned by an earlier `retrieve`, and `$AGENT_ID` was captured as in [Extract a scalar](#extract-a-scalar): + +```bash +echo '{"description": "Updated test agent.", "version": 1}' | \ + ant beta:agents update --agent-id "$AGENT_ID" +``` + +Heredocs work the same way and are convenient for multiline YAML. Quote the delimiter (as in `<<'YAML'`) to disable variable expansion inside the body. + +```bash +ant beta:agents create <<'YAML' +name: Research Agent +model: claude-opus-4-8 +system: | + You are a research assistant. Cite sources for every claim. +tools: + - type: agent_toolset_20260401 +YAML +``` + +### File references + +Flags that take a file path, such as `--file` on the upload command, accept a bare path: + +```bash +ant beta:files upload --file ./report.pdf +``` + +To inline a file's contents into a string-valued field, prefix the path with `@`: + +```bash +ant beta:agents create \ + --name "Researcher" --model '{id: claude-opus-4-8}' \ + --system @./prompts/researcher.txt +``` + +Inside structured flag values, wrap the path in quotes. To send a PDF to the Messages API: + +```bash +ant messages create \ + --model claude-opus-4-8 \ + --max-tokens 1024 \ + --message '{role: user, content: [ + {type: document, source: {type: base64, media_type: application/pdf, data: "@./scan.pdf"}}, + {type: text, text: "Extract the text from this scanned document."} + ]}' \ + --transform 'content.0.text' --raw-output +``` + +The CLI detects the file type and encodes binary files as base64 automatically. To force a specific encoding use `@file://` for plain text or `@data://` for base64. Escape a literal leading `@` with a backslash (`\@username`). + +## Debugging + +Add `--debug` to any command to print the exact HTTP request and response (headers and body) to stderr. API keys are redacted. + +```bash +ant --debug beta:agents list +``` + +```text Output wrap +GET /v1/agents?beta=true HTTP/1.1 +Host: api.anthropic.com +Anthropic-Beta: managed-agents-2026-04-01 +Anthropic-Version: 2023-06-01 +X-Api-Key: +... +``` + +## Available resources + +Every API resource the CLI exposes is documented in the [API reference](/docs/en/api/cli/messages/create). For a local listing, run `ant --help`, and append `--help` to any subcommand for its flags and parameters. + +## Next steps + + + + Version-control API resources, scripting patterns, and use from Claude Code + + + + Endpoint-specific parameters, request fields, and response schemas + + + + API keys, headless hosts, multiple workspaces, and named profiles + + diff --git a/content/en/cli-sdks-libraries/libraries/apple-foundation-models.md b/content/en/cli-sdks-libraries/libraries/apple-foundation-models.md new file mode 100644 index 000000000..5be22de96 --- /dev/null +++ b/content/en/cli-sdks-libraries/libraries/apple-foundation-models.md @@ -0,0 +1,233 @@ +# Apple Foundation Models + +Use Claude on Apple platforms through the Foundation Models framework with the Claude for Foundation Models Swift package. + +--- + +[Claude for Foundation Models](https://github.com/anthropics/ClaudeForFoundationModels) is a Swift package that makes Claude available as a server-side language model in Apple's [Foundation Models](https://developer.apple.com/documentation/foundationmodels) framework. The package conforms Claude to the framework's `LanguageModel` protocol, so you drive it with the same `LanguageModelSession` API you use for Apple's on-device model: `respond(to:)`, streaming, guided generation, and tool calling all work the same way. + +Requests go directly from your app to the Claude API; Apple is not in the request path and does not see prompts or responses. Usage is billed to your Anthropic account at [standard API pricing](/docs/en/about-claude/pricing). Your app decides when to use Claude and when to use Apple's on-device model: pass whichever model you want to each session. + + + **Beta.** This package targets the Foundation Models server-side language model API introduced in the OS 27 betas. APIs might change before general availability. + + + + Claude for Foundation Models is **not** a general-purpose Messages API client. Its public surface is the Foundation Models provider conformance plus the configuration types that reach it (`ClaudeLanguageModel`, `ClaudeModel`, `AuthMode`, `ClaudeServerTool`). For direct access to the Messages API in another language, see the [Client SDKs](/docs/en/cli-sdks-libraries/overview#client-sdks). + + +## Requirements + +* iOS 27, macOS 27, visionOS 27, or watchOS 27 (all in beta): the OS releases whose Foundation Models framework supports server-side language models +* Xcode 27 (beta) +* A Claude API key from the [Claude Console](https://platform.claude.com/) for development. See [Authentication](#authentication) for production options. + +## Install the package + +Add the package to your `Package.swift`: + +```swift +dependencies: [ + .package(url: "https://github.com/anthropics/ClaudeForFoundationModels.git", from: "0.1.0") +] +``` + +Or in Xcode: **File** > **Add Package Dependencies…** and enter the repository URL. + +Then add `ClaudeForFoundationModels` to your target's dependencies and import it alongside `FoundationModels`: + +```swift +import FoundationModels +import ClaudeForFoundationModels +``` + +## Quick start + +`ClaudeLanguageModel` is the entry point. Pass it to `LanguageModelSession` and use the session exactly as you would with any Foundation Models provider: + +```swift +import FoundationModels +import ClaudeForFoundationModels + +let model = ClaudeLanguageModel( + name: .sonnet4_6, + auth: .apiKey(ProcessInfo.processInfo.environment["ANTHROPIC_API_KEY"] ?? "") +) + +let session = LanguageModelSession(model: model) +let response = try await session.respond(to: "Plan a 4-day trip to Buenos Aires.") +print(response.content) +``` + +The initializer also accepts `baseURL` (default `https://api.anthropic.com`), `timeout`, and `serverTools` (see [Server-side tools](#server-side-tools)). + +For a complete working program, the repository includes [`Examples/ClaudeExample`](https://github.com/anthropics/ClaudeForFoundationModels/tree/main/Examples/ClaudeExample), a runnable command-line target that streams a chat turn to the terminal, with a `--search` flag that enables server-side web search for the turn. Running it requires a macOS 27 host. + +## Choosing a model + +Model identifiers are values of `ClaudeModel`. Use a compiled-in constant, or construct one with explicit capabilities for an ID that isn't compiled in yet (see [Capabilities](#capabilities)): + +```swift +ClaudeLanguageModel(name: .opus4_8, auth: auth) +``` + +Constants mirror API model IDs (`.opus4_8` is `claude-opus-4-8`) and carry each model's capabilities. New models ship as new constants in package releases; check `ClaudeModel` in Xcode for the current list, and the [Models overview](/docs/en/about-claude/models/overview) to compare models. + +### Capabilities + +Each `ClaudeModel` declares what it accepts: sampling parameters, effort levels, adaptive thinking, structured output, and image input. The package uses this to determine which request fields to send, because sending a field a model rejects is a hard error. The constants carry the right capabilities. For an ID that isn't compiled in, declare what the model accepts (there is deliberately no shorthand that guesses): + +```swift +let model = ClaudeModel( + id: "claude-experimental-x", + capabilities: .init(samplingParams: false, effortLevels: [.low, .high]) +) +ClaudeLanguageModel(name: model, auth: auth) +``` + +### Effort + +Pin a Claude [effort level](/docs/en/build-with-claude/effort) for every request with `fixedEffort:`. It takes precedence over the framework's per-request reasoning hints, and it's the only way to request `.xhigh` or `.max`, because the framework's reasoning levels stop at high. The API defaults to `high` when no effort is sent: + +```swift +ClaudeLanguageModel(name: .opus4_8, auth: auth, fixedEffort: .xhigh) +``` + +The level must be one the model accepts. Each `ClaudeModel` declares which of the five levels (`low`, `medium`, `high`, `xhigh`, `max`) its model takes, if any: some models don't accept effort at all. + +### When to use Claude versus the on-device model + +Apple's on-device model is fast, private, and available offline, but it is sized for lightweight tasks. Escalate to Claude when you need larger context, frontier reasoning, or server-side tools such as web search and code execution. Because both use the same `LanguageModelSession` API, you can switch by swapping the `model:` argument. + +## Authentication + +Set the credential with the `auth:` parameter. + +### API key (development) + +Pass an API key directly while developing: + +```swift +ClaudeLanguageModel(name: .sonnet4_6, auth: .apiKey("YOUR_API_KEY")) +``` + + + A key bundled into an app is extractable from the shipping binary, and anyone who extracts it can make requests billed to your account. Use `.apiKey` for development only, and switch to a proxy before release. + + +### Proxy (production) + +For production, route requests through your own back end with `.proxied`. The relay at `baseURL` adds the Claude API credential server-side, so the app ships no key. The `headers` you provide are sent on every request so your proxy can authorize the caller. Pass `[:]` if it needs none: + +```swift +ClaudeLanguageModel( + name: .sonnet4_6, + auth: .proxied(headers: ["X-App-Token": "..."]), + baseURL: URL(string: "https://api.yourapp.com/claude")! +) +``` + +Your proxy receives standard [Messages API](/docs/en/api/messages/create) requests, attaches the `x-api-key` header, and forwards them to `https://api.anthropic.com`. + +## Streaming + +`streamResponse(to:)` returns the response incrementally. Each element is a cumulative snapshot of the response so far, not a delta: + +```swift +let stream = session.streamResponse(to: "Summarize today's top science stories.") +for try await partial in stream { + print(partial.content) +} +``` + +## Structured output + +Annotate a type with `@Generable` and request it with `generating:`. The model returns a value of that type through [structured outputs](/docs/en/build-with-claude/structured-outputs): + +```swift +@Generable +struct Trip { + @Guide(description: "Destination city") var destination: String + @Guide(description: "Length in days") var days: Int +} + +let response = try await session.respond(to: "Plan a trip to Tokyo.", generating: Trip.self) +print(response.content.destination) +``` + +Structured output requires a model whose capabilities include it (all compiled-in constants do). If the chosen model does not, the package throws `LanguageModelError.unsupportedGenerationGuide` rather than silently degrading. + +## Tool use + +### Client-side tools + +The framework's `tools:` array works unchanged. Conform your types to `Tool`, pass them to `LanguageModelSession`, and the framework invokes them on the device when Claude calls them. See [Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview). + +```swift +let session = LanguageModelSession(model: model, tools: [FindRestaurantsTool()]) +``` + +### Server-side tools + +[Server tools](/docs/en/agents-and-tools/tool-use/server-tools) (web search, web fetch, and code execution) run on Anthropic's infrastructure within a single round trip, with nothing for the framework to invoke on the device. Configure them for each model with `serverTools:`: + +```swift +let model = ClaudeLanguageModel( + name: .sonnet4_6, + auth: auth, + serverTools: [ + .webSearch(maxUses: 5), + .codeExecution, + ] +) +``` + +`.webSearch` and `.webFetch` accept optional `allowedDomains`, `blockedDomains`, and `maxUses`. Server tool activity surfaces in the transcript as `ClaudeServerToolSegment` custom segments. + + + `serverTools` is configured on `ClaudeLanguageModel` rather than on `LanguageModelSession` because the session type is Apple's. To use different server-tool sets for each conversation, construct multiple `ClaudeLanguageModel` instances. + + +## Images + +Models whose capabilities include image input declare the framework's vision capability. Pass image content through the framework's standard session API; the package converts it to the Claude API's image format. See [Vision](/docs/en/build-with-claude/vision) for image requirements. + +## Error handling + +The package maps Claude API errors onto Apple's `LanguageModelError` cases where one fits: context-window overflow surfaces as `.contextSizeExceeded`, HTTP 429 as `.rateLimited`, a request past the configured timeout as `.timeout`. Provider errors with no framework equivalent surface as `ClaudeError`. Pattern-match to drive product flows: + +```swift +do { + let response = try await session.respond(to: prompt) + print(response.content) +} catch ClaudeError.missingCredential { + // Prompt for an API key. +} catch let error as LanguageModelError { + // Framework-shaped errors (rate limits, guardrails, context length, decoding). +} catch { + // Transport errors. +} +``` + +A common pattern is to catch `.rateLimited` and fall back to `SystemLanguageModel` for that turn, queue the request, or surface a retry affordance. + +## Feature support + +The package surfaces the Messages API capabilities that the Foundation Models provider protocol can express. Features with no representation in Apple's protocol are not available through it, including: + +* Prompt caching controls (the package applies prompt caching automatically; cache TTL and breakpoint placement are not configurable) +* Stop sequences +* Batch processing +* Files API +* Token counting +* Beta headers + +## Additional resources + +| Reference | Covers | +| --------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- | +| [Apple Foundation Models documentation](https://developer.apple.com/documentation/foundationmodels) | `LanguageModelSession`, `@Generable`, `Transcript`, `Tool`, and the rest of the framework surface | +| [`ClaudeForFoundationModels` on GitHub](https://github.com/anthropics/ClaudeForFoundationModels) | Source, the runnable example, and the issue tracker | +| [Claude API reference](/docs/en/api/overview) | The underlying Messages API | + +The package is licensed under Apache 2.0. Bug reports are welcome through GitHub issues. External pull requests are not being accepted during the beta period. diff --git a/content/en/cli-sdks-libraries/libraries/openai-sdk.md b/content/en/cli-sdks-libraries/libraries/openai-sdk.md new file mode 100644 index 000000000..1cc511c29 --- /dev/null +++ b/content/en/cli-sdks-libraries/libraries/openai-sdk.md @@ -0,0 +1,318 @@ +# OpenAI SDK compatibility + +Anthropic provides a compatibility layer that enables you to use the OpenAI SDK to test the Claude API. With a few code changes, you can quickly evaluate Anthropic model capabilities. + +--- + + + This compatibility layer is primarily intended to test and compare model capabilities, and is not considered a long-term or production-ready solution for most use cases. While it is intended to remain fully functional and not have breaking changes, the priority is the reliability and effectiveness of the [Claude API](/docs/en/api/overview). + + For more information on known compatibility limitations, see [Important OpenAI compatibility limitations](#important-openai-compatibility-limitations). + + If you encounter any issues with the OpenAI SDK compatibility feature, please share your feedback via this [compatibility feedback form](https://forms.gle/oQV4McQNiuuNbz9n8). + + + + For the best experience and access to Claude API full feature set ([PDF processing](/docs/en/build-with-claude/pdf-support), [citations](/docs/en/build-with-claude/citations), [extended thinking](/docs/en/build-with-claude/extended-thinking), and [prompt caching](/docs/en/build-with-claude/prompt-caching)), use the native [Claude API](/docs/en/api/overview). + + +## Getting started with the OpenAI SDK + +To use the OpenAI SDK compatibility feature, you'll need to: + +1. Use an official OpenAI SDK + +2. Change the following + + * Update your base URL to point to the Claude API + * Replace your API key with a [Claude API key](/settings/keys) + * Update your model name to use a [Claude model](/docs/en/about-claude/models/overview) + +3. Review the documentation below for what features are supported + +### Quick start example + + + ```python Python + import os + + from openai import OpenAI + + client = OpenAI( + api_key=os.environ.get("ANTHROPIC_API_KEY"), # Your Claude API key + base_url="https://api.anthropic.com/v1/", # the Claude API endpoint + ) + + response = client.chat.completions.create( + model="claude-opus-4-8", # Claude model name + messages=[ + {"role": "system", "content": "You are a helpful assistant."}, + {"role": "user", "content": "Who are you?"}, + ], + ) + + print(response.choices[0].message.content) + ``` + + ```typescript TypeScript + import OpenAI from "openai"; + + const openai = new OpenAI({ + apiKey: "ANTHROPIC_API_KEY", // Your Claude API key + baseURL: "https://api.anthropic.com/v1/" // Claude API endpoint + }); + + const response = await openai.chat.completions.create({ + messages: [{ role: "user", content: "Who are you?" }], + model: "claude-opus-4-8" // Claude model name + }); + + console.log(response.choices[0].message.content); + ``` + + +## Important OpenAI compatibility limitations + +### API behavior + +Here are the most substantial differences from using OpenAI: + +* The `strict` parameter for function calling is ignored, which means the tool use JSON is not guaranteed to follow the supplied schema. For guaranteed schema conformance, use the native [Claude API with Structured Outputs](/docs/en/build-with-claude/structured-outputs). +* Audio input is not supported; it will simply be ignored and stripped from input +* Prompt caching is not supported, but it is supported in the [Anthropic SDKs](/docs/en/cli-sdks-libraries/overview) +* System/developer messages are hoisted and concatenated to the beginning of the conversation, as Anthropic only supports a single initial system message. + +Most unsupported fields are silently ignored rather than producing errors. These are all documented below. + +### Output quality considerations + +If you’ve done lots of tweaking to your prompt, it’s likely to be well-tuned to OpenAI specifically. Consider using the [prompt improver in the Claude Console](/dashboard) as a good starting point. + +### System / developer message hoisting + +Most of the inputs to the OpenAI SDK clearly map directly to Anthropic’s API parameters, but one distinct difference is the handling of system / developer prompts. These two prompts can be put throughout a chat conversation via OpenAI. Since Anthropic only supports an initial system message, the API takes all system/developer messages and concatenates them together with a single newline (`\n`) in between them. This full string is then supplied as a single system message at the start of the messages. + +### Extended thinking support + +You can enable [extended thinking](/docs/en/build-with-claude/extended-thinking) capabilities by adding the `thinking` parameter. While this improves Claude's reasoning for complex tasks, the OpenAI SDK doesn't return Claude's detailed thought process. For full extended thinking features, including access to Claude's step-by-step reasoning output, use the native Claude API. + + + ```python Python + response = client.chat.completions.create( + model="claude-sonnet-4-6", + messages=[{"role": "user", "content": "Who are you?"}], + extra_body={"thinking": {"type": "enabled", "budget_tokens": 2000}}, + ) + ``` + + ```typescript TypeScript + const response = await openai.chat.completions.create({ + messages: [{ role: "user", content: "Who are you?" }], + model: "claude-sonnet-4-6", + // @ts-expect-error + thinking: { type: "enabled", budget_tokens: 2000 } + }); + ``` + + +## Rate limits + +Rate limits follow Anthropic's [standard limits](/docs/en/api/rate-limits) for the `/v1/messages` endpoint. + +## Detailed OpenAI compatible API support + +### Request fields + +#### Simple fields + +| Field | Support status | +| ----------------------- | ---------------------------------------------------------------------------------------------------------------------------- | +| `model` | Use Claude model names | +| `max_tokens` | Fully supported | +| `max_completion_tokens` | Fully supported | +| `stream` | Fully supported | +| `stream_options` | Fully supported | +| `top_p` | Fully supported | +| `parallel_tool_calls` | Fully supported | +| `stop` | All non-whitespace stop sequences work | +| `temperature` | Between 0 and 1 (inclusive). Values greater than 1 are capped at 1. | +| `n` | Must be exactly 1 | +| `logprobs` | Ignored | +| `metadata` | Ignored | +| `response_format` | Ignored. For JSON output, use [Structured Outputs](/docs/en/build-with-claude/structured-outputs) with the native Claude API | +| `prediction` | Ignored | +| `presence_penalty` | Ignored | +| `frequency_penalty` | Ignored | +| `seed` | Ignored | +| `service_tier` | Ignored | +| `audio` | Ignored | +| `logit_bias` | Ignored | +| `store` | Ignored | +| `user` | Ignored | +| `modalities` | Ignored | +| `top_logprobs` | Ignored | +| `reasoning_effort` | Ignored | + +#### `tools` / `functions` fields + + + + + `tools[n].function` fields + + | Field | Support status | + | ------------- | ------------------------------------------------------------------------------------------------------------------------------------ | + | `name` | Fully supported | + | `description` | Fully supported | + | `parameters` | Fully supported | + | `strict` | Ignored. Use [Structured Outputs](/docs/en/build-with-claude/structured-outputs) with native Claude API for strict schema validation | + + + + `functions[n]` fields + + + OpenAI has deprecated the `functions` field and suggests using `tools` instead. + + + | Field | Support status | + | ------------- | ------------------------------------------------------------------------------------------------------------------------------------ | + | `name` | Fully supported | + | `description` | Fully supported | + | `parameters` | Fully supported | + | `strict` | Ignored. Use [Structured Outputs](/docs/en/build-with-claude/structured-outputs) with native Claude API for strict schema validation | + + + + +#### `messages` array fields + + + + + Fields for `messages[n].role == "developer"` + + + Developer messages are hoisted to beginning of conversation as part of the initial system message + + + | Field | Support status | + | --------- | ---------------------------- | + | `content` | Fully supported, but hoisted | + | `name` | Ignored | + + + + Fields for `messages[n].role == "system"` + + + System messages are hoisted to beginning of conversation as part of the initial system message + + + | Field | Support status | + | --------- | ---------------------------- | + | `content` | Fully supported, but hoisted | + | `name` | Ignored | + + + + Fields for `messages[n].role == "user"` + + | Field | Variant | Sub-field | Support status | + | --------- | -------------------------------- | --------- | --------------- | + | `content` | `string` | | Fully supported | + | | `array`, `type == "text"` | | Fully supported | + | | `array`, `type == "image_url"` | `url` | Fully supported | + | | | `detail` | Ignored | + | | `array`, `type == "input_audio"` | | Ignored | + | | `array`, `type == "file"` | | Ignored | + | `name` | | | Ignored | + + + + Fields for `messages[n].role == "assistant"` + + | Field | Variant | Support status | + | --------------- | ---------------------------- | --------------- | + | `content` | `string` | Fully supported | + | | `array`, `type == "text"` | Fully supported | + | | `array`, `type == "refusal"` | Ignored | + | `tool_calls` | | Fully supported | + | `function_call` | | Fully supported | + | `audio` | | Ignored | + | `refusal` | | Ignored | + + + + Fields for `messages[n].role == "tool"` + + | Field | Variant | Support status | + | -------------- | ------------------------- | --------------- | + | `content` | `string` | Fully supported | + | | `array`, `type == "text"` | Fully supported | + | `tool_call_id` | | Fully supported | + | `tool_choice` | | Fully supported | + | `name` | | Ignored | + + + + Fields for `messages[n].role == "function"` + + | Field | Variant | Support status | + | ------------- | ------------------------- | --------------- | + | `content` | `string` | Fully supported | + | | `array`, `type == "text"` | Fully supported | + | `tool_choice` | | Fully supported | + | `name` | | Ignored | + + + + +### Response fields + +| Field | Support status | +| --------------------------------- | ------------------------------ | +| `id` | Fully supported | +| `choices[]` | Will always have a length of 1 | +| `choices[].finish_reason` | Fully supported | +| `choices[].index` | Fully supported | +| `choices[].message.role` | Fully supported | +| `choices[].message.content` | Fully supported | +| `choices[].message.tool_calls` | Fully supported | +| `object` | Fully supported | +| `created` | Fully supported | +| `model` | Fully supported | +| `finish_reason` | Fully supported | +| `content` | Fully supported | +| `usage.completion_tokens` | Fully supported | +| `usage.prompt_tokens` | Fully supported | +| `usage.total_tokens` | Fully supported | +| `usage.completion_tokens_details` | Always empty | +| `usage.prompt_tokens_details` | Always empty | +| `choices[].message.refusal` | Always empty | +| `choices[].message.audio` | Always empty | +| `logprobs` | Always empty | +| `service_tier` | Always empty | +| `system_fingerprint` | Always empty | + +### Error message compatibility + +The compatibility layer maintains consistent error formats with the OpenAI API. However, the detailed error messages will not be equivalent. Only use the error messages for logging and debugging. + +### Header compatibility + +While the OpenAI SDK automatically manages headers, here is the complete list of headers supported by the Claude API for developers who need to work with them directly. + +| Header | Support Status | +| -------------------------------- | ------------------- | +| `x-ratelimit-limit-requests` | Fully supported | +| `x-ratelimit-limit-tokens` | Fully supported | +| `x-ratelimit-remaining-requests` | Fully supported | +| `x-ratelimit-remaining-tokens` | Fully supported | +| `x-ratelimit-reset-requests` | Fully supported | +| `x-ratelimit-reset-tokens` | Fully supported | +| `retry-after` | Fully supported | +| `request-id` | Fully supported | +| `openai-version` | Always `2020-10-01` | +| `authorization` | Fully supported | +| `openai-processing-ms` | Always empty | diff --git a/content/en/cli-sdks-libraries/middleware.md b/content/en/cli-sdks-libraries/middleware.md new file mode 100644 index 000000000..368f2faa2 --- /dev/null +++ b/content/en/cli-sdks-libraries/middleware.md @@ -0,0 +1,178 @@ +# SDK middleware + +Intercept and modify requests and responses in the Anthropic SDKs. + +--- + +The Anthropic SDKs provide a middleware (or interceptor) hook that lets you run code before a request is sent and after the response is received. Use middleware for cross-cutting concerns such as logging, custom retries, request annotation, and refusal fallback handling. + +```mermaid +sequenceDiagram + autonumber + participant App as Your code + participant M1 as Middleware A + participant M2 as Middleware B + participant Core as SDK core + participant API as Claude API + App->>M1: request + M1->>M2: next(request) + M2->>Core: next(request) + Core->>API: HTTP request + API-->>Core: HTTP response + Core-->>M2: response + M2-->>M1: response + M1-->>App: response +``` + +Each middleware can inspect or replace the request before calling `next()`, and the response after `next()` returns. + +## Registering middleware + +Each middleware is a function that receives the outgoing request and a `next` callable. Call `next` to forward the request to the rest of the chain (or directly to the SDK core if this is the last middleware), and return its response. Anything before the `next` call runs on the way out; anything after runs on the way back. + + + ```python Python + def logging_middleware(request: APIRequest, call_next: CallNext) -> APIResponse[Any]: + # Before the request + print(f"-> {request.method} {request.url}") + + # Forward the request to the rest of the chain + response = call_next(request) + + # After the request + print(f"<- {response.status_code}") + + return response + + + client = Anthropic(middleware=[logging_middleware]) + ``` + + ```typescript TypeScript + import type { Middleware } from "@anthropic-ai/sdk"; + + const loggingMiddleware: Middleware = async (request, next, ctx) => { + // Before the request + ctx.logger.debug("->", request.method, request.url); + + // Forward the request to the rest of the chain + const response = await next(request); + + // After the request + ctx.logger.debug("<-", response.status, request.url); + + return response; + }; + + const client = new Anthropic({ middleware: [loggingMiddleware] }); + ``` + + ```csharp C# + AnthropicClient client = new() + { + Handlers = + [ + Handler.Create(async (request, next, cancellationToken) => + { + // Before the request + Console.WriteLine($"Sending {request.Method} {request.RequestUri}"); + + // Forward the request to the next handler + var response = await next(request, cancellationToken); + + // After the request + Console.WriteLine($"Received {(int)response.StatusCode}"); + + return response; + }), + ], + }; + ``` + + ```go Go + client := anthropic.NewClient( + option.WithMiddleware(func(req *http.Request, next option.MiddlewareNext) (*http.Response, error) { + // Before the request + start := time.Now() + slog.Info("sending request", "method", req.Method, "url", req.URL) + + // Forward the request to the rest of the chain + res, err := next(req) + if err != nil { + return nil, err + } + + // After the request + slog.Info("received response", "status", res.StatusCode, "duration", time.Since(start)) + + return res, nil + }), + ) + ``` + + ```java Java + AnthropicClient client = AnthropicOkHttpClient.builder() + .fromEnv() + .addInterceptor(Interceptor.syncOnly((nextClient, request, requestOptions) -> { + // Before the request + IO.println(request.method() + " /" + String.join("/", request.pathSegments())); + + // Forward the request to the next handler + HttpResponse response = nextClient.execute(request, requestOptions); + + // After the request + IO.println(response.statusCode()); + + return response; + })) + .build(); + ``` + + ```php PHP + $loggingMiddleware = function (RequestInterface $request, callable $next): ResponseInterface { + // Before the request + error_log("-> {$request->getMethod()} {$request->getUri()}"); + + // Forward the request to the rest of the chain + $response = $next($request); + + // After the request + error_log("<- {$response->getStatusCode()}"); + + return $response; + }; + + $client = new Client(requestOptions: ['middleware' => [$loggingMiddleware]]); + ``` + + ```ruby Ruby + logging_middleware = lambda do |request, call_next| + # Before the request + puts "-> #{request.method.upcase} #{request.url}" + + # Forward the request to the rest of the chain + response = call_next.call(request) + + # After the request + puts "<- #{response.status}" + + response + end + + client = Anthropic::Client.new(middleware: [logging_middleware]) + ``` + + +## Middleware ordering + +When you register multiple middleware, they apply in the order given: the first middleware's "before" code runs first, and its "after" code runs last. Middleware registered on the client runs before middleware passed as a per-request option. + +In the Go SDK, repeated `option.WithMiddleware` calls concatenate (client first, then method). In the other SDKs, pass an array; later entries wrap inner. + +## Replacing the HTTP client + +Each SDK also accepts a custom HTTP client (for proxy configuration, custom TLS, or connection pooling). Only one HTTP client is used per SDK client; setting it replaces the default. The custom HTTP client receives requests after all middleware has run. + +## Built-in middleware + +The SDKs ship a refusal-fallback middleware that automatically retries requests Claude Fable 5 declines on a fallback model. See [Detect and retry on a fallback model](/docs/en/build-with-claude/refusals-and-fallback#client-side-fallback) for setup and per-language examples. diff --git a/content/en/cli-sdks-libraries/overview.md b/content/en/cli-sdks-libraries/overview.md new file mode 100644 index 000000000..f29391f44 --- /dev/null +++ b/content/en/cli-sdks-libraries/overview.md @@ -0,0 +1,87 @@ +# CLI, SDKs, and libraries + +Official tools for building with the Claude API: the ant CLI, client SDKs in seven languages, and framework-specific libraries. + +--- + +Anthropic provides three kinds of official tooling for building with the Claude API: + +* **CLI:** The `ant` command-line tool for shell scripting and interactive use. +* **Client SDKs:** General-purpose Messages API clients for Python, TypeScript, C#, Go, Java, PHP, and Ruby. Each SDK provides idiomatic interfaces, type safety, and built-in support for streaming, retries, and error handling. +* **Libraries and integrations:** Packages and compatibility layers that expose Claude inside another framework's API surface rather than the Messages API directly. + + + For the full API specification, see the [API reference](/docs/en/api/overview). + + +## CLI + + + + Shell scripting, typed flags, response transforms + + + +## Client SDKs + + + + Sync and async clients, Pydantic models + + + + Node.js, Deno, Bun, and browser support + + + + .NET Standard 2.0+, IChatClient integration + + + + Context-based cancellation, functional options + + + + Builder pattern, CompletableFuture async + + + + Value objects, builder pattern + + + + Sorbet types, streaming helpers + + + +## Libraries and integrations + +Libraries and integrations expose Claude through another framework's API surface. They are not general-purpose Messages API clients. + + + + Swift package for Apple's `LanguageModelSession` API + + + + Use Claude through the OpenAI SDK surface + + + +## Building agents or using Claude Code? + +The CLI, client SDKs, and libraries are for calling the Claude API yourself: you send each request and handle each response. Claude Code, the Claude Agent SDK, and Claude Managed Agents work at a higher level, providing the agent loop, tool execution, and runtime. + + + + Agentic coding tool for delegating coding tasks to Claude + + + + Build agents that run in a process you operate + + + + Run agents in Anthropic's managed infrastructure + + diff --git a/content/en/cli-sdks-libraries/sdks/csharp.md b/content/en/cli-sdks-libraries/sdks/csharp.md new file mode 100644 index 000000000..21c448453 --- /dev/null +++ b/content/en/cli-sdks-libraries/sdks/csharp.md @@ -0,0 +1,467 @@ +# C# SDK + +Install and configure the Anthropic C# SDK for .NET applications with IChatClient integration + +--- + +The Anthropic C# SDK provides convenient access to the Anthropic REST API from applications written in C#. + + + The C# SDK is currently in beta. APIs may change between versions. + + + + For API feature documentation with code examples, see the [API reference](/docs/en/api/overview). This page covers C#-specific SDK features and configuration. + + + + As of version 10+, the `Anthropic` package is now the official Anthropic SDK for C#. Package versions 3.X and below were previously used for the tryAGI community-built SDK, which has moved to [`tryAGI.Anthropic`](https://www.nuget.org/packages/tryagi.Anthropic/). If you need to continue using the former client in your project, update your package reference to `tryAGI.Anthropic`. + + +## Installation + +Install the package from [NuGet](https://www.nuget.org/packages/Anthropic): + +```bash +dotnet add package Anthropic +``` + +## Requirements + +This library requires .NET Standard 2.0 or later. + +## Usage + +```csharp +using System; +using Anthropic; +using Anthropic.Models.Messages; + +AnthropicClient client = new(); + +MessageCreateParams parameters = new() +{ + MaxTokens = 1024, + Messages = + [ + new() + { + Role = Role.User, + Content = "Hello, Claude", + }, + ], + Model = Model.ClaudeOpus4_8, +}; + +var message = await client.Messages.Create(parameters); + +foreach (var block in message.Content) +{ + if (block.TryPickText(out var textBlock)) + { + Console.WriteLine(textBlock.Text); + } +} +``` + +For authentication options including Workload Identity Federation, see [Authentication](/docs/en/manage-claude/authentication). + +## Client configuration + +Configure the client using environment variables: + +```csharp +using Anthropic; + +// Configured using the ANTHROPIC_API_KEY, ANTHROPIC_AUTH_TOKEN and ANTHROPIC_BASE_URL environment variables +AnthropicClient client = new(); +``` + +Or manually: + +```csharp +using Anthropic; + +AnthropicClient client = new() { ApiKey = "my-anthropic-api-key" }; +``` + +Or using a combination of the two approaches. + +See this table for the available options: + +| Property | Environment variable | Required | Default value | +| ----------- | ---------------------- | -------- | ----------------------------- | +| `ApiKey` | `ANTHROPIC_API_KEY` | false | - | +| `AuthToken` | `ANTHROPIC_AUTH_TOKEN` | false | - | +| `BaseUrl` | `ANTHROPIC_BASE_URL` | true | `"https://api.anthropic.com"` | + +### Modifying configuration + +To temporarily use a modified client configuration, while reusing the same connection and thread pools, call `WithOptions` on any client or service: + +```csharp +using System; + +var message = await client + .WithOptions(options => + options with + { + BaseUrl = "https://example.com", + Timeout = TimeSpan.FromSeconds(42), + } + ) + .Messages.Create(parameters); + +Console.WriteLine(message); +``` + +Using a [`with` expression](https://learn.microsoft.com/en-us/dotnet/csharp/language-reference/operators/with-expression) makes it easy to construct the modified options. + +The `WithOptions` method does not affect the original client or service. + +## Streaming + +The SDK defines methods that return response "chunk" streams, where each chunk can be individually processed as soon as it arrives instead of waiting on the full response. Streaming methods generally correspond to [SSE](https://developer.mozilla.org/en-US/docs/Web/API/Server-sent_events) or [JSONL](https://jsonlines.org) responses. + +A streaming method always has a `Streaming` suffix in its name, even if it doesn't have a non-streaming variant. + +These streaming methods return [`IAsyncEnumerable`](https://learn.microsoft.com/en-us/dotnet/api/system.collections.generic.iasyncenumerable-1): + +```csharp +using System; +using Anthropic.Models.Messages; + +MessageCreateParams parameters = new() +{ + MaxTokens = 1024, + Messages = + [ + new() + { + Role = Role.User, + Content = "Hello, Claude", + }, + ], + Model = Model.ClaudeOpus4_8, +}; + +await foreach (var message in client.Messages.CreateStreaming(parameters)) +{ + Console.WriteLine(message); +} +``` + +## Error handling + +The SDK throws custom unchecked exception types: + +* `AnthropicApiException`: Base class for API errors. See this table for which exception subclass is thrown for each HTTP status code: + +| Status | Exception | +| ------ | ---------------------------------------- | +| 400 | `AnthropicBadRequestException` | +| 401 | `AnthropicUnauthorizedException` | +| 403 | `AnthropicForbiddenException` | +| 404 | `AnthropicNotFoundException` | +| 422 | `AnthropicUnprocessableEntityException` | +| 429 | `AnthropicRateLimitException` | +| 5xx | `Anthropic5xxException` | +| others | `AnthropicUnexpectedStatusCodeException` | + +Additionally, all 4xx errors inherit from `Anthropic4xxException`. + +* `AnthropicSseException`: thrown for errors encountered during SSE streaming after a successful initial HTTP response. + +* `AnthropicIOException`: I/O networking errors. + +* `AnthropicInvalidDataException`: Failure to interpret successfully parsed data. For example, when accessing a property that's supposed to be required, but the API unexpectedly omitted it from the response. + +* `AnthropicException`: Base class for all exceptions. + +## Retries + +The SDK automatically retries 2 times by default, with a short exponential backoff between requests. + +Only the following error types are retried: + +* Connection errors (for example, because of a network connectivity problem) +* 408 Request Timeout +* 409 Conflict +* 429 Rate Limit +* 5xx Internal + +The API may also explicitly instruct the SDK to retry or not retry a request. + +To set a custom number of retries, configure the client using the `MaxRetries` property: + +```csharp +using Anthropic; + +AnthropicClient client = new() { MaxRetries = 3 }; +``` + +Or configure a single method call using `WithOptions`: + +```csharp +using System; + +var message = await client + .WithOptions(options => + options with { MaxRetries = 3 } + ) + .Messages.Create(parameters); + +Console.WriteLine(message); +``` + +## Timeouts + +Requests time out after 10 minutes by default. + +To set a custom timeout, configure the client using the `Timeout` option: + +```csharp +using System; +using Anthropic; + +AnthropicClient client = new() { Timeout = TimeSpan.FromSeconds(42) }; +``` + +Or configure a single method call using `WithOptions`: + +```csharp +using System; + +var message = await client + .WithOptions(options => + options with { Timeout = TimeSpan.FromSeconds(42) } + ) + .Messages.Create(parameters); + +Console.WriteLine(message); +``` + +## Pagination + +The SDK defines methods that return paginated lists of results. It provides convenient ways to access the results either one page at a time or item-by-item across all pages. + +### Auto-pagination + +To iterate through all results across all pages, use the `Paginate` method, which automatically fetches more pages as needed. The method returns an [`IAsyncEnumerable`](https://learn.microsoft.com/en-us/dotnet/api/system.collections.generic.iasyncenumerable-1): + +```csharp +using System; + +var page = await client.Messages.Batches.List(parameters); +await foreach (var item in page.Paginate()) +{ + Console.WriteLine(item); +} +``` + +### Manual pagination + +To access individual page items and manually request the next page, use the `Items` property, and `HasNext` and `Next` methods: + +```csharp +var page = await client.Messages.Batches.List(); +while (true) +{ + foreach (var item in page.Items) + { + Console.WriteLine(item); + } + if (!page.HasNext()) + { + break; + } + page = await page.Next(); +} +``` + +## Response validation + +In rare cases, the API may return a response that doesn't match the expected type. By default, the SDK does not throw an exception in this case. It throws `AnthropicInvalidDataException` only if you directly access the property. + +If you would prefer to check that the response is completely well-typed upfront, then either call `Validate`: + +```csharp +var message = await client.Messages.Create(parameters); +message.Validate(); +``` + +Or configure the client using the `ResponseValidation` option: + +```csharp +using Anthropic; + +AnthropicClient client = new() { ResponseValidation = true }; +``` + +Or configure a single method call using `WithOptions`: + +```csharp +using System; + +var message = await client + .WithOptions(options => + options with { ResponseValidation = true } + ) + .Messages.Create(parameters); + +Console.WriteLine(message); +``` + +## IChatClient integration + +The SDK provides an implementation of the `IChatClient` interface from the `Microsoft.Extensions.AI.Abstractions` library. This enables `AnthropicClient` (and `Anthropic.Services.IBetaService`) to be used with other libraries that integrate with these core abstractions. For example, tools in the MCP C# SDK (`ModelContextProtocol`) library can be used directly with an `AnthropicClient` exposed through `IChatClient`. + +```csharp +using Anthropic; +using Microsoft.Extensions.AI; +using ModelContextProtocol.Client; + +// Configured using the ANTHROPIC_API_KEY, ANTHROPIC_AUTH_TOKEN and ANTHROPIC_BASE_URL environment variables +AnthropicClient client = new(); + +IChatClient chatClient = client.AsIChatClient("claude-opus-4-8") + .AsBuilder() + .UseFunctionInvocation() + .Build(); + +// Using McpClient from the MCP C# SDK +McpClient learningServer = await McpClient.CreateAsync( + new HttpClientTransport(new() { Endpoint = new("https://learn.microsoft.com/api/mcp") })); + +ChatOptions options = new() { Tools = [.. await learningServer.ListToolsAsync()] }; + +Console.WriteLine(await chatClient.GetResponseAsync("Tell me about IChatClient", options)); +``` + +## Requests and responses + +To send a request to the Claude API, build an instance of a `Params` class and pass it to the corresponding client method. When the response is received, it's deserialized into an instance of a C# class. + +For example, `client.Messages.Create` should be called with an instance of `MessageCreateParams`, and it will return an instance of `Task`. + +## Advanced usage + +### Binary responses + +The SDK defines methods that return binary responses, which are used for API responses that shouldn't necessarily be parsed, like non-JSON data. + +These methods return `HttpResponse`: + +```csharp +using System; +using Anthropic.Models.Beta.Files; + +FileDownloadParams parameters = new() { FileID = "file_id" }; + +var response = await client.Beta.Files.Download(parameters); + +Console.WriteLine(response); +``` + +To save the response content to a file, or any [`Stream`](https://learn.microsoft.com/en-us/dotnet/api/system.io.stream), use the [`CopyToAsync`](https://learn.microsoft.com/en-us/dotnet/api/system.io.stream.copytoasync) method: + +```csharp +using System.IO; + +using var response = await client.Beta.Files.Download(parameters); +using var contentStream = await response.ReadAsStream(); +using var fileStream = File.Open(path, FileMode.OpenOrCreate); +await contentStream.CopyToAsync(fileStream); // Or any other Stream +``` + +### Raw responses + +The SDK defines methods that deserialize responses into instances of C# classes. To access response headers, status code, or the raw response body, prefix any HTTP method call on a client or service with `WithRawResponse`: + +```csharp +var response = await client.WithRawResponse.Messages.Create(parameters); +var statusCode = response.StatusCode; +var headers = response.Headers; +``` + +The raw `HttpResponseMessage` can also be accessed through the `RawMessage` property. + +For non-streaming responses, you can deserialize the response into an instance of a C# class if needed: + +```csharp +using System; +using Anthropic.Models.Messages; + +var response = await client.WithRawResponse.Messages.Create(parameters); +Message deserialized = await response.Deserialize(); +Console.WriteLine(deserialized); +``` + +For streaming responses, you can deserialize the response to an `IAsyncEnumerable` if needed: + +```csharp +using System; + +var response = await client.WithRawResponse.Messages.CreateStreaming(parameters); +await foreach (var item in response.Enumerate()) +{ + Console.WriteLine(item); +} +``` + +### Logging + + + All log messages are intended for debugging only. The format and content of log messages may change between releases. + + +Enable debug logging by setting an environment variable: + +```bash +export ANTHROPIC_LOG=debug +``` + +### Undocumented API functionality + +The SDK is typed for convenient usage of the documented API. However, it also supports working with undocumented or not yet supported parts of the API. + +## Platform integrations + + + For detailed platform setup guides with code examples, see: + + * [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) + * [Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy) + * [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) + * [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) + * [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) + + +The C# SDK supports the following platforms through separate NuGet packages: + +* **Agent Platform:** `Anthropic.Vertex`. See [Claude on Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) for client setup. +* **Bedrock:** `Anthropic.Bedrock`. Use `AnthropicBedrockMantleClient` for the Messages-API Bedrock endpoint, or `AnthropicBedrockClient` (`bedrock-runtime` path). `AnthropicBedrockMantleClient` takes an optional `MantleAwsClientOptions` config object; `AnthropicBedrockClient` accepts `AnthropicBedrockCredentialsHelper.FromEnv()` or explicit credentials. +* **Claude Platform on AWS:** `Anthropic.Aws`. Use `AnthropicAwsClient`; set `WorkspaceId` on the client or the `ANTHROPIC_AWS_WORKSPACE_ID` environment variable (see [Workspaces](/docs/en/build-with-claude/claude-platform-on-aws#workspaces)). Available in beta. +* **Foundry:** `Anthropic.Foundry`. Use `AnthropicFoundryClient` with `DefaultAnthropicFoundryCredentials.FromEnv()` or explicit credentials. + +Use `AnthropicBedrockMantleClient` for new projects; `AnthropicBedrockClient` remains for existing applications using the Bedrock `InvokeModel` API. + +## Semantic versioning + + + Although this package is versioned as 10+, it's currently in beta. During the beta period, breaking changes may occur in minor or patch releases. Once the library reaches stable release, SemVer conventions will be followed more strictly. Share feedback by [filing an issue](https://github.com/anthropics/anthropic-sdk-csharp/issues/new). + + +This package generally follows [SemVer](https://semver.org/spec/v2.0.0.html) conventions, though certain backwards-incompatible changes may be released as minor versions: + +1. Changes to library internals that are technically public but not intended or documented for external use. +2. Changes that aren't expected to impact the vast majority of users in practice. + +Backwards-compatibility is taken seriously to ensure you can rely on a smooth upgrade experience. + +## Additional resources + +* [GitHub repository](https://github.com/anthropics/anthropic-sdk-csharp) +* [NuGet package](https://www.nuget.org/packages/Anthropic) +* [API reference](/docs/en/api/overview) +* [Streaming Messages](/docs/en/build-with-claude/streaming) diff --git a/content/en/cli-sdks-libraries/sdks/go.md b/content/en/cli-sdks-libraries/sdks/go.md new file mode 100644 index 000000000..c89835f27 --- /dev/null +++ b/content/en/cli-sdks-libraries/sdks/go.md @@ -0,0 +1,752 @@ +# Go SDK + +Install and configure the Anthropic Go SDK with context-based cancellation and functional options + +--- + +The Anthropic Go library provides convenient access to the Anthropic REST API from applications written in Go. + + + For API feature documentation with code examples, see the [API reference](/docs/en/api/overview). This page covers Go-specific SDK features and configuration. + + +## Installation + +```go +import ( + "github.com/anthropics/anthropic-sdk-go" // imported as anthropic +) +``` + +Install with `go get`: + +```bash +go get github.com/anthropics/anthropic-sdk-go +``` + +## Requirements + +This library requires Go 1.23+. + +## Usage + +```go +package main + +import ( + "context" + "fmt" + + "github.com/anthropics/anthropic-sdk-go" + "github.com/anthropics/anthropic-sdk-go/option" +) + +func main() { + client := anthropic.NewClient( + option.WithAPIKey("my-anthropic-api-key"), // defaults to os.LookupEnv("ANTHROPIC_API_KEY") + ) + message, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + MaxTokens: 1024, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("What is a quaternion?")), + }, + Model: anthropic.ModelClaudeOpus4_8, + }) + if err != nil { + panic(err.Error()) + } + for _, block := range message.Content { + if textBlock, ok := block.AsAny().(anthropic.TextBlock); ok { + fmt.Println(textBlock.Text) + } + } +} +``` + +For authentication options including Workload Identity Federation, see [Authentication](/docs/en/manage-claude/authentication). + + + + ```go + messages := []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock("What is my first name?")), + } + + message, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + Messages: messages, + MaxTokens: 1024, + }) + if err != nil { + panic(err) + } + + fmt.Printf("%+v\n", message.Content) + + messages = append(messages, message.ToParam()) + messages = append(messages, anthropic.NewUserMessage( + anthropic.NewTextBlock("My full name is John Doe"), + )) + + message, err = client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + Messages: messages, + MaxTokens: 1024, + }) + if err != nil { + panic(err) + } + + fmt.Printf("%+v\n", message.Content) + ``` + + + + ```go + message, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 1024, + System: []anthropic.TextBlockParam{ + {Text: "Be very serious at all times."}, + }, + Messages: messages, + }) + if err != nil { + panic(err) + } + fmt.Printf("%+v\n", message.Content) + ``` + + + + ```go + content := "What is a quaternion?" + + stream := client.Messages.NewStreaming(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 1024, + Messages: []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock(content)), + }, + }) + + message := anthropic.Message{} + for stream.Next() { + event := stream.Current() + err := message.Accumulate(event) + if err != nil { + panic(err) + } + + switch eventVariant := event.AsAny().(type) { + case anthropic.ContentBlockDeltaEvent: + switch deltaVariant := eventVariant.Delta.AsAny().(type) { + case anthropic.TextDelta: + print(deltaVariant.Text) + } + + } + } + + if stream.Err() != nil { + panic(stream.Err()) + } + ``` + + + + ```go + messages := []anthropic.MessageParam{ + anthropic.NewUserMessage(anthropic.NewTextBlock(content)), + } + + toolParams := []anthropic.ToolParam{ + { + Name: "get_coordinates", + Description: anthropic.String("Accepts a place as an address, then returns the latitude and longitude coordinates."), + InputSchema: GetCoordinatesInputSchema, + }, + } + tools := make([]anthropic.ToolUnionParam, len(toolParams)) + for i, toolParam := range toolParams { + tools[i] = anthropic.ToolUnionParam{OfTool: &toolParam} + } + + for { + message, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + Model: anthropic.ModelClaudeOpus4_8, + MaxTokens: 1024, + Messages: messages, + Tools: tools, + }) + + if err != nil { + panic(err) + } + + print(color("[assistant]: ")) + for _, block := range message.Content { + switch block := block.AsAny().(type) { + case anthropic.TextBlock: + println(block.Text) + println() + case anthropic.ToolUseBlock: + inputJSON, _ := json.Marshal(block.Input) + println(block.Name + ": " + string(inputJSON)) + println() + } + } + + messages = append(messages, message.ToParam()) + toolResults := []anthropic.ContentBlockParamUnion{} + + for _, block := range message.Content { + switch variant := block.AsAny().(type) { + case anthropic.ToolUseBlock: + print(color("[user (" + block.Name + ")]: ")) + + var response interface{} + switch block.Name { + case "get_coordinates": + var input struct { + Location string `json:"location"` + } + + err := json.Unmarshal([]byte(variant.JSON.Input.Raw()), &input) + if err != nil { + panic(err) + } + + response = GetCoordinates(input.Location) + } + + b, err := json.Marshal(response) + if err != nil { + panic(err) + } + + println(string(b)) + + toolResults = append(toolResults, anthropic.NewToolResultBlock(block.ID, string(b), false)) + } + + } + if len(toolResults) == 0 { + break + } + messages = append(messages, anthropic.NewUserMessage(toolResults...)) + } + ``` + + + +## Request fields + +The anthropic library uses the [`omitzero`](https://tip.golang.org/doc/go1.24#encodingjsonpkgencodingjson) semantics from the Go 1.24+ `encoding/json` release for request fields. + +Required primitive fields (`int64`, `string`, etc.) feature the tag `` `json:"...,required"` ``. These fields are always serialized, even their zero values. + +Optional primitive types are wrapped in a `param.Opt[T]`. These fields can be set with the provided constructors, `anthropic.String(string)`, `anthropic.Int(int64)`, etc. + +Any `param.Opt[T]`, map, slice, struct or string enum uses the tag `` `json:"...,omitzero"` ``. Its zero value is considered omitted. + +The `param.IsOmitted(any)` function can confirm the presence of any `omitzero` field. + +```go +p := anthropic.ExampleParams{ + ID: "id_xxx", // required property + Name: anthropic.String("..."), // optional property + + Point: anthropic.Point{ + X: 0, // required field will serialize as 0 + Y: anthropic.Int(1), // optional field will serialize as 1 + // ... omitted non-required fields will not be serialized + }, + + Origin: anthropic.Origin{}, // the zero value of [Origin] is considered omitted +} +``` + +To send `null` instead of a `param.Opt[T]`, use `param.Null[T]()`. To send `null` instead of a struct `T`, use `param.NullStruct[T]()`. + +```go +p.Name = param.Null[string]() // 'null' instead of string +p.Point = param.NullStruct[Point]() // 'null' instead of struct + +param.IsNull(p.Name) // true +param.IsNull(p.Point) // true +``` + +Request structs contain a `.SetExtraFields(map[string]any)` method which can send non-conforming fields in the request body. Extra fields overwrite any struct fields with a matching key. + + + For security reasons, only use `SetExtraFields` with trusted data. + + +To send a custom value instead of a struct, use the generic function `param.Override` (for example, `param.Override[anthropic.FooParams](12)`). + +```go +// In cases where the API specifies a given type, +// but you want to send something else, use [SetExtraFields]: +p.SetExtraFields(map[string]any{ + "x": 0.01, // send "x" as a float instead of int +}) + +// Send a number instead of an object +custom := param.Override[anthropic.FooParams](12) +``` + +### Request unions + +Unions are represented as a struct with fields prefixed by "Of" for each of its variants, only one field can be non-zero. The non-zero field will be serialized. + +Subproperties of the union can be accessed through methods on the union struct. These methods return a mutable pointer to the underlying data, if present. + +```go +// Only one field can be non-zero, use param.IsOmitted() to check if a field is set +type AnimalUnionParam struct { + OfCat *Cat `json:",omitzero,inline"` + OfDog *Dog `json:",omitzero,inline"` +} + +animal := AnimalUnionParam{ + OfCat: &Cat{ + Name: "Whiskers", + Owner: PersonParam{ + Address: AddressParam{Street: "3333 Coyote Hill Rd", ZipCode: 0}, + }, + }, +} + +// Mutating a field +if address := animal.GetOwner().GetAddress(); address != nil { + address.ZipCode = 94304 +} +``` + +### Deserializing params + + + `param.SetJSON` requires SDK v1.20.0 or later. + + +Param types (types ending in `Param`, such as `MessageNewParams` or `ToolUnionParam`) are designed for outgoing requests only. They marshal correctly to JSON but do not fully support round-trip deserialization. If you unmarshal raw JSON into a param struct, typed union fields like `OfBashTool20250124` will be nil even when the underlying JSON is valid. + +If you need to reconstruct params from raw JSON (for example, from a database, middleware, or a previous request), call `UnmarshalJSON` to populate non-union fields, then use `param.SetJSON` to attach the raw bytes for correct re-serialization: + +```go +// Serialize params (for example, for storage or forwarding) +b, err := json.Marshal(original) +if err != nil { + panic(err) +} + +// Later, reconstruct params from the stored JSON +var params anthropic.MessageNewParams +if err := params.UnmarshalJSON(b); err != nil { + panic(err) +} +param.SetJSON(b, ¶ms) + +// params.Model and other scalar fields are populated by UnmarshalJSON. +// params.Tools[0].OfBashTool20250124 is nil (the union limitation), +// but the raw JSON is preserved. When params is marshaled again +// for the API call, the tools serialize correctly. +b2, _ := json.Marshal(params) +fmt.Println(string(b) == string(b2)) // true +``` + +For this use case, `param.SetJSON` (available since v1.20.0) is preferred over the more general `param.Override[T](any)` because it doesn't require spelling out the type parameter and makes the round-trip intent explicit. + +## Response objects + +All fields in response structs are ordinary value types (not pointers or wrappers). Response structs also include a special `JSON` field containing metadata about each property. + +```go +type Animal struct { + Name string `json:"name,nullable"` + Owners int `json:"owners"` + Age int `json:"age"` + JSON struct { + Name respjson.Field + Owners respjson.Field + Age respjson.Field + ExtraFields map[string]respjson.Field + } `json:"-"` +} +``` + +To handle optional data, use the `.Valid()` method on the JSON field. `.Valid()` returns true when the field is present, non-`null`, and was unmarshaled successfully. + +If `.Valid()` is false, the corresponding field will be its zero value. + +```go +raw := `{"owners": 1, "name": null}` + +var res Animal +json.Unmarshal([]byte(raw), &res) + +// Accessing regular fields + +res.Owners // 1 +res.Name // "" +res.Age // 0 + +// Optional field checks + +res.JSON.Owners.Valid() // true +res.JSON.Name.Valid() // false +res.JSON.Age.Valid() // false + +// Raw JSON values + +res.JSON.Owners.Raw() // "1" +res.JSON.Name.Raw() == "null" // true +res.JSON.Name.Raw() == respjson.Null // true +res.JSON.Age.Raw() == "" // true +res.JSON.Age.Raw() == respjson.Omitted // true +``` + +These `.JSON` structs also include an `ExtraFields` map containing any properties in the json response that were not specified in the struct. This can be useful for API features not yet present in the SDK. + +```go +body := res.JSON.ExtraFields["my_unexpected_field"].Raw() +``` + +### Response unions + +In responses, unions are represented by a flattened struct containing all possible fields from each of the object variants. To convert it to a variant use the `.AsFooVariant()` method or the `.AsAny()` method if present. + +If a response value union contains primitive values, primitive fields will be alongside the properties but prefixed with `Of` and feature the tag `json:"...,inline"`. + +```go +type AnimalUnion struct { + // From variants [Dog], [Cat] + Owner Person `json:"owner"` + // From variant [Dog] + DogBreed string `json:"dog_breed"` + // From variant [Cat] + CatBreed string `json:"cat_breed"` + // ... + + JSON struct { + Owner respjson.Field + // ... + } `json:"-"` +} + +// If animal variant +if animal.Owner.Address.ZipCode == "" { + panic("missing zip code") +} + +// Switch on the variant +switch variant := animal.AsAny().(type) { +case Dog: +case Cat: +default: + panic("unexpected type") +} +``` + +## Error handling + +When the API returns a non-success status code, the SDK returns an error with type `*anthropic.Error`. This contains the `StatusCode`, `*http.Request`, and `*http.Response` values of the request, as well as the JSON of the error body (much like other response objects in the SDK). The error also includes the `RequestID` from the response headers, which is useful for troubleshooting with Anthropic support. + +To handle errors, use the `errors.As` pattern: + +```go +_, err := client.Messages.New(context.TODO(), anthropic.MessageNewParams{ + MaxTokens: 1024, + Messages: []anthropic.MessageParam{{ + Content: []anthropic.ContentBlockParamUnion{{ + OfText: &anthropic.TextBlockParam{ + Text: "What is a quaternion?", + }, + }}, + Role: anthropic.MessageParamRoleUser, + }}, + Model: anthropic.ModelClaudeOpus4_8, +}) +if err != nil { + var apierr *anthropic.Error + if errors.As(err, &apierr) { + println("Request ID:", apierr.RequestID) + println(string(apierr.DumpRequest(true))) // Prints the serialized HTTP request + println(string(apierr.DumpResponse(true))) // Prints the serialized HTTP response + } + panic(err.Error()) // POST "/v1/messages": 400 Bad Request (Request-ID: req_xxx) { ... } +} +``` + +When other errors occur, they are returned unwrapped; for example, if HTTP transport fails, you might receive `*url.Error` wrapping `*net.OpError`. + +## Retries + +Certain errors will be automatically retried 2 times by default, with a short exponential backoff. The SDK retries by default all connection errors, 408 Request Timeout, 409 Conflict, 429 Rate Limit, and >=500 Internal errors. + +You can use the `WithMaxRetries` option to configure or disable this: + +```go +// Configure the default for all requests: +client := anthropic.NewClient( + option.WithMaxRetries(0), // default is 2 +) + +// Override per-request: +// ... + client.Messages.New( + context.TODO(), + anthropic.MessageNewParams{ + MaxTokens: 1024, + Messages: []anthropic.MessageParam{{ + Content: []anthropic.ContentBlockParamUnion{{ + OfText: &anthropic.TextBlockParam{ + Text: "What is a quaternion?", + }, + }}, + Role: anthropic.MessageParamRoleUser, + }}, + Model: anthropic.ModelClaudeOpus4_8, + }, + option.WithMaxRetries(5), + ) +``` + +## Timeouts + +Non-streaming Messages requests time out after 10 minutes by default; other requests have no default timeout. Use context to configure a timeout for a request lifecycle. + +Note that if a request is [retried](#retries), the context timeout does not start over. To set a per-retry timeout, use `option.WithRequestTimeout()`. + +```go +// This sets the timeout for the request, including all the retries. +ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute) +defer cancel() +// ... + client.Messages.New( + ctx, + anthropic.MessageNewParams{ + MaxTokens: 1024, + Messages: []anthropic.MessageParam{{ + Content: []anthropic.ContentBlockParamUnion{{ + OfText: &anthropic.TextBlockParam{ + Text: "What is a quaternion?", + }, + }}, + Role: anthropic.MessageParamRoleUser, + }}, + Model: anthropic.ModelClaudeOpus4_8, + }, + // This sets the per-retry timeout + option.WithRequestTimeout(20*time.Second), + ) +``` + +## Long requests + + + Consider using the streaming Messages API for longer running requests. + + +Avoid setting a large `MaxTokens` value without using streaming as some networks may drop idle connections after a certain period of time, which can cause the request to fail or [timeout](#timeouts) without receiving a response from Anthropic. + +This SDK will also return an error if a non-streaming request is expected to be above roughly 10 minutes long. Calling `.Messages.NewStreaming()` or [setting a custom timeout](#timeouts) disables this error. + +## File uploads + +Request parameters that correspond to file uploads in multipart requests are typed as `io.Reader`. The contents of the `io.Reader` will by default be sent as a multipart form part with the file name of "anonymous\_file" and content-type of "application/octet-stream", so the recommended approach is to specify a custom content-type with the `anthropic.File(reader io.Reader, filename string, contentType string)` helper, which wraps any `io.Reader` with the appropriate file name and content type. + +```go +// A file from the file system +file, err := os.Open("/path/to/file.json") +anthropic.BetaFileUploadParams{ + File: anthropic.File(file, "custom-name.json", "application/json"), +} + +// A file from a string +anthropic.BetaFileUploadParams{ + File: anthropic.File(strings.NewReader("my file contents"), "custom-name.json", "application/json"), +} +``` + +The file name and content-type can also be customized by implementing `Name() string` or `ContentType() string` on the run-time type of `io.Reader`. Note that `os.File` implements `Name() string`, so a file returned by `os.Open` will be sent with the file name on disk. + +## Pagination + +This library provides some conveniences for working with paginated list endpoints. + +You can use `.ListAutoPaging()` methods to iterate through items across all pages: + +```go +iter := client.Messages.Batches.ListAutoPaging(context.TODO(), anthropic.MessageBatchListParams{ + Limit: anthropic.Int(20), +}) +// Automatically fetches more pages as needed. +for iter.Next() { + messageBatch := iter.Current() + fmt.Printf("%+v\n", messageBatch) +} +if err := iter.Err(); err != nil { + panic(err.Error()) +} +``` + +Or you can use simple `.List()` methods to fetch a single page and receive a standard response object with additional helper methods like `.GetNextPage()`: + +```go +page, err := client.Messages.Batches.List(context.TODO(), anthropic.MessageBatchListParams{ + Limit: anthropic.Int(20), +}) +for page != nil { + for _, batch := range page.Data { + fmt.Printf("%+v\n", batch) + } + page, err = page.GetNextPage() +} +if err != nil { + panic(err.Error()) +} +``` + +## RequestOptions + +This library uses the functional options pattern. Functions defined in the `option` package return a `RequestOption`, which is a closure that mutates a `RequestConfig`. These options can be supplied to the client or at individual requests. For example: + +```go +client := anthropic.NewClient( + // Adds a header to every request made by the client + option.WithHeader("X-Some-Header", "custom_header_info"), +) + +client.Messages.New(context.TODO(), // ..., + // Override the header + option.WithHeader("X-Some-Header", "some_other_custom_header_info"), + // Add an undocumented field to the request body, using sjson syntax + option.WithJSONSet("some.json.path", map[string]string{"my": "object"}), +) +``` + +The request option `option.WithDebugLog(nil)` may be helpful while debugging. + +See the [full list of request options](https://pkg.go.dev/github.com/anthropics/anthropic-sdk-go/option). + +## HTTP client customization + +For request middleware (`option.WithMiddleware`) and replacing the default `http.Client` (`option.WithHTTPClient`), see [SDK middleware](/docs/en/cli-sdks-libraries/middleware). + +## Platform integrations + + + For detailed platform setup guides with code examples, see: + + * [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) + * [Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy) + * [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) + * [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) + + +The Go SDK supports the following platforms: + +* **Agent Platform:** `import "github.com/anthropics/anthropic-sdk-go/vertex"`. Use `vertex.WithGoogleAuth(ctx, region, projectID)` or `vertex.WithCredentials(ctx, region, projectID, creds)`. +* **Bedrock:** `import "github.com/anthropics/anthropic-sdk-go/bedrock"`. Use `bedrock.NewMantleClient` for the Messages-API Bedrock endpoint (streams over SSE), or `bedrock.WithLoadDefaultConfig(ctx)` / `bedrock.WithConfig(cfg)` (`bedrock-runtime` path). Importing the `bedrock` package globally registers a decoder for `application/vnd.amazon.eventstream` with the SDK's streaming layer (through package `init()`). This applies whether you use the `bedrock-runtime` `WithConfig`/`WithLoadDefaultConfig` path or `NewMantleClient`. +* **Claude Platform on AWS:** `import anthropicaws "github.com/anthropics/anthropic-sdk-go/aws"`. Use `anthropicaws.NewClient(ctx, cfg)` with an `anthropicaws.ClientConfig` value to construct a client; set `WorkspaceID` on the config or the `ANTHROPIC_AWS_WORKSPACE_ID` environment variable. The `anthropicaws` import alias avoids a name collision with `github.com/aws/aws-sdk-go-v2/aws` when both are imported. Available in beta. +* **Foundry:** Not currently supported in the Go SDK. See [Claude in Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) for supported SDKs. + +Use `bedrock.NewMantleClient` for new projects; `bedrock.WithLoadDefaultConfig`/`WithConfig` remain for existing applications using the Bedrock `InvokeModel` API. + +## Advanced usage + +### Accessing raw response data (for example, response headers) + +You can access the raw HTTP response data by using the `option.WithResponseInto()` request option. This is useful when you need to examine response headers, status codes, or other details. + +```go +// Create a variable to store the HTTP response +var response *http.Response +message, err := client.Messages.New( + context.TODO(), + anthropic.MessageNewParams{ + MaxTokens: 1024, + Messages: []anthropic.MessageParam{{ + Content: []anthropic.ContentBlockParamUnion{{ + OfText: &anthropic.TextBlockParam{ + Text: "What is a quaternion?", + }, + }}, + Role: anthropic.MessageParamRoleUser, + }}, + Model: anthropic.ModelClaudeOpus4_8, + }, + option.WithResponseInto(&response), +) +if err != nil { + // handle error +} +fmt.Printf("%+v\n", message) + +fmt.Printf("Status Code: %d\n", response.StatusCode) +fmt.Printf("Headers: %+#v\n", response.Header) +``` + +### Making custom/undocumented requests + +This library is typed for convenient access to the documented API. If you need to access undocumented endpoints, params, or response properties, the library can still be used. + +#### Undocumented endpoints + +To make requests to undocumented endpoints, you can use `client.Get`, `client.Post`, and other HTTP verbs. `RequestOptions` on the client, such as retries, will be respected when making these requests. + +```go +var ( + // params can be an io.Reader, a []byte, an encoding/json serializable object, + // or a "...Params" struct defined in this library. + params map[string]any + + // result can be an []byte, *http.Response, a encoding/json deserializable object, + // or a model defined in this library. + result *http.Response +) +err := client.Post(context.Background(), "/unspecified", params, &result) +if err != nil { + // ... +} +``` + +#### Undocumented request params + +To make requests using undocumented parameters, you may use either the `option.WithQuerySet()` or the `option.WithJSONSet()` methods. + +```go +params := FooNewParams{ + ID: "id_xxxx", + Data: FooNewParamsData{ + FirstName: anthropic.String("John"), + }, +} +client.Foo.New(context.Background(), params, option.WithJSONSet("data.last_name", "Doe")) +``` + +#### Undocumented response properties + +To access undocumented response properties, you may either access the raw JSON of the response as a string with `result.JSON.RawJSON()`, or get the raw JSON of a particular field on the result with `result.JSON.Foo.Raw()`. + +Any fields that are not present on the response struct are saved and can be accessed through `result.JSON.ExtraFields`, which is a `map[string]respjson.Field`. + +## Semantic versioning + +This package generally follows [SemVer](https://semver.org/spec/v2.0.0.html) conventions, though certain backwards-incompatible changes may be released as minor versions: + +1. Changes to library internals that are technically public but not intended or documented for external use. +2. Changes that aren't expected to impact the vast majority of users in practice. + +Backwards-compatibility is taken seriously to ensure you can rely on a smooth upgrade experience. + +Your feedback is welcome; open an [issue](https://github.com/anthropics/anthropic-sdk-go/issues) with questions, bugs, or suggestions. + +## Additional resources + +* [GitHub repository](https://github.com/anthropics/anthropic-sdk-go) +* [Go package documentation](https://pkg.go.dev/github.com/anthropics/anthropic-sdk-go) +* [API reference](/docs/en/api/overview) +* [Streaming Messages](/docs/en/build-with-claude/streaming) diff --git a/content/en/cli-sdks-libraries/sdks/java.md b/content/en/cli-sdks-libraries/sdks/java.md new file mode 100644 index 000000000..506491f1e --- /dev/null +++ b/content/en/cli-sdks-libraries/sdks/java.md @@ -0,0 +1,1280 @@ +# Java SDK + +Install and configure the Anthropic Java SDK with builder patterns and async support + +--- + +The Anthropic Java SDK provides convenient access to the Anthropic REST API from applications written in Java. It uses the builder pattern for creating requests and supports both synchronous and asynchronous operations. + + + For API feature documentation with code examples, see the [API reference](/docs/en/api/overview). This page covers Java-specific SDK features and configuration. + + +## Installation + + + + ```kotlin + implementation("com.anthropic:anthropic-java:2.48.0") + ``` + + + + ```xml + + com.anthropic + anthropic-java + 2.48.0 + + ``` + + + +## Requirements + +This library requires Java 8 or later. + + + The SDK supports Java 8 and later. Code examples in this documentation are written as [JDK 25 compact source files](https://openjdk.org/jeps/512), using a bare `void main()` entry point and `IO.println()` for output. The API calls themselves are identical on every supported JDK; to compile an example on an earlier version, replace `IO.println(...)` with `System.out.println(...)` and place the body inside `public static void main(String[] args)` within a class. + + +## Quick start + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; +import com.anthropic.models.messages.Message; +import com.anthropic.models.messages.MessageCreateParams; +import com.anthropic.models.messages.Model; + +// Configures using the `anthropic.apiKey`, `anthropic.authToken` and `anthropic.baseUrl` system properties +// Or configures using the `ANTHROPIC_API_KEY`, `ANTHROPIC_AUTH_TOKEN` and `ANTHROPIC_BASE_URL` environment variables +AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + +MessageCreateParams params = MessageCreateParams.builder() + .maxTokens(1024L) + .addUserMessage("Hello, Claude") + .model(Model.CLAUDE_OPUS_4_8) + .build(); + +Message message = client.messages().create(params); +``` + +## Client configuration + +### API key setup + +Configure the client using system properties or environment variables: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; + +// Configures using the `anthropic.apiKey`, `anthropic.authToken` and `anthropic.baseUrl` system properties +// Or configures using the `ANTHROPIC_API_KEY`, `ANTHROPIC_AUTH_TOKEN` and `ANTHROPIC_BASE_URL` environment variables +AnthropicClient client = AnthropicOkHttpClient.fromEnv(); +``` + +Or configure manually: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; + +AnthropicClient client = AnthropicOkHttpClient.builder() + .apiKey("my-anthropic-api-key") + .build(); +``` + +Or use a combination of both approaches: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; + +AnthropicClient client = AnthropicOkHttpClient.builder() + // Configures using system properties or environment variables + .fromEnv() + .apiKey("my-anthropic-api-key") + .build(); +``` + +For authentication options including Workload Identity Federation, see [Authentication](/docs/en/manage-claude/authentication). + +### Configuration options + +| Setter | System property | Environment variable | Required | Default value | +| ----------- | --------------------- | ---------------------- | -------- | ----------------------------- | +| `apiKey` | `anthropic.apiKey` | `ANTHROPIC_API_KEY` | false | - | +| `authToken` | `anthropic.authToken` | `ANTHROPIC_AUTH_TOKEN` | false | - | +| `baseUrl` | `anthropic.baseUrl` | `ANTHROPIC_BASE_URL` | true | `"https://api.anthropic.com"` | + +System properties take precedence over environment variables. + + + Don't create more than one client in the same application. Each client has a connection pool and thread pools, which are more efficient to share between requests. + + +### Modifying configuration + +To temporarily use a modified client configuration while reusing the same connection and thread pools, call `withOptions()` on any client or service: + +```java +import com.anthropic.client.AnthropicClient; + +AnthropicClient clientWithOptions = client.withOptions(optionsBuilder -> { + optionsBuilder.baseUrl("https://example.com"); + optionsBuilder.maxRetries(42); +}); +``` + +The `withOptions()` method does not affect the original client or service. + +## Async usage + +The default client is synchronous. To switch to asynchronous execution, call the `async()` method: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; +import com.anthropic.models.messages.Message; +import com.anthropic.models.messages.MessageCreateParams; +import com.anthropic.models.messages.Model; + +AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + +MessageCreateParams params = MessageCreateParams.builder() + .maxTokens(1024L) + .addUserMessage("Hello, Claude") + .model(Model.CLAUDE_OPUS_4_8) + .build(); + +CompletableFuture message = client.async().messages().create(params); +``` + +Or create an asynchronous client from the beginning: + +```java +import com.anthropic.client.AnthropicClientAsync; +import com.anthropic.client.okhttp.AnthropicOkHttpClientAsync; +import com.anthropic.models.messages.Message; +import com.anthropic.models.messages.MessageCreateParams; +import com.anthropic.models.messages.Model; + +AnthropicClientAsync client = AnthropicOkHttpClientAsync.fromEnv(); + +MessageCreateParams params = MessageCreateParams.builder() + .maxTokens(1024L) + .addUserMessage("Hello, Claude") + .model(Model.CLAUDE_OPUS_4_8) + .build(); + +CompletableFuture message = client.messages().create(params); +``` + +The asynchronous client supports the same options as the synchronous one, except most methods return `CompletableFuture`s. + +## Streaming + +The SDK defines methods that return response "chunk" streams, where each chunk can be individually processed as soon as it arrives instead of waiting on the full response. + +### Synchronous streaming + +These streaming methods return `StreamResponse` for synchronous clients: + +```java +import com.anthropic.core.http.StreamResponse; +import com.anthropic.models.messages.RawMessageStreamEvent; + +try (StreamResponse streamResponse = client.messages().createStreaming(params)) { + streamResponse.stream().forEach(chunk -> { + IO.println(chunk); + }); + IO.println("No more chunks!"); +} +``` + +### Asynchronous streaming + +For asynchronous clients, the method returns `AsyncStreamResponse`: + +```java +import com.anthropic.core.http.AsyncStreamResponse; +import com.anthropic.models.messages.RawMessageStreamEvent; + +client.async().messages().createStreaming(params).subscribe(chunk -> { + IO.println(chunk); +}); + +// If you need to handle errors or completion of the stream +client.async().messages().createStreaming(params).subscribe(new AsyncStreamResponse.Handler<>() { + @Override + public void onNext(RawMessageStreamEvent chunk) { + IO.println(chunk); + } + + @Override + public void onComplete(Optional error) { + if (error.isPresent()) { + IO.println("Something went wrong!"); + throw new RuntimeException(error.get()); + } else { + IO.println("No more chunks!"); + } + } +}); + +// Or use futures +client.async().messages().createStreaming(params) + .subscribe(chunk -> { + IO.println(chunk); + }) + .onCompleteFuture() + .whenComplete((unused, error) -> { + if (error != null) { + IO.println("Something went wrong!"); + throw new RuntimeException(error); + } else { + IO.println("No more chunks!"); + } + }); +``` + +Async streaming uses a dedicated per-client cached thread pool `Executor` to stream without blocking the current thread. To use a different `Executor`: + +```java +Executor executor = Executors.newFixedThreadPool(4); +client.async().messages().createStreaming(params).subscribe( + chunk -> IO.println(chunk), executor +); +``` + +Or configure the client globally using the `streamHandlerExecutor` method: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; + +AnthropicClient client = AnthropicOkHttpClient.builder() + .fromEnv() + .streamHandlerExecutor(Executors.newFixedThreadPool(4)) + .build(); +``` + +### Streaming with message accumulator + +A `MessageAccumulator` can record the stream of events in the response as they are processed and accumulate a `Message` object similar to what would have been returned by the non-streaming API. + +For a synchronous response, add a `Stream.peek()` call to the stream pipeline to accumulate each event: + +```java +import com.anthropic.core.http.StreamResponse; +import com.anthropic.helpers.MessageAccumulator; +import com.anthropic.models.messages.Message; +import com.anthropic.models.messages.RawMessageStreamEvent; + +MessageAccumulator messageAccumulator = MessageAccumulator.create(); + +try (StreamResponse streamResponse = + client.messages().createStreaming(createParams)) { + streamResponse.stream() + .peek(messageAccumulator::accumulate) + .flatMap(event -> event.contentBlockDelta().stream()) + .flatMap(deltaEvent -> deltaEvent.delta().text().stream()) + .forEach(textDelta -> IO.print(textDelta.text())); +} + +Message message = messageAccumulator.message(); +``` + +For an asynchronous response, add the `MessageAccumulator` to the `subscribe()` call: + +```java +import com.anthropic.helpers.MessageAccumulator; +import com.anthropic.models.messages.Message; + +MessageAccumulator messageAccumulator = MessageAccumulator.create(); + +client.async().messages() + .createStreaming(createParams) + .subscribe(event -> messageAccumulator.accumulate(event).contentBlockDelta().stream() + .flatMap(deltaEvent -> deltaEvent.delta().text().stream()) + .forEach(textDelta -> IO.print(textDelta.text()))) + .onCompleteFuture() + .join(); + +Message message = messageAccumulator.message(); +``` + +A `BetaMessageAccumulator` is also available for the accumulation of a `BetaMessage` object. It is used in the same manner as the `MessageAccumulator`. + +## Structured outputs + +For complete structured outputs documentation including Java examples, see [Structured outputs](/docs/en/build-with-claude/structured-outputs). + +## Tool use + +[Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview) lets you integrate external tools and functions directly into the AI model's responses. Instead of producing plain text, the model can output instructions (with parameters) for calling a tool or function when appropriate. You define JSON schemas for tools, and the model uses the schemas to determine when and how to use these tools. + +The tool use feature supports a "strict" mode that guarantees that the JSON output from the AI model will conform to the JSON schema you provide in the input parameters. + +The SDK can derive a tool and its parameters automatically from the structure of an arbitrary Java class: the class's name (converted to snake case) provides the tool name, and the class's fields define the tool's parameters. + + + Declare your tool classes as top-level classes or `static` nested classes. This requirement comes from the Jackson Databind library (`com.fasterxml.jackson.databind`), which the SDK uses to deserialize tool inputs into your class instances and cannot instantiate non-static inner classes. + + +### Defining tools with annotations + +```java +import com.fasterxml.jackson.annotation.JsonClassDescription; +import com.fasterxml.jackson.annotation.JsonPropertyDescription; + +enum Unit { + CELSIUS, + FAHRENHEIT; + + public String toString() { + return switch (this) { + case CELSIUS -> "C"; + case FAHRENHEIT -> "F"; + }; + } + + public double fromKelvin(double temperatureK) { + return switch (this) { + case CELSIUS -> temperatureK - 273.15; + case FAHRENHEIT -> (temperatureK - 273.15) * 1.8 + 32.0; + }; + } +} + +@JsonClassDescription("Get the weather in a given location") +static class GetWeather { + + @JsonPropertyDescription("The city and state, e.g. San Francisco, CA") + public String location; + + @JsonPropertyDescription("The unit of temperature") + public Unit unit; + + public Weather execute() { + double temperatureK = switch (location) { + case "San Francisco, CA" -> 300.0; + case "New York, NY" -> 310.0; + case "Dallas, TX" -> 305.0; + default -> 295; + }; + return new Weather(String.format("%.0f%s", unit.fromKelvin(temperatureK), unit)); + } +} + +static class Weather { + + public String temperature; + + public Weather(String temperature) { + this.temperature = temperature; + } +} +``` + +### Calling tools + +When your tool classes are defined, add them to the message parameters using `MessageCreateParams.Builder.addTool(Class)` and then call them if requested to do so in the AI model's response. `BetaToolUseBlock.input(Class)` can be used to parse a tool's parameters in JSON form to an instance of your tool-defining class. + +After calling the tool, use `BetaToolResultBlockParam.Builder.contentAsJson(Object)` to pass the tool's result back to the AI model: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; +import com.anthropic.models.beta.messages.*; +import com.anthropic.models.messages.Model; + +AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + +MessageCreateParams.Builder createParamsBuilder = MessageCreateParams.builder() + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(2048) + .addTool(GetWeather.class) + .addUserMessage("What's the temperature in New York?"); + +client.beta().messages().create(createParamsBuilder.build()).content().stream() + .flatMap(contentBlock -> contentBlock.toolUse().stream()) + .forEach(toolUseBlock -> createParamsBuilder + // Add a message indicating that the tool use was requested. + .addAssistantMessageOfBetaContentBlockParams( + List.of(BetaContentBlockParam.ofToolUse(BetaToolUseBlockParam.builder() + .name(toolUseBlock.name()) + .id(toolUseBlock.id()) + .input(toolUseBlock._input()) + .build()))) + // Add a message with the result of the requested tool use. + .addUserMessageOfBetaContentBlockParams( + List.of(BetaContentBlockParam.ofToolResult(BetaToolResultBlockParam.builder() + .toolUseId(toolUseBlock.id()) + .contentAsJson(callTool(toolUseBlock)) + .build())))); + +client.beta().messages().create(createParamsBuilder.build()).content().stream() + .flatMap(contentBlock -> contentBlock.text().stream()) + .forEach(textBlock -> IO.println(textBlock.text())); + +private static Object callTool(BetaToolUseBlock toolUseBlock) { + if (!"get_weather".equals(toolUseBlock.name())) { + throw new IllegalArgumentException("Unknown tool: " + toolUseBlock.name()); + } + + GetWeather tool = toolUseBlock.input(GetWeather.class); + return tool != null ? tool.execute() : new Weather("unknown"); +} +``` + +### Tool name conversion + +Tool names are derived from the camel case tool class names (for example, `GetWeather`) and converted to snake case (for example, `get_weather`). Word boundaries begin where the current character is not the first character, is upper-case, and either the preceding character is lower-case, or the following character is lower-case. For example, `MyJSONParser` becomes `my_json_parser` and `ParseJSON` becomes `parse_json`. This conversion can be overridden using the `@JsonTypeName` annotation. + +### Local tool JSON schema validation + +You can perform local validation to check that the JSON schema derived from your tool class respects Anthropic's restrictions. Local validation is enabled by default, but it can be disabled: + +```java +MessageCreateParams.Builder createParamsBuilder = MessageCreateParams.builder() + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(2048) + .addTool(GetWeather.class, JsonSchemaLocalValidation.NO) + .addUserMessage("What's the temperature in New York?"); +``` + +### Annotating tool classes + +You can use annotations to add further information about tools to the JSON schemas: + +* `@JsonClassDescription` - Add a description to a tool class detailing when and how to use that tool. +* `@JsonTypeName` - Set the tool name to something other than the simple name of the class converted to snake case. +* `@JsonPropertyDescription` - Add a detailed description to a tool parameter. +* `@JsonIgnore` - Exclude a `public` field or getter method from the generated JSON schema for a tool's parameters. +* `@JsonProperty` - Include a non-`public` field or getter method in the generated JSON schema for a tool's parameters. + +## Message batches + +The SDK provides support for [Batch processing](/docs/en/build-with-claude/batch-processing) under the `client.messages().batches()` namespace. See [Pagination](#pagination) for how to list and paginate through batches. + +## File uploads + +The SDK defines methods that accept files through the `MultipartField` class: + +```java +import com.anthropic.core.MultipartField; +import com.anthropic.models.beta.files.FileMetadata; +import com.anthropic.models.beta.files.FileUploadParams; + +FileUploadParams params = FileUploadParams.builder() + .file( + MultipartField.builder() + .value(Files.newInputStream(Paths.get("/path/to/file.pdf"))) + .contentType("application/pdf") + .build() + ) + .build(); + +FileMetadata fileMetadata = client.beta().files().upload(params); +``` + +Or from an `InputStream`: + +```java +import com.anthropic.core.MultipartField; +import com.anthropic.models.beta.files.FileMetadata; +import com.anthropic.models.beta.files.FileUploadParams; + +FileUploadParams params = FileUploadParams.builder() + .file( + MultipartField.builder() + .value(URI.create("https://example.com/path/to/file").toURL().openStream()) + .filename("document.pdf") + .contentType("application/pdf") + .build() + ) + .build(); + +FileMetadata fileMetadata = client.beta().files().upload(params); +``` + +Or from in-memory bytes: + +```java +import com.anthropic.core.MultipartField; +import com.anthropic.models.beta.files.FileMetadata; +import com.anthropic.models.beta.files.FileUploadParams; + +FileUploadParams params = FileUploadParams.builder() + .file( + MultipartField.builder() + .value(new ByteArrayInputStream("content".getBytes())) + .filename("document.txt") + .contentType("text/plain") + .build() + ) + .build(); + +FileMetadata fileMetadata = client.beta().files().upload(params); +``` + +### Binary responses + +The SDK defines methods that return binary responses for API responses that aren't necessarily parsed as JSON: + +```java +import com.anthropic.core.http.HttpResponse; + +HttpResponse response = client.beta().files().download("file_abc123"); +``` + +To save the response content to a file: + +```java +import com.anthropic.core.http.HttpResponse; + +try (HttpResponse response = client.beta().files().download(params)) { + Files.copy( + response.body(), + Paths.get(path), + StandardCopyOption.REPLACE_EXISTING + ); +} catch (Exception e) { + IO.println("Something went wrong!"); + throw new RuntimeException(e); +} +``` + +Or transfer the response content to any `OutputStream`: + +```java +import com.anthropic.core.http.HttpResponse; + +try (HttpResponse response = client.beta().files().download(params)) { + response.body().transferTo(Files.newOutputStream(Paths.get(path))); +} catch (Exception e) { + IO.println("Something went wrong!"); + throw new RuntimeException(e); +} +``` + +## Error handling + +The SDK throws custom unchecked exception types: + +* `AnthropicServiceException` - Base class for HTTP errors. +* `AnthropicIoException` - I/O networking errors. +* `AnthropicRetryableException` - Generic error indicating a failure that could be retried. +* `AnthropicInvalidDataException` - Failure to interpret successfully parsed data (for example, when accessing a property that's supposed to be required, but the API unexpectedly omitted it). +* `AnthropicException` - Base class for all exceptions. + +### Status code mapping + +| Status | Exception | +| ------ | ------------------------------- | +| 400 | `BadRequestException` | +| 401 | `UnauthorizedException` | +| 403 | `PermissionDeniedException` | +| 404 | `NotFoundException` | +| 422 | `UnprocessableEntityException` | +| 429 | `RateLimitException` | +| 5xx | `InternalServerException` | +| others | `UnexpectedStatusCodeException` | + +`SseException` is thrown for errors encountered during SSE streaming after a successful initial HTTP response. + +```java +import com.anthropic.errors.*; + +try { + Message message = client.messages().create(params); +} catch (RateLimitException e) { + IO.println("Rate limited, retry after: " + e.headers()); +} catch (UnauthorizedException e) { + IO.println("Invalid API key"); +} catch (AnthropicServiceException e) { + IO.println("API error: " + e.statusCode()); +} catch (AnthropicIoException e) { + IO.println("Network error: " + e.getMessage()); +} +``` + +## Request IDs + +When using [raw responses](#raw-response-access), you can access the `request-id` response header using the `requestId()` method: + +```java +import com.anthropic.core.http.HttpResponseFor; +import com.anthropic.models.messages.Message; + +HttpResponseFor message = client.messages().withRawResponse().create(params); + +Optional requestId = message.requestId(); +``` + +This can be used to quickly log failing requests and report them back to Anthropic. For more information on debugging requests, see [Request ID](/docs/en/api/errors#request-id). + +## Retries + +The SDK automatically retries 2 times by default, with a short exponential backoff between requests. + +Only the following error types are retried: + +* Connection errors (for example, because of a network connectivity problem) +* 408 Request Timeout +* 409 Conflict +* 429 Rate Limit +* 5xx Internal + +The API may also explicitly instruct the SDK to retry or not retry a request. + +To set a custom number of retries, configure the client using the `maxRetries` method: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; + +AnthropicClient client = AnthropicOkHttpClient.builder().fromEnv().maxRetries(4).build(); +``` + +## Timeouts + +Requests time out after 10 minutes by default. + +However, for methods that accept `maxTokens`, if you specify a large `maxTokens` value and are streaming, then the default timeout will be calculated dynamically using this formula: + +```java +Duration.ofSeconds( + Math.min( + 60 * 60, // 1 hour max + Math.max( + 10 * 60, // 10 minute minimum + 60 * 60 * maxTokens / 128_000 + ) + ) +) +``` + +This results in a timeout of up to 60 minutes, scaled by the `maxTokens` parameter, unless overridden. + +For non-streaming requests, the dynamic timeout scales from a 30 second minimum up to a 10 minute maximum based on `maxTokens`. + +To set a custom timeout per-request: + +```java +import com.anthropic.models.messages.Message; + +Message message = client + .messages() + .create(params, RequestOptions.builder().timeout(Duration.ofSeconds(30)).build()); +``` + +Or configure the default for all method calls at the client level: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; + +AnthropicClient client = AnthropicOkHttpClient.builder() + .fromEnv() + .timeout(Duration.ofSeconds(30)) + .build(); +``` + +## Long requests + + + Consider using [streaming](#streaming) for longer running requests. + + +Avoid setting a large `maxTokens` value without using streaming. Some networks may drop idle connections after a certain period of time, which can cause the request to fail or [timeout](#timeouts) without receiving a response from Anthropic. The SDK periodically pings the API to keep the connection alive and reduce the impact of these networks. + +The SDK throws an error if a non-streaming request is expected to take longer than 10 minutes. Using a [streaming method](#streaming) or [overriding the timeout](#timeouts) at the client or request level disables the error. + +## Pagination + +The SDK provides convenient ways to access paginated results either one page at a time or item-by-item across all pages. + +### Auto-pagination + +To iterate through all results across all pages, use the `autoPager()` method, which automatically fetches more pages as needed. + +```java +import com.anthropic.models.messages.batches.BatchListPage; +import com.anthropic.models.messages.batches.MessageBatch; + +BatchListPage page = client.messages().batches().list(); + +// Process as an Iterable +for (MessageBatch batch : page.autoPager()) { + IO.println(batch); +} + +// Process as a Stream +page.autoPager() + .stream() + .limit(50) + .forEach(batch -> IO.println(batch)); +``` + +When using the asynchronous client, the method returns an `AsyncStreamResponse`: + +```java +import com.anthropic.core.http.AsyncStreamResponse; +import com.anthropic.models.messages.batches.BatchListPageAsync; +import com.anthropic.models.messages.batches.MessageBatch; + +CompletableFuture pageFuture = client.async().messages().batches().list(); + +pageFuture.thenAccept(page -> page.autoPager().subscribe(batch -> { + IO.println(batch); +})); + +// If you need to handle errors or completion of the stream +pageFuture.thenAccept(page -> page.autoPager().subscribe(new AsyncStreamResponse.Handler<>() { + @Override + public void onNext(MessageBatch batch) { + IO.println(batch); + } + + @Override + public void onComplete(Optional error) { + if (error.isPresent()) { + IO.println("Something went wrong!"); + throw new RuntimeException(error.get()); + } else { + IO.println("No more!"); + } + } +})); + +// Or use futures +pageFuture.thenAccept(page -> page.autoPager() + .subscribe(batch -> { + IO.println(batch); + }) + .onCompleteFuture() + .whenComplete((unused, error) -> { + if (error != null) { + IO.println("Something went wrong!"); + throw new RuntimeException(error); + } else { + IO.println("No more!"); + } + })); +``` + +### Manual pagination + +To access individual page items and manually request the next page: + +```java +import com.anthropic.models.messages.batches.BatchListPage; +import com.anthropic.models.messages.batches.MessageBatch; + +BatchListPage page = client.messages().batches().list(); +while (true) { + for (MessageBatch batch : page.items()) { + IO.println(batch); + } + + if (!page.hasNextPage()) { + break; + } + + page = page.nextPage(); +} +``` + +## Type system + +### Immutability and builders + +Each class in the SDK has an associated builder for constructing it. Each class is immutable once constructed. If the class has an associated builder, then it has a `toBuilder()` method, which can be used to convert it back to a builder for making a modified copy. + +```java +MessageCreateParams params = MessageCreateParams.builder() + .maxTokens(1024L) + .addUserMessage("Hello, Claude") + .model(Model.CLAUDE_OPUS_4_8) + .build(); + +// Create a modified copy using toBuilder() +MessageCreateParams modified = params.toBuilder().maxTokens(2048L).build(); +``` + +Because each class is immutable, builder modification never affects already built class instances. + +### Requests and responses + +To send a request to the Claude API, build an instance of some `Params` class and pass it to the corresponding client method. When the response is received, it is deserialized into an instance of a Java class. + +For example, `client.messages().create(...)` should be called with an instance of `MessageCreateParams`, and it returns an instance of `Message`. + +### Undocumented parameters + +To set undocumented parameters, call the `putAdditionalHeader`, `putAdditionalQueryParam`, or `putAdditionalBodyProperty` methods on any `Params` class: + +```java +import com.anthropic.core.JsonValue; +import com.anthropic.models.messages.MessageCreateParams; + +MessageCreateParams params = MessageCreateParams.builder() + .putAdditionalHeader("Secret-Header", "42") + .putAdditionalQueryParam("secret_query_param", "42") + .putAdditionalBodyProperty("secretProperty", JsonValue.from("42")) + .build(); +``` + +These can be accessed on the built object later using the `_additionalHeaders()`, `_additionalQueryParams()`, and `_additionalBodyProperties()` methods. + + + The values passed to these methods overwrite values passed to earlier methods. For security reasons, ensure these methods are only used with trusted input data. + + +To set undocumented parameters on nested headers, query params, or body classes: + +```java +import com.anthropic.core.JsonValue; +import com.anthropic.models.messages.MessageCreateParams; +import com.anthropic.models.messages.Metadata; + +MessageCreateParams params = MessageCreateParams.builder() + .metadata( + Metadata.builder().putAdditionalProperty("secretProperty", JsonValue.from("42")).build() + ) + .build(); +``` + +These properties can be accessed on the nested built object later using the `_additionalProperties()` method. + +To set a documented parameter or property to an undocumented or not yet supported value, pass a `JsonValue` object to its setter: + +```java +import com.anthropic.core.JsonValue; +import com.anthropic.models.messages.MessageCreateParams; +import com.anthropic.models.messages.Model; + +MessageCreateParams params = MessageCreateParams.builder() + .maxTokens(JsonValue.from(3.14)) + .addUserMessage("Hello, Claude") + .model(Model.CLAUDE_OPUS_4_8) + .build(); +``` + +### JsonValue creation + +The most straightforward way to create a `JsonValue` is using its `from(...)` method: + +```java +import com.anthropic.core.JsonValue; + +// Create primitive JSON values +JsonValue nullValue = JsonValue.from(null); + +JsonValue booleanValue = JsonValue.from(true); + +JsonValue numberValue = JsonValue.from(42); + +JsonValue stringValue = JsonValue.from("Hello World!"); + +// Create a JSON array value equivalent to `["Hello", "World"]` +JsonValue arrayValue = JsonValue.from(List.of("Hello", "World")); + +// Create a JSON object value equivalent to `{ "a": 1, "b": 2 }` +JsonValue objectValue = JsonValue.from(Map.of("a", 1, "b", 2)); + +// Create an arbitrarily nested JSON equivalent to: +// { "a": [1, 2], "b": [3, 4] } +JsonValue complexValue = JsonValue.from(Map.of("a", List.of(1, 2), "b", List.of(3, 4))); +``` + +### Forcibly omitting required parameters + +Normally a `Builder` class's `build` method will throw `IllegalStateException` if any required parameter or property is unset. To forcibly omit a required parameter or property, pass `JsonMissing`: + +```java +import com.anthropic.core.JsonMissing; +import com.anthropic.models.messages.MessageCreateParams; +import com.anthropic.models.messages.Model; + +MessageCreateParams params = MessageCreateParams.builder() + .addUserMessage("Hello, world") + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(JsonMissing.of()) + .build(); +``` + +### Response properties + +To access undocumented response properties, call the `_additionalProperties()` method: + +```java +import com.anthropic.core.JsonValue; + +Map additionalProperties = client + .messages() + .create(params) + ._additionalProperties(); + +JsonValue secretPropertyValue = additionalProperties.get("secretProperty"); + +String result = secretPropertyValue.accept(new JsonValue.Visitor<>() { + @Override + public String visitNull() { + return "It's null!"; + } + + @Override + public String visitBoolean(boolean value) { + return "It's a boolean!"; + } + + @Override + public String visitNumber(Number value) { + return "It's a number!"; + } + + // Other methods include `visitMissing`, `visitString`, `visitArray`, and `visitObject` + // The default implementation of each unimplemented method delegates to `visitDefault`, + // which throws by default, but can also be overridden +}); +``` + +To access a property's raw JSON value, call its `_` prefixed method: + +```java +import com.anthropic.core.JsonField; +import com.anthropic.models.messages.StopReason; + +JsonField stopReason = client.messages().create(params)._stopReason(); + +if (stopReason.isMissing()) { + // The property is absent from the JSON response +} else if (stopReason.isNull()) { + // The property was set to literal null +} else { + // Check if value was provided as a string + // Other methods include `asNumber()`, `asBoolean()`, etc. + Optional jsonString = stopReason.asString(); + + // Try to deserialize into a custom type + MyClass myObject = stopReason.asUnknown().orElseThrow().convert(MyClass.class); +} +``` + +### Response validation + +By default, the SDK does not throw an exception when the API returns a response that doesn't match the expected type. It throws `AnthropicInvalidDataException` only if you directly access the property. + +To check that the response is completely well-typed upfront, call `validate()`: + +```java +import com.anthropic.models.messages.Message; + +Message message = client.messages().create(params).validate(); +``` + +Or configure per-request: + +```java +import com.anthropic.models.messages.Message; + +Message message = client + .messages() + .create(params, RequestOptions.builder().responseValidation(true).build()); +``` + +Or configure the default for all method calls at the client level: + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; + +AnthropicClient client = AnthropicOkHttpClient.builder() + .fromEnv() + .responseValidation(true) + .build(); +``` + +## HTTP client customization + +### Proxy configuration + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; +import java.net.Proxy; + +AnthropicClient client = AnthropicOkHttpClient.builder() + .fromEnv() + .proxy(new Proxy(Proxy.Type.HTTP, new InetSocketAddress("https://example.com", 8080))) + .build(); +``` + +### HTTPS / SSL configuration + + + Most applications should not call these methods, and instead use the system defaults. The defaults include special optimizations that can be lost if the implementations are modified. + + +```java +import com.anthropic.client.AnthropicClient; +import com.anthropic.client.okhttp.AnthropicOkHttpClient; + +AnthropicClient client = AnthropicOkHttpClient.builder() + .fromEnv() + .sslSocketFactory(yourSSLSocketFactory) + .trustManager(yourTrustManager) + .hostnameVerifier(yourHostnameVerifier) + .build(); +``` + +### Custom HTTP client + +The SDK consists of three artifacts: + +* `anthropic-java-core` - Contains core SDK logic, does not depend on OkHttp. Exposes `AnthropicClient`, `AnthropicClientAsync`, and their implementation classes, all of which can work with any HTTP client. +* `anthropic-java-client-okhttp` - Depends on OkHttp. Exposes `AnthropicOkHttpClient` and `AnthropicOkHttpClientAsync`. +* `anthropic-java` - Depends on and exposes the APIs of both `anthropic-java-core` and `anthropic-java-client-okhttp`. Does not have its own logic. + +This structure allows replacing the SDK's default HTTP client without pulling in unnecessary dependencies. + +#### Customized OkHttpClient + + + Try the available [network options](#retries) before replacing the default client. + + +To use a customized `OkHttpClient`: + +1. Replace your `anthropic-java` dependency with `anthropic-java-core`. +2. Copy `anthropic-java-client-okhttp`'s `OkHttpClient` class into your code and customize it. +3. Construct `AnthropicClientImpl` or `AnthropicClientAsyncImpl` using your customized client. + +#### Completely custom HTTP client + +To use a completely custom HTTP client: + +1. Replace your `anthropic-java` dependency with `anthropic-java-core`. +2. Write a class that implements the `HttpClient` interface. +3. Construct `AnthropicClientImpl` or `AnthropicClientAsyncImpl` using your new client class. + +## Platform integrations + + + For detailed platform setup guides with code examples, see: + + * [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) + * [Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy) + * [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) + * [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) + * [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) + + +The Java SDK supports the following platforms through separate dependencies that provide platform-specific `Backend` implementations: + +* **Agent Platform:** `com.anthropic:anthropic-java-vertex`: Use `VertexBackend.fromEnv()` or `VertexBackend.builder()`. +* **Bedrock:** `com.anthropic:anthropic-java-bedrock`: Use `BedrockMantleBackend.fromEnv()` or `BedrockMantleBackend.builder()` for the Messages-API Bedrock endpoint, or `BedrockBackend.fromEnv()` / `BedrockBackend.builder()` (`bedrock-runtime` path). +* **Claude Platform on AWS:** `com.anthropic:anthropic-java-aws`: Use `AwsBackend.fromEnv()` (reads `ANTHROPIC_AWS_WORKSPACE_ID` and the AWS default region/credential chain) or `AwsBackend.builder()`. Available in beta. +* **Foundry:** `com.anthropic:anthropic-java-foundry`: Use `FoundryBackend.fromEnv()` or `FoundryBackend.builder()`. + +Use `BedrockMantleBackend` for new projects; `BedrockBackend` remains for existing applications using the Bedrock `InvokeModel` API. + +Each `Backend` implementation is passed to the client with `.backend()` on `AnthropicOkHttpClient.builder()`. Each cloud backend pulls in its respective cloud-platform SDK classes as transitive dependencies. + +## Advanced usage + +### Raw response access + +To access HTTP headers, status codes, and the raw response body, prefix any HTTP method call with `withRawResponse()`: + +```java +import com.anthropic.core.http.Headers; +import com.anthropic.core.http.HttpResponseFor; +import com.anthropic.models.messages.Message; +import com.anthropic.models.messages.MessageCreateParams; +import com.anthropic.models.messages.Model; + +MessageCreateParams params = MessageCreateParams.builder() + .maxTokens(1024L) + .addUserMessage("Hello, Claude") + .model(Model.CLAUDE_OPUS_4_8) + .build(); + +HttpResponseFor message = client.messages().withRawResponse().create(params); + +int statusCode = message.statusCode(); + +Headers headers = message.headers(); +``` + +You can still deserialize the response into an instance of a Java class if needed: + +```java +import com.anthropic.models.messages.Message; + +Message parsedMessage = message.parse(); +``` + +### Logging + +The SDK uses the standard OkHttp logging interceptor. + +Enable logging by setting the `ANTHROPIC_LOG` environment variable to `info`: + +```bash +export ANTHROPIC_LOG=info +``` + +Or to `debug` for more verbose logging: + +```bash +export ANTHROPIC_LOG=debug +``` + + + The SDK depends on Jackson for JSON serialization/deserialization. It is compatible with version 2.13.4 or higher, but depends on version 2.18.2 by default. + + The SDK throws an exception if it detects an incompatible Jackson version at runtime (for example, if the default version was overridden in your Maven or Gradle config). + + If the SDK threw an exception, but you're certain the version is compatible, then disable the version check using `checkJacksonVersionCompatibility` on `AnthropicOkHttpClient` or `AnthropicOkHttpClientAsync`. + + + There is no guarantee that the SDK works correctly when the Jackson version check is disabled. + + + There are also bugs in older Jackson versions that can affect the SDK. The SDK doesn't work around all Jackson bugs and expects users to upgrade Jackson for those instead. + + + + Although the SDK uses reflection, it is still usable with ProGuard and R8 because `anthropic-java-core` is published with a configuration file containing keep rules. + + ProGuard and R8 should automatically detect and use the published rules, but you can also manually copy the keep rules if necessary. + + +### Undocumented API functionality + +The SDK is typed for convenient usage of the documented API. However, it also supports working with undocumented or not yet supported parts of the API. + +#### Undocumented request parameters + +To set undocumented request parameters, use the `putAdditionalHeader`, `putAdditionalQueryParam`, or `putAdditionalBodyProperty` methods as described in [Undocumented parameters](#undocumented-parameters). + +#### Undocumented response properties + +To access undocumented response properties, use the `_additionalProperties()` method as described in [Response properties](#response-properties). + +#### New or unreleased enum values + +Enum-like classes in the SDK, such as `Model` and `AnthropicBeta`, are not closed Java `enum` types. Each one provides an `of(String)` factory method that accepts any string, so you can use values that have not been added to the SDK yet, such as a model or beta header released after your SDK version: + +```java +import com.anthropic.models.beta.AnthropicBeta; +import com.anthropic.models.messages.Model; + +Model model = Model.of("some-new-model"); +AnthropicBeta beta = AnthropicBeta.of("some-new-beta-2026-01-01"); +``` + +Builder methods that take these types often also provide a `String` overload that calls `of(...)` for you: + +```java +import com.anthropic.models.messages.MessageCreateParams; + +MessageCreateParams params = MessageCreateParams.builder() + .model("some-new-model") // same as .model(Model.of("some-new-model")) + .maxTokens(1024L) + .addUserMessage("Hello, Claude") + .build(); +``` + +Prefer the well-typed constants (for example, `Model.CLAUDE_OPUS_4_7`) so you get autocomplete and deprecation warnings. The `String` overloads and `of(...)` are primarily for setting the field to an undocumented or not yet supported value while waiting for an SDK release that includes it. + +## Beta features + +Beta features are available before general release to get early feedback and test new functionality. You can check the availability of all of Claude's capabilities and tools in the [build with Claude overview](/docs/en/build-with-claude/overview). + +You can access most beta API features through the `beta()` method on the client. To enable a particular beta feature, add the appropriate [beta header](/docs/en/api/beta-headers) with `.addBeta()` when building the message params. + +For example, to use the [Files API](/docs/en/build-with-claude/files): + +```java +import com.anthropic.models.beta.AnthropicBeta; +import com.anthropic.models.beta.messages.BetaContentBlockParam; +import com.anthropic.models.beta.messages.BetaMessage; +import com.anthropic.models.beta.messages.BetaRequestDocumentBlock; +import com.anthropic.models.beta.messages.BetaTextBlockParam; +import com.anthropic.models.beta.messages.MessageCreateParams; +// ... +void main() { + AnthropicClient client = AnthropicOkHttpClient.fromEnv(); + + BetaMessage message = client.beta().messages().create( + MessageCreateParams.builder() + .model(Model.CLAUDE_OPUS_4_8) + .maxTokens(1024L) + .addBeta(AnthropicBeta.FILES_API_2025_04_14) + .addUserMessageOfBetaContentBlockParams(List.of( + BetaContentBlockParam.ofText( + BetaTextBlockParam.builder() + .text("Please summarize this document for me.") + .build()), + BetaContentBlockParam.ofDocument( + BetaRequestDocumentBlock.builder() + .fileSource("file_abc123") + .build()))) + .build()); +} +``` + +## Frequently asked questions + + + + Java `enum` classes are not trivially forward compatible. Using them in the SDK could cause runtime exceptions if the API is updated to respond with a new enum value. + + Because these classes are open, you can also construct them with any string value through their `of(String)` factory method. See [New or unreleased enum values](#new-or-unreleased-enum-values) if you need to use a value that isn't in your SDK version yet. + + + + Using `JsonField` enables a few features: + + * Allowing usage of undocumented API functionality + * Lazily validating the API response against the expected shape + * Representing absent vs explicitly null values + + + + It is not backwards compatible to add new fields to a data class, and the SDK avoids introducing a breaking change every time a field is added to a class. + + + + Checked exceptions are widely considered a mistake in the Java programming language. In fact, they were omitted from Kotlin for this reason. + + Checked exceptions: + + * Are verbose to handle + * Encourage error handling at the wrong level of abstraction, where nothing can be done about the error + * Are tedious to propagate because of the function coloring problem + * Don't play well with lambdas (also because of the function coloring problem) + + + +## Semantic versioning + +This package generally follows [SemVer](https://semver.org/spec/v2.0.0.html) conventions, though certain backwards-incompatible changes may be released as minor versions: + +1. Changes to library internals which are technically public but not intended or documented for external use. +2. Changes that aren't expected to impact the vast majority of users in practice. + +## Additional resources + +* [GitHub repository](https://github.com/anthropics/anthropic-sdk-java) +* [Javadocs](https://javadoc.io/doc/com.anthropic/anthropic-java) +* [API reference](/docs/en/api/overview) +* [Streaming Messages](/docs/en/build-with-claude/streaming) +* [Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview) diff --git a/content/en/cli-sdks-libraries/sdks/php.md b/content/en/cli-sdks-libraries/sdks/php.md new file mode 100644 index 000000000..530d7d8cf --- /dev/null +++ b/content/en/cli-sdks-libraries/sdks/php.md @@ -0,0 +1,247 @@ +# PHP SDK + +Install and configure the Anthropic PHP SDK with value objects and builder patterns + +--- + +The Anthropic PHP library provides convenient access to the Anthropic REST API from any PHP 8.1.0+ application. + + + The PHP SDK is currently in beta. APIs might change between versions. + + + + For API feature documentation with code examples, see the [API reference](/docs/en/api/overview). This page covers PHP-specific SDK features and configuration. + + +## Installation + +The SDK uses [PSR-18](https://www.php-fig.org/psr/psr-18/) for HTTP and discovers any installed PSR-18 client automatically. [Guzzle](https://docs.guzzlephp.org/) is recommended because the SDK configures it for streaming with no additional setup: + +```bash +composer require "anthropic-ai/sdk" "guzzlehttp/guzzle:^7" +``` + +## Requirements + +PHP 8.1.0 or higher. + +## Usage + +This library uses named parameters to specify optional arguments. Parameters with a default value must be set by name. + +```php +$client = new Client(); + +$message = $client->messages->create( + maxTokens: 1024, + messages: [['role' => 'user', 'content' => 'Hello, Claude']], + model: 'claude-opus-4-8', +); + +echo $message->content[0]->text; +``` + +For authentication options including Workload Identity Federation, see [Authentication](/docs/en/manage-claude/authentication). + +## Value objects + +It is recommended to use the static `with` constructor `Base64ImageSource::with(data: "U3RhaW5sZXNzIHJvY2tz", ...)` and named parameters to initialize value objects. + +However, builders are also provided `(new Base64ImageSource)->withData("U3RhaW5sZXNzIHJvY2tz")`. + +## Streaming + +The SDK provides support for streaming responses using Server-Sent Events (SSE). + +```php +$client = new Client(); + +$stream = $client->messages->createStream( + maxTokens: 1024, + messages: [['role' => 'user', 'content' => 'Hello, Claude']], + model: 'claude-opus-4-8', +); + +foreach ($stream as $event) { + echo $event->type . PHP_EOL; +} +``` + +Streaming requires an HTTP client that returns the response body incrementally. When Guzzle is the discovered PSR-18 client, the SDK configures it for streaming automatically. With a buffering client, the `foreach` loop yields every event at once when the response completes instead of incrementally; if you observe that symptom, install Guzzle or supply a streaming-capable PSR-18 client through the `streamingTransporter` request option: + +```php +$client = new Anthropic\Client( + requestOptions: Anthropic\RequestOptions::with(streamingTransporter: $myStreamingClient), +); +``` + +## Error handling + +When the library is unable to connect to the API, or if the API returns a non-success status code (that is, a 4xx or 5xx response), a subclass of `Anthropic\Core\Exceptions\APIException` is thrown: + +```php +messages->create( + maxTokens: 1024, + messages: [['role' => 'user', 'content' => 'Hello, Claude']], + model: 'claude-opus-4-8', + ); +} catch (APIConnectionException $e) { + echo "The server could not be reached", PHP_EOL; + echo $e->getPrevious()?->getMessage(), PHP_EOL; +} catch (RateLimitException $_) { + echo "A 429 status code was received; we should back off a bit.", PHP_EOL; +} catch (APIStatusException $e) { + echo "Another non-200-range status code was received", PHP_EOL; + echo $e->getMessage(); +} +``` + +Error codes are as follows: + +| Cause | Error Type | +| ---------------- | ------------------------------ | +| HTTP 400 | `BadRequestException` | +| HTTP 401 | `AuthenticationException` | +| HTTP 403 | `PermissionDeniedException` | +| HTTP 404 | `NotFoundException` | +| HTTP 409 | `ConflictException` | +| HTTP 422 | `UnprocessableEntityException` | +| HTTP 429 | `RateLimitException` | +| HTTP >= 500 | `InternalServerException` | +| Other HTTP error | `APIStatusException` | +| Timeout | `APITimeoutException` | +| Network error | `APIConnectionException` | + +## Retries + +Certain errors are automatically retried two times by default, with a short exponential backoff. + +Connection errors (for example, because of a network connectivity problem), 408 Request Timeout, 409 Conflict, 429 Rate Limit, >=500 Internal errors, and timeouts are all retried by default. + +You can use the `maxRetries` option to configure or disable this: + +```php +use Anthropic\RequestOptions; +// ... +// Configure the default for all requests: +$client = new Client(requestOptions: RequestOptions::with(maxRetries: 0)); + +// Or, configure per-request: +$result = $client->messages->create( + maxTokens: 1024, + messages: [['role' => 'user', 'content' => 'Hello, Claude']], + model: 'claude-opus-4-8', + requestOptions: RequestOptions::with(maxRetries: 5), +); +``` + +## Pagination + +List methods in the Claude API are paginated. + +This library provides auto-paginating iterators with each list response, so you do not have to request successive pages manually: + +```php +$client = new Client(); + +$page = $client->beta->messages->batches->list(limit: 20); + +// fetch items from the current page +foreach ($page->getItems() as $item) { + echo $item->id, PHP_EOL; +} +// make additional network requests to fetch items from all pages, including and after the current page +foreach ($page->pagingEachItem() as $item) { + echo $item->id, PHP_EOL; +} +``` + +## Advanced usage + +### Undocumented properties + +You can send undocumented parameters to any endpoint, and read undocumented response properties, as follows: + + + The `extra*` parameters of the same name override the documented parameters. + + +```php +messages->create( + maxTokens: 1024, + messages: [['role' => 'user', 'content' => 'Hello, Claude']], + model: 'claude-opus-4-8', + requestOptions: RequestOptions::with( + extraQueryParams: ['my_query_parameter' => 'value'], + extraBodyParams: ['my_body_parameter' => 'value'], + extraHeaders: ['my-header' => 'value'], + ), +); +``` + +### Undocumented request parameters + +If you want to explicitly send an extra parameter, you can do so with the `extraQueryParams`, `extraBodyParams`, and `extraHeaders` options under `RequestOptions::with()` when making a request, as seen in the preceding example. + +### Undocumented endpoints + +To make requests to undocumented endpoints while retaining the benefit of authentication, retries, and other client features, you can make requests using `client->request`, as follows: + +```php +$client = new Client(); + +$response = $client->request( + method: "post", + path: '/undocumented/endpoint', + query: ['dog' => 'woof'], + headers: ['useful-header' => 'interesting-value'], + body: ['hello' => 'world'] +); +``` + +## Platform integrations + + + For detailed platform setup guides with code examples, see: + + * [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) + * [Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy) + * [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) + * [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) + * [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) + + +The PHP SDK supports the following platforms: + +* **Agent Platform:** `Anthropic\Vertex\Client`. Use `::fromEnvironment()`. +* **Bedrock:** `Anthropic\Bedrock\MantleClient`. Use `new MantleClient(awsRegion: ...)`. +* **Bedrock (legacy):** `Anthropic\Bedrock\Client`. Use `::fromEnvironment()` or `::withCredentials()`. +* **Claude Platform on AWS:** `Anthropic\Aws\Client` (requires `aws/aws-sdk-php` as a soft dependency). Use `new Anthropic\Aws\Client(workspaceId: ...)` or set `ANTHROPIC_AWS_WORKSPACE_ID`. Available in beta. +* **Foundry:** `Anthropic\Foundry\Client`. Use `::withCredentials()`. + +Use `MantleClient` for new projects; `Anthropic\Bedrock\Client` remains for existing applications using the Bedrock `InvokeModel` API. + +## Semantic versioning + +This package follows [SemVer](https://semver.org/spec/v2.0.0.html) conventions. As the library is in initial development and has a major version of `0`, APIs might change at any time. + +This package considers improvements to the (non-runtime) PHPDoc type definitions to be non-breaking changes. + +## Additional resources + +* [GitHub repository](https://github.com/anthropics/anthropic-sdk-php) +* [Packagist](https://packagist.org/packages/anthropic-ai/sdk) +* [API reference](/docs/en/api/overview) +* [Streaming Messages](/docs/en/build-with-claude/streaming) diff --git a/content/en/cli-sdks-libraries/sdks/python.md b/content/en/cli-sdks-libraries/sdks/python.md new file mode 100644 index 000000000..47a693adc --- /dev/null +++ b/content/en/cli-sdks-libraries/sdks/python.md @@ -0,0 +1,790 @@ +# Python SDK + +Install and configure the Anthropic Python SDK with sync and async client support + +--- + +The Anthropic Python SDK provides convenient access to the Anthropic REST API from Python applications. It supports both synchronous and asynchronous operations, streaming, and integrations with Amazon Bedrock, Claude Platform on AWS, Google Cloud, and Microsoft Foundry. + + + For API feature documentation with code examples, see the [API reference](/docs/en/api/overview). This page covers Python-specific SDK features and configuration. + + +## Installation + +```bash +pip install anthropic +``` + +For platform-specific integrations or improved async performance, install with extras: + +```bash +# For Amazon Bedrock support +pip install "anthropic[bedrock]" + +# For Google Cloud support +pip install "anthropic[vertex]" + +# For Claude Platform on AWS support +pip install "anthropic[aws]" + +# Microsoft Foundry support is included in the base package + +# For improved async performance with aiohttp +pip install "anthropic[aiohttp]" +``` + +## Requirements + +Python 3.9 or later is required. + +## Usage + +```python +import os +from anthropic import Anthropic + +client = Anthropic( + # This is the default and can be omitted + api_key=os.environ.get("ANTHROPIC_API_KEY"), +) + +message = client.messages.create( + max_tokens=1024, + messages=[ + { + "role": "user", + "content": "Hello, Claude", + } + ], + model="claude-opus-4-8", +) + +for block in message.content: + if block.type == "text": + print(block.text) +``` + + + Consider using [python-dotenv](https://pypi.org/project/python-dotenv/) to add `ANTHROPIC_API_KEY="my-anthropic-api-key"` to your `.env` file so that your API key isn't stored in source control. + + +For authentication options including Workload Identity Federation, see [Authentication](/docs/en/manage-claude/authentication). + +## Async usage + +```python +import os +import asyncio +from anthropic import AsyncAnthropic + +client = AsyncAnthropic( + api_key=os.environ.get("ANTHROPIC_API_KEY"), +) + + +async def main() -> None: + message = await client.messages.create( + max_tokens=1024, + messages=[ + { + "role": "user", + "content": "Hello, Claude", + } + ], + model="claude-opus-4-8", + ) + print(message.content) + + +asyncio.run(main()) +``` + +### Using aiohttp for better concurrency + +For improved async performance, you can use the `aiohttp` HTTP backend instead of the default `httpx`: + +```python +import os +import asyncio +from anthropic import AsyncAnthropic, DefaultAioHttpClient + + +async def main() -> None: + async with AsyncAnthropic( + api_key=os.environ.get("ANTHROPIC_API_KEY"), + http_client=DefaultAioHttpClient(), + ) as client: + message = await client.messages.create( + max_tokens=1024, + messages=[ + { + "role": "user", + "content": "Hello, Claude", + } + ], + model="claude-opus-4-8", + ) + print(message.content) + + +asyncio.run(main()) +``` + +## Streaming responses + +The SDK provides support for streaming responses using Server-Sent Events (SSE). + +```python +client = Anthropic() + +stream = client.messages.create( + max_tokens=1024, + messages=[ + { + "role": "user", + "content": "Hello, Claude", + } + ], + model="claude-opus-4-8", + stream=True, +) +for event in stream: + print(event.type) +``` + +The async client uses the exact same interface: + +```python +client = AsyncAnthropic() + +stream = await client.messages.create( + max_tokens=1024, + messages=[ + { + "role": "user", + "content": "Hello, Claude", + } + ], + model="claude-opus-4-8", + stream=True, +) +async for event in stream: + print(event.type) +``` + +### Streaming helpers + +The SDK also provides streaming helpers that use context managers and provide access to the accumulated text and the final message: + +```python +async def main() -> None: + async with client.messages.stream( + max_tokens=1024, + messages=[ + { + "role": "user", + "content": "Say hello there!", + } + ], + model="claude-opus-4-8", + ) as stream: + async for text in stream.text_stream: + print(text, end="", flush=True) + print() + + message = await stream.get_final_message() + print(message.to_json()) + + +asyncio.run(main()) +``` + +Streaming with `client.messages.stream(...)` exposes various helpers including accumulation and SDK-specific events. + +Alternatively, you can use `client.messages.create(..., stream=True)` which only returns an iterable of the events in the stream and uses less memory (it doesn't build up a final message object for you). + +## Token counting + +You can see the exact usage for a given request through the `usage` response property: + +```python +message = client.messages.create(...) +print(message.usage) +# Usage(input_tokens=25, output_tokens=13) +``` + +You can also count tokens before making a request: + +```python +count = client.messages.count_tokens( + model="claude-opus-4-8", messages=[{"role": "user", "content": "Hello, world"}] +) +print(count.input_tokens) # 10 +``` + +## Tool use + +This SDK provides support for tool use, also known as function calling. For more details, see [Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview). + +### Tool helpers + +The SDK provides helpers for defining and running tools as pure Python functions. The `@beta_tool` decorator generates the tool schema from the function signature and docstring: + +```python +import json +from anthropic import Anthropic, beta_tool + +client = Anthropic() + + +@beta_tool +def get_weather(location: str) -> str: + """Get the weather for a given location. + + Args: + location: The city and state, for example, San Francisco, CA + Returns: + A JSON-encoded string with the location, temperature, and weather condition. + """ + return json.dumps( + { + "location": location, + "temperature": "68°F", + "condition": "Sunny", + } + ) + + +# Use the tool_runner to automatically handle tool calls +runner = client.beta.messages.tool_runner( + max_tokens=1024, + model="claude-opus-4-8", + tools=[get_weather], + messages=[ + {"role": "user", "content": "What is the weather in SF?"}, + ], +) +for message in runner: + print(message) +``` + +On every iteration, an API request is made. If the response includes a call to one of the given tools, the tool is automatically called, and the result is returned directly to the model in the next iteration. + +## Message batches + +This SDK provides support for the [Message Batches API](/docs/en/build-with-claude/batch-processing) under `client.messages.batches`. + +### Creating a batch + +Message Batches takes an array of requests, where each object has a `custom_id` identifier and the same request `params` as the standard Messages API: + +```python +client.messages.batches.create( + requests=[ + { + "custom_id": "my-first-request", + "params": { + "model": "claude-opus-4-8", + "max_tokens": 1024, + "messages": [{"role": "user", "content": "Hello, world"}], + }, + }, + { + "custom_id": "my-second-request", + "params": { + "model": "claude-opus-4-8", + "max_tokens": 1024, + "messages": [{"role": "user", "content": "Hi again, friend"}], + }, + }, + ] +) +``` + +### Getting results from a batch + +Once a Message Batch has been processed, indicated by `.processing_status == 'ended'`, you can access the results with `.batches.results()`: + +```python +client = anthropic.Anthropic() +batch_id = "batch_abc123" +result_stream = client.messages.batches.results(batch_id) +for entry in result_stream: + if entry.result.type == "succeeded": + print(entry.result.message.content) +``` + +## File uploads + +Request parameters that correspond to file uploads can be passed in many different forms: + +* A `PathLike` object (for example, `pathlib.Path`) +* A tuple of `(filename, content, content_type)` +* A `BinaryIO` file-like object + +```python +from pathlib import Path +from anthropic import Anthropic + +client = Anthropic() + +# Upload using a file path +client.beta.files.upload( + file=Path("/path/to/file"), +) + +# Upload using bytes +client.beta.files.upload( + file=("file.txt", b"my bytes", "text/plain"), +) +``` + +The async client uses the exact same interface. If you pass a `PathLike` instance, the file contents are read asynchronously automatically. + +## Handling errors + +When the library is unable to connect to the API, or if the API returns a non-success status code (that is, 4xx or 5xx response), a subclass of `APIError` is raised: + +```python +import anthropic +# ... +try: + message = client.messages.create( + max_tokens=1024, + messages=[ + { + "role": "user", + "content": "Hello, Claude", + } + ], + model="claude-opus-4-8", + ) +except anthropic.APIConnectionError as e: + print("The server could not be reached") + print(e.__cause__) # an underlying Exception, likely raised within httpx +except anthropic.RateLimitError as e: + print("A 429 status code was received; we should back off a bit.") +except anthropic.APIStatusError as e: + print("Another non-200-range status code was received") + print(e.status_code) + print(e.response) +``` + +Error codes are as follows: + +| Status code | Error type | +| ----------- | -------------------------- | +| 400 | `BadRequestError` | +| 401 | `AuthenticationError` | +| 403 | `PermissionDeniedError` | +| 404 | `NotFoundError` | +| 409 | `ConflictError` | +| 422 | `UnprocessableEntityError` | +| 429 | `RateLimitError` | +| >=500 | `InternalServerError` | +| N/A | `APIConnectionError` | + +## Request IDs + +> For more information on debugging requests, see [Request ID](/docs/en/api/errors#request-id). + +All object responses in the SDK provide a `_request_id` property which is added from the `request-id` response header so that you can quickly log failing requests and report them back to Anthropic. + +```python +message = client.messages.create( + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude"}], + model="claude-opus-4-8", +) +print(message._request_id) # e.g., req_018EeWyXxfu5pfWkrYcMdjWG +``` + + + Unlike other properties that use an `_` prefix, the `_request_id` property is public. Unless documented otherwise, all other `_` prefix properties, methods, and modules are private. + + +## Retries + +Certain errors are automatically retried 2 times by default, with a short exponential backoff. Connection errors (for example, because of a network connectivity problem), 408 Request Timeout, 409 Conflict, 429 Rate Limit, and >=500 Internal errors are all retried by default. + +You can use the `max_retries` option to configure or disable this: + +```python +# Configure the default for all requests: +client = Anthropic( + max_retries=0, # default is 2 +) + +# Or, configure per-request: +client.with_options(max_retries=5).messages.create( + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude"}], + model="claude-opus-4-8", +) +``` + +## Timeouts + +By default requests time out after 10 minutes. You can configure this with a `timeout` option, which accepts a float or an `httpx.Timeout` object: + +```python +import httpx +from anthropic import Anthropic + +# Configure the default for all requests: +client = Anthropic( + timeout=20.0, # 20 seconds (default is 10 minutes) +) + +# More granular control: +client = Anthropic( + timeout=httpx.Timeout(60.0, read=5.0, write=10.0, connect=2.0), +) + +# Override per-request: +client.with_options(timeout=5.0).messages.create( + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude"}], + model="claude-opus-4-8", +) +``` + +On timeout, the SDK throws an `APITimeoutError`. + +Note that requests that time out are [retried twice by default](#retries). + +## Long requests + + + Consider using the streaming [Messages API](#streaming-responses) for longer running requests. + + +Avoid setting a large `max_tokens` value without using streaming. Some networks may drop idle connections after a certain period of time, which can cause the request to fail or [timeout](#timeouts) without receiving a response from Anthropic. + +The SDK will throw a `ValueError` if a non-streaming request is expected to take longer than approximately 10 minutes. Passing `stream=True` or overriding the `timeout` option at the client or request level disables this error. + +An expected request latency longer than the [timeout](#timeouts) for a non-streaming request will result in the client terminating the connection and retrying without receiving a response. + +The SDK sets a [TCP socket keep-alive](https://tldp.org/HOWTO/TCP-Keepalive-HOWTO/overview.html) option to reduce the impact of idle connection timeouts on some networks. This can be overridden by passing a custom `http_client` option to the client. + +## Auto-pagination + +List methods in the Claude API are paginated. You can use the `for` syntax to iterate through items across all pages: + +```python +client = Anthropic() + +all_batches = [] +# Automatically fetches more pages as needed. +for batch in client.messages.batches.list(limit=20): + all_batches.append(batch) +print(all_batches) +``` + +For async iteration: + +```python +async def main() -> None: + all_batches = [] + async for batch in client.messages.batches.list(limit=20): + all_batches.append(batch) + print(all_batches) + + +asyncio.run(main()) +``` + +Alternatively, you can use the `.has_next_page()`, `.next_page_info()`, or `.get_next_page()` methods for more granular control working with pages: + +```python +first_page = await client.messages.batches.list(limit=20) + +if first_page.has_next_page(): + print(f"will fetch next page using these details: {first_page.next_page_info()}") + next_page = await first_page.get_next_page() + print(f"number of items we just fetched: {len(next_page.data)}") + +# Remove `await` for non-async usage. +``` + +Or work directly with the returned data: + +```python +first_page = await client.messages.batches.list(limit=20) + +print(f"next page cursor: {first_page.last_id}") +for batch in first_page.data: + print(batch.id) + +# Remove `await` for non-async usage. +``` + +## Default headers + +The SDK automatically sends the `anthropic-version` header set to `2023-06-01`. + +If you need to, you can override it by setting default headers on the client object or per-request. + + + Overriding default headers may result in incorrect types and other unexpected or undefined behavior in the SDK. + + +```python +# Set default headers for all requests on the client +client = Anthropic( + default_headers={"anthropic-version": "My-Custom-Value"}, +) + +# Or override per-request +client.messages.with_raw_response.create( + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude"}], + model="claude-opus-4-8", + extra_headers={"anthropic-version": "My-Custom-Value"}, +) +``` + +## Type system + +### Request parameters + +Nested request parameters are [TypedDicts](https://docs.python.org/3/library/typing.html#typing.TypedDict). Responses are [Pydantic models](https://docs.pydantic.dev) which also have helper methods for things like serializing back into JSON ([`v1`](https://docs.pydantic.dev/1.10/usage/models/), [`v2`](https://docs.pydantic.dev/latest/concepts/serialization/)). + +Typed requests and responses provide autocomplete and documentation within your editor. If you'd like to see type errors in VS Code to help catch bugs earlier, set `python.analysis.typeCheckingMode` to `basic`. + +### Response models + +To convert a Pydantic model to a dictionary, use the helper methods: + +```python +message = client.messages.create(...) + +# Convert to JSON string +json_str = message.to_json() + +# Convert to dictionary +data = message.to_dict() +``` + +### Handling null vs missing fields + +In responses, you can distinguish between fields that are explicitly `null` versus fields that were not returned (missing): + +```python +response = client.messages.create( + model="claude-opus-4-8", + max_tokens=1024, + messages=[{"role": "user", "content": "Hello"}], +) +if response.my_field is None: + if "my_field" not in response.model_fields_set: + print("field was not in the response") + else: + print("field was null") +``` + +## Advanced usage + +### Accessing raw response data (for example, headers) + +The "raw" `Response` returned by `httpx` can be accessed through the `.with_raw_response` property on the client. This is useful for accessing response headers or other metadata: + +```python +client = Anthropic() + +response = client.messages.with_raw_response.create( + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude"}], + model="claude-opus-4-8", +) + +print(response.headers.get("request-id")) +message = ( + response.parse() +) # get the object that `messages.create()` would have returned +print(message.content) +``` + +These methods return an `APIResponse` object. + +### Streaming response body + +The `.with_raw_response` approach eagerly reads the full response body when you make the request. To stream the response body instead, use `.with_streaming_response`, which requires a context manager and only reads the response body once you call `.read()`, `.text()`, `.json()`, `.iter_bytes()`, `.iter_text()`, `.iter_lines()`, or `.parse()`. In the async client, these are async methods. + +```python +with client.messages.with_streaming_response.create( + max_tokens=1024, + messages=[{"role": "user", "content": "Hello, Claude"}], + model="claude-opus-4-8", +) as response: + print(response.headers.get("request-id")) + + for line in response.iter_lines(): + print(line) +``` + +The context manager is required so that the response will reliably be closed. + +### Logging + +The SDK uses the standard library `logging` module. + +You can enable logging by setting the environment variable `ANTHROPIC_LOG` to `debug` or `info`: + +```bash +export ANTHROPIC_LOG=debug +``` + +### Making custom/undocumented requests + +This library is typed for convenient access to the documented API. If you need to access undocumented endpoints, params, or response properties, the library can still be used. + +#### Undocumented endpoints + +To make requests to undocumented endpoints, you can use `client.get`, `client.post`, and other HTTP verbs. Options on the client, such as retries, are respected when making these requests. + +```python +import httpx + +response = client.post( + "/foo", + cast_to=httpx.Response, + body={"my_param": True}, +) + +print(response.json()) +``` + +#### Undocumented request params + +If you want to explicitly send an extra parameter, you can do so with the `extra_query`, `extra_body`, and `extra_headers` request options. + + + The `extra_` parameters override documented parameters of the same name. For security reasons, ensure these methods are only used with trusted input data. + + +#### Undocumented response properties + +To access undocumented response properties, you can access the extra fields like `response.unknown_prop`. You can also get all extra fields on the Pydantic model as a dict with `response.model_extra`. + +### Configuring the HTTP client + +You can directly override the [httpx client](https://www.python-httpx.org/api/#client) to customize it for your use case, including support for proxies and transports: + +```python +import httpx +from anthropic import Anthropic, DefaultHttpxClient + +client = Anthropic( + # Or use the `ANTHROPIC_BASE_URL` env var + base_url="http://my.test.server.example.com:8083", + http_client=DefaultHttpxClient( + proxy="http://my.test.proxy.example.com", + transport=httpx.HTTPTransport(local_address="0.0.0.0"), + ), +) +``` + +You can also customize the client on a per-request basis by using `with_options()`: + +```python +client.with_options(http_client=DefaultHttpxClient(...)) +``` + + + Use `DefaultHttpxClient` and `DefaultAsyncHttpxClient` instead of raw `httpx.Client` and `httpx.AsyncClient` to ensure the SDK's default configuration (such as timeouts and connection limits) is preserved. + + +### Managing HTTP resources + +By default the library closes underlying HTTP connections whenever the client is [garbage collected](https://docs.python.org/3/reference/datamodel.html#object.__del__). You can manually close the client using the `.close()` method if desired, or with a context manager that closes when exiting. + +```python +with Anthropic() as client: + message = client.messages.create(...) + +# HTTP client is automatically closed +``` + +## Beta features + +Beta features are available before general release to get early feedback and test new functionality. You can check the availability of all of Claude's capabilities and tools in the [build with Claude overview](/docs/en/build-with-claude/overview). + +You can access most beta API features through the `beta` property of the client. To enable a particular beta feature, you need to add the appropriate [beta header](/docs/en/api/beta-headers) to the `betas` field when creating a message. + +For example, to use the [Files API](/docs/en/build-with-claude/files): + +```python +client = Anthropic() + +response = client.beta.messages.create( + model="claude-opus-4-8", + max_tokens=1024, + messages=[ + { + "role": "user", + "content": [ + {"type": "text", "text": "Please summarize this document for me."}, + { + "type": "document", + "source": { + "type": "file", + "file_id": "file_abc123", + }, + }, + ], + }, + ], + betas=["files-api-2025-04-14"], +) +``` + +## Platform integrations + + + For detailed platform setup guides with code examples, see: + + * [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) + * [Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy) + * [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) + * [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) + * [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) + + +All five client classes are included in the base `anthropic` package: + +| Provider | Client | Extra dependencies | +| -------------------------------- | ---------------------------------------------- | ---------------------------------- | +| Agent Platform | `from anthropic import AnthropicVertex` | `pip install "anthropic[vertex]"` | +| Bedrock | `from anthropic import AnthropicBedrockMantle` | `pip install "anthropic[bedrock]"` | +| Bedrock (`bedrock-runtime` path) | `from anthropic import AnthropicBedrock` | `pip install "anthropic[bedrock]"` | +| Claude Platform on AWS | `from anthropic import AnthropicAWS` | `pip install "anthropic[aws]"` | +| Foundry | `from anthropic import AnthropicFoundry` | None | + +The `AnthropicAWS` client is in beta. Pass `workspace_id` to the constructor or set the `ANTHROPIC_AWS_WORKSPACE_ID` environment variable. + +Use `AnthropicBedrockMantle` for new projects; `AnthropicBedrock` remains for existing applications using the Bedrock `InvokeModel` API. + +## Semantic versioning + +This package generally follows [SemVer](https://semver.org/spec/v2.0.0.html) conventions, though certain backward-incompatible changes may be released as minor versions: + +1. Changes that only affect static types, without breaking runtime behavior. +2. Changes to library internals which are technically public but not intended or documented for external use. +3. Changes that aren't expected to impact the vast majority of users in practice. + +### Determining the installed version + +If you've upgraded to the latest version but aren't seeing new features you were expecting, your Python environment is likely still using an older version. You can determine the version being used at runtime with: + +```python +print(anthropic.__version__) +``` + +## Additional resources + +* [GitHub repository](https://github.com/anthropics/anthropic-sdk-python) +* [API reference](/docs/en/api/overview) +* [Streaming Messages](/docs/en/build-with-claude/streaming) +* [Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview) diff --git a/content/en/cli-sdks-libraries/sdks/ruby.md b/content/en/cli-sdks-libraries/sdks/ruby.md new file mode 100644 index 000000000..d043677a0 --- /dev/null +++ b/content/en/cli-sdks-libraries/sdks/ruby.md @@ -0,0 +1,416 @@ +# Ruby SDK + +Install and configure the Anthropic Ruby SDK with Sorbet types, streaming helpers, and connection pooling + +--- + +The Anthropic Ruby library provides convenient access to the Anthropic REST API from any Ruby 3.2.0+ application. It ships with comprehensive types and docstrings in Yard, RBS, and RBI. The standard library's `net/http` is used as the HTTP transport, with connection pooling through the `connection_pool` gem. + + + For API feature documentation with code examples, see the [API reference](/docs/en/api/overview). This page covers Ruby-specific SDK features and configuration. + + +## Installation + +Add the gem to your application's `Gemfile` with Bundler: + +```bash +bundle add anthropic +``` + +## Requirements + +Ruby 3.2.0 or higher. + +## Usage + +```ruby +anthropic = Anthropic::Client.new( + api_key: ENV["ANTHROPIC_API_KEY"] # This is the default and can be omitted +) + +message = anthropic.messages.create( + max_tokens: 1024, + messages: [{role: "user", content: "Hello, Claude"}], + model: :"claude-opus-4-8" +) + +message.content.each do |block| + puts block.text if block.type == :text +end +``` + +For authentication options including Workload Identity Federation, see [Authentication](/docs/en/manage-claude/authentication). + +## Streaming + +The SDK provides support for streaming responses using Server-Sent Events (SSE). + +```ruby +anthropic = Anthropic::Client.new +stream = anthropic.messages.stream( + max_tokens: 1024, + messages: [{role: "user", content: "Hello, Claude"}], + model: :"claude-opus-4-8" +) + +stream.each do |message| + puts(message.type) +end +``` + +### Streaming helpers + +This library provides several conveniences for streaming messages, for example: + +```ruby +anthropic = Anthropic::Client.new +stream = anthropic.messages.stream( + max_tokens: 1024, + messages: [{role: :user, content: "Say hello there!"}], + model: :"claude-opus-4-8" +) + +stream.text.each do |text| + print(text) +end +``` + +Streaming with `anthropic.messages.stream(...)` exposes various helpers including accumulation and SDK-specific events. + +## Input schema and tool calling + +The SDK provides helper mechanisms to define structured data classes for tools and let Claude automatically execute them. For detailed documentation on tool use patterns including the tool runner, see [Tool Runner (SDK)](/docs/en/agents-and-tools/tool-use/tool-runner). + +```ruby +anthropic = Anthropic::Client.new +class CalculatorInput < Anthropic::BaseModel + required :lhs, Float + required :rhs, Float + required :operator, Anthropic::InputSchema::EnumOf[:+, :-, :*, :/] +end + +class Calculator < Anthropic::BaseTool + input_schema CalculatorInput + + def call(expr) + expr.lhs.public_send(expr.operator, expr.rhs) + end +end + +# Automatically handles tool execution loop +anthropic.beta.messages.tool_runner( + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [{role: "user", content: "What's 15 * 7?"}], + tools: [Calculator.new] +).each_message { |message| puts message.content } +``` + +## Structured outputs + +For complete structured outputs documentation including Ruby examples, see [Structured outputs](/docs/en/build-with-claude/structured-outputs). + +## Handling errors + +When the library is unable to connect to the API, or if the API returns a non-success status code (that is, 4xx or 5xx response), a subclass of `Anthropic::Errors::APIError` is raised: + +```ruby +anthropic = Anthropic::Client.new +begin + message = anthropic.messages.create( + max_tokens: 1024, + messages: [{role: "user", content: "Hello, Claude"}], + model: :"claude-opus-4-8" + ) +rescue Anthropic::Errors::APIConnectionError => e + puts("The server could not be reached") + puts(e.cause) # an underlying Exception, likely raised within `net/http` +rescue Anthropic::Errors::RateLimitError => e + puts("A 429 status code was received; we should back off a bit.") +rescue Anthropic::Errors::APIStatusError => e + puts("Another non-200-range status code was received") + puts(e.status) +end +``` + +Error codes are as follows: + +| Cause | Error Type | +| ---------------- | -------------------------- | +| HTTP 400 | `BadRequestError` | +| HTTP 401 | `AuthenticationError` | +| HTTP 403 | `PermissionDeniedError` | +| HTTP 404 | `NotFoundError` | +| HTTP 409 | `ConflictError` | +| HTTP 422 | `UnprocessableEntityError` | +| HTTP 429 | `RateLimitError` | +| HTTP >= 500 | `InternalServerError` | +| Other HTTP error | `APIStatusError` | +| Timeout | `APITimeoutError` | +| Network error | `APIConnectionError` | + +## Retries + +Certain errors will be automatically retried 2 times by default, with a short exponential backoff. + +Connection errors (for example, because of a network connectivity problem), 408 Request Timeout, 409 Conflict, 429 Rate Limit, >=500 Internal errors, and timeouts are all retried by default. + +You can use the `max_retries` option to configure or disable this: + +```ruby +# Configure the default for all requests: +anthropic = Anthropic::Client.new( + max_retries: 0 # default is 2 +) + +# Or, configure per-request: +anthropic.messages.create( + max_tokens: 1024, + messages: [{role: "user", content: "Hello, Claude"}], + model: :"claude-opus-4-8", + request_options: {max_retries: 5} +) +``` + +## Timeouts + +By default, requests time out after 10 minutes. You can use the `timeout` option to configure this: + +```ruby +# Configure the default for all requests: +anthropic = Anthropic::Client.new( + timeout: 20 # 20 seconds (default is 10 minutes) +) + +# Or, configure per-request: +anthropic.messages.create( + max_tokens: 1024, + messages: [{role: "user", content: "Hello, Claude"}], + model: :"claude-opus-4-8", + request_options: {timeout: 5} +) +``` + +On timeout, `Anthropic::Errors::APITimeoutError` is raised. + +Note that requests that time out are retried by default. + +## Pagination + +List methods in the Claude API are paginated. + +This library provides auto-paginating iterators with each list response, so you do not have to request successive pages manually: + +```ruby +anthropic = Anthropic::Client.new +page = anthropic.messages.batches.list(limit: 20) + +# Fetch single item from page. +batch = page.data[0] +puts(batch.id) + +# Automatically fetches more pages as needed. +page.auto_paging_each do |batch| + puts(batch.id) +end +``` + +Alternatively, you can use the `#next_page?` and `#next_page` methods for more granular control working with pages. + +```ruby +anthropic = Anthropic::Client.new +page = anthropic.messages.batches.list(limit: 20) +loop do + page.data&.each { |batch| puts(batch.id) } + break unless page.next_page? + page = page.next_page +end +``` + +## File uploads + +Request parameters that correspond to file uploads can be passed as raw contents, a [`Pathname`](https://rubyapi.org/3.2/o/pathname) instance, [`StringIO`](https://rubyapi.org/3.2/o/stringio), or more. + +```ruby +anthropic = Anthropic::Client.new +require "pathname" + +# Use `Pathname` to send the filename and/or avoid paging a large file into memory: +file_metadata = anthropic.beta.files.upload(file: Pathname("/path/to/file")) + +# Alternatively, pass file contents or a `StringIO` directly: +file_metadata = anthropic.beta.files.upload(file: File.read("/path/to/file")) + +# Or, to control the filename and/or content type: +file = Anthropic::FilePart.new(File.read("/path/to/file"), filename: "/path/to/file", content_type: "...") +file_metadata = anthropic.beta.files.upload(file: file) + +puts(file_metadata.id) +``` + +Note that you can also pass a raw `IO` descriptor, but this disables retries, as the library can't be sure if the descriptor is a file or pipe (which cannot be rewound). + +## Sorbet + +This library provides comprehensive [RBI](https://sorbet.org/docs/rbi) definitions, and has no dependency on sorbet-runtime. + +You can provide typesafe request parameters like so: + +```ruby +anthropic = Anthropic::Client.new +anthropic.messages.create( + max_tokens: 1024, + messages: [Anthropic::MessageParam.new(role: "user", content: "Hello, Claude")], + model: :"claude-opus-4-8" +) +``` + +Or, equivalently: + +```ruby +anthropic = Anthropic::Client.new +# Hashes work, but are not typesafe: +anthropic.messages.create( + max_tokens: 1024, + messages: [{role: "user", content: "Hello, Claude"}], + model: :"claude-opus-4-8" +) + +# You can also splat a full Params class: +params = Anthropic::MessageCreateParams.new( + max_tokens: 1024, + messages: [Anthropic::MessageParam.new(role: "user", content: "Hello, Claude")], + model: :"claude-opus-4-8" +) +anthropic.messages.create(**params) +``` + +### Enums + +Since this library does not depend on `sorbet-runtime`, it cannot provide [`T::Enum`](https://sorbet.org/docs/tenum) instances. Instead, the SDK provides "tagged symbols", which is always a primitive at runtime: + +```ruby +# :auto +puts(Anthropic::MessageCreateParams::ServiceTier::AUTO) + +# Revealed type: `T.all(Anthropic::MessageCreateParams::ServiceTier, Symbol)` +T.reveal_type(Anthropic::MessageCreateParams::ServiceTier::AUTO) +``` + +Enum parameters have a "relaxed" type, so you can either pass in enum constants or their literal value: + +```ruby +# Using the enum constants preserves the tagged type information: +anthropic.messages.create( + service_tier: Anthropic::MessageCreateParams::ServiceTier::AUTO, + # ... +) + +# Literal values are also permissible: +anthropic.messages.create( + service_tier: :auto, + # ... +) +``` + +## BaseModel + +All parameter and response objects inherit from `Anthropic::Internal::Type::BaseModel`, which provides several conveniences, including: + +1. All fields, including unknown ones, are accessible with `obj[:prop]` syntax, and can be destructured with `obj => {prop: prop}` or pattern-matching syntax. + +2. Structural equivalence for equality; if two API calls return the same values, comparing the responses with == will return true. + +3. Both instances and the classes themselves can be pretty-printed. + +4. Helpers such as `#to_h`, `#deep_to_h`, `#to_json`, and `#to_yaml`. + +## Concurrency and connection pooling + +The `Anthropic::Client` instances are threadsafe, but are only fork-safe when there are no in-flight HTTP requests. + +Each instance of `Anthropic::Client` has its own HTTP connection pool with a default size of 99. As such, the recommendation is to create the client once per application in most settings. + +When all available connections from the pool are checked out, requests wait for a new connection to become available, with queue time counting toward the request timeout. + +Unless otherwise specified, other classes in the SDK do not have locks protecting their underlying data structure. + +## Making custom or undocumented requests + +### Undocumented properties + +You can send undocumented parameters to any endpoint, and read undocumented response properties, like so: + + + The `extra_` parameters of the same name override the documented parameters. For security reasons, ensure these methods are only used with trusted input data. + + +```ruby +anthropic = Anthropic::Client.new +value = "example" +message = + anthropic.messages.create( + max_tokens: 1024, + messages: [{role: "user", content: "Hello, Claude"}], + model: :"claude-opus-4-8", + request_options: { + extra_query: {my_query_parameter: value}, + extra_body: {my_body_parameter: value}, + extra_headers: {"my-header": value} + } + ) + +puts(message[:my_undocumented_property]) +``` + +### Undocumented request params + +If you want to explicitly send an extra param, you can do so with the `extra_query`, `extra_body`, and `extra_headers` under the `request_options:` parameter when making a request, as seen in the examples above. + +### Undocumented endpoints + +To make requests to undocumented endpoints while retaining the benefit of auth, retries, and so on, you can make requests using `anthropic.request`, like so: + +```ruby +response = anthropic.request( + method: :post, + path: '/undocumented/endpoint', + query: {"dog": "woof"}, + headers: {"useful-header": "interesting-value"}, + body: {"hello": "world"} +) +``` + +## Platform integrations + + + For detailed platform setup guides with code examples, see: + + * [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) + * [Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy) + * [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) + * [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) + + +The Ruby SDK supports the following platforms: + +* **Agent Platform:** `Anthropic::VertexClient`. Requires the `googleauth` gem. +* **Bedrock:** `Anthropic::BedrockMantleClient`, or `Anthropic::BedrockClient` for the `bedrock-runtime` path. `Anthropic::BedrockMantleClient` requires the `aws-sdk-core` gem; `Anthropic::BedrockClient` requires the `aws-sdk-bedrockruntime` gem. +* **Claude Platform on AWS:** Part of the main `anthropic` gem (requires the `aws-sdk-core` gem). Provides `Anthropic::AWSClient`. Pass `workspace_id:` to the constructor or set the `ANTHROPIC_AWS_WORKSPACE_ID` environment variable (see [Workspaces](/docs/en/build-with-claude/claude-platform-on-aws#workspaces)). Available in beta. +* **Foundry:** Not currently supported in the Ruby SDK. See [Claude in Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) for supported SDKs. + +Use `Anthropic::BedrockMantleClient` for new projects; `Anthropic::BedrockClient` remains for existing applications using the Bedrock `InvokeModel` API. + +## Semantic versioning + +This package follows [SemVer](https://semver.org/spec/v2.0.0.html) conventions. As the library is in initial development and has a major version of `0`, APIs may change at any time. + +This package considers improvements to the (non-runtime) `*.rbi` and `*.rbs` type definitions to be non-breaking changes. + +## Additional resources + +* [GitHub repository](https://github.com/anthropics/anthropic-sdk-ruby) +* [YARD documentation](https://gemdocs.org/gems/anthropic) +* [API reference](/docs/en/api/overview) +* [Streaming Messages](/docs/en/build-with-claude/streaming) diff --git a/content/en/cli-sdks-libraries/sdks/typescript.md b/content/en/cli-sdks-libraries/sdks/typescript.md new file mode 100644 index 000000000..46733f1b0 --- /dev/null +++ b/content/en/cli-sdks-libraries/sdks/typescript.md @@ -0,0 +1,812 @@ +# TypeScript SDK + +Install and configure the Anthropic TypeScript SDK for Node.js, Deno, Bun, and browser environments + +--- + +This library provides convenient access to the Anthropic REST API from TypeScript or JavaScript. + + + For API feature documentation with code examples, see the [API reference](/docs/en/api/overview). This page covers TypeScript-specific SDK features and configuration. + + +## Installation + +```bash +npm install @anthropic-ai/sdk +``` + +## Requirements + +TypeScript >= 4.9 is supported. + +The following runtimes are supported: + +* Node.js 20 LTS or later ([non-EOL](https://endoflife.date/nodejs)) versions. +* Deno v1.28.0 or higher. +* Bun 1.0 or later. +* Cloudflare Workers. +* Vercel Edge Runtime. +* Jest 28 or greater with the `"node"` environment (`"jsdom"` is not supported at this time). +* Nitro v2.6 or greater. +* Web browsers: disabled by default to avoid exposing your secret API credentials (see [API key best practices](https://support.claude.com/en/articles/9767949-api-key-best-practices-keeping-your-keys-safe-and-secure)). Enable browser support by explicitly setting `dangerouslyAllowBrowser` to `true`. + +Note that React Native is not supported at this time. + +If you are interested in other runtime environments, open or upvote an issue on the [GitHub repository](https://github.com/anthropics/anthropic-sdk-typescript). + +## Usage + +```typescript +const client = new Anthropic({ + apiKey: process.env["ANTHROPIC_API_KEY"] // This is the default and can be omitted +}); + +const message = await client.messages.create({ + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" +}); + +for (const block of message.content) { + if (block.type === "text") { + console.log(block.text); + } +} +``` + +For authentication options including Workload Identity Federation, see [Authentication](/docs/en/manage-claude/authentication). + +## Request and response types + +This library includes TypeScript definitions for all request parameters and response fields. You may import and use them like so: + +```typescript +const client = new Anthropic({ + apiKey: process.env["ANTHROPIC_API_KEY"] // This is the default and can be omitted +}); + +const params: Anthropic.MessageCreateParams = { + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" +}; +const message: Anthropic.Message = await client.messages.create(params); +``` + +Documentation for each method, request parameter, and response field is available in docstrings and appears on hover in most modern editors. + +## Counting tokens + +You can see the exact usage for a given request through the `usage` response property, for example: + +```typescript +const message = await client.messages.create(/* ... */); +console.log(message.usage); +// { input_tokens: 25, output_tokens: 13 } +``` + +## Streaming responses + +The SDK provides support for streaming responses using Server Sent Events (SSE). + +```typescript +const client = new Anthropic(); + +const stream = await client.messages.create({ + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8", + stream: true +}); +for await (const messageStreamEvent of stream) { + console.log(messageStreamEvent.type); +} +``` + +If you need to cancel a stream, you can `break` from the loop or call `stream.controller.abort()`. + +## Streaming helpers + +This library provides several conveniences for streaming messages, for example: + +```typescript +const anthropic = new Anthropic(); + +const stream = anthropic.messages + .stream({ + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [ + { + role: "user", + content: "Say hello there!" + } + ] + }) + .on("text", (text) => { + console.log(text); + }); + +const message = await stream.finalMessage(); +console.log(message); +``` + +Streaming with `client.messages.stream(...)` exposes various helpers for your convenience including event handlers and accumulation. + +Alternatively, you can use `client.messages.create({ ..., stream: true })` which only returns an async iterable of the events in the stream and thus uses less memory (it does not build up a final message object for you). + +## Tool helpers + +This SDK provides helpers for making it easy to create and run tools in the Messages API. You can use Zod schemas or JSON Schemas to describe the input to a tool. You can then run those tools using the `client.beta.messages.toolRunner()` method. This method handles passing the inputs generated by the chosen model into the right tool and passing the result back to the model. + +For more details on tool use, see [Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview). + +```typescript +import { betaZodTool } from "@anthropic-ai/sdk/helpers/beta/zod"; +import { z } from "zod"; + +const anthropic = new Anthropic(); + +const weatherTool = betaZodTool({ + name: "get_weather", + inputSchema: z.object({ + location: z.string() + }), + description: "Get the current weather in a given location", + run: (input) => { + return `The weather in ${input.location} is foggy and 60°F`; + } +}); + +const finalMessage = await anthropic.beta.messages.toolRunner({ + model: "claude-opus-4-8", + max_tokens: 1000, + messages: [{ role: "user", content: "What is the weather in San Francisco?" }], + tools: [weatherTool] +}); + +console.log(finalMessage.content); +``` + +### Tool errors + +To report an error from a tool back to the model, throw a `ToolError` from the `run` function. Unlike a plain `Error`, `ToolError` accepts content blocks, allowing you to include images or other structured content in the error response: + +```typescript +import { ToolError } from "@anthropic-ai/sdk/lib/tools/BetaRunnableTool"; + +const screenshotTool = betaZodTool({ + name: "take_screenshot", + inputSchema: z.object({ url: z.string() }), + run: async (input) => { + if (!isValidUrl(input.url)) { + throw new ToolError(`Invalid URL: ${input.url}`); + } + const result = await takeScreenshot(input.url); + if (result.error) { + // Include the error screenshot so the model can see what went wrong + throw new ToolError([ + { type: "text", text: `Failed to load page: ${result.error}` }, + { + type: "image", + source: { type: "base64", data: result.screenshot, media_type: "image/png" } + } + ]); + } + return { + type: "image", + source: { type: "base64", data: result.screenshot, media_type: "image/png" } + }; + } +}); +``` + +If a plain `Error` is thrown, the message will be converted to a text content block. + +## Tool use + +This SDK provides support for tool use, also known as function calling. For more details, see [Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview). + +## MCP helpers + +This SDK provides helpers for integrating with [Model Context Protocol (MCP)](https://modelcontextprotocol.io/) servers. These helpers convert MCP types to Claude API types, reducing boilerplate when working with MCP tools, prompts, and resources. + + + The Claude API also supports an [`mcp_servers` parameter](/docs/en/agents-and-tools/mcp-connector) that lets Claude connect directly to remote MCP servers. Use `mcp_servers` when you have remote servers accessible by URL and only need tool support. Use the MCP helpers when you need local MCP servers, prompts, resources, or more control over the MCP connection. + + +```typescript +import { + mcpTools, + mcpMessages, + mcpResourceToContent, + mcpResourceToFile +} from "@anthropic-ai/sdk/helpers/beta/mcp"; +import { Client } from "@modelcontextprotocol/sdk/client/index.js"; +import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"; + +const anthropic = new Anthropic(); + +// Connect to an MCP server +const transport = new StdioClientTransport({ command: "mcp-server", args: [] }); +const mcpClient = new Client({ name: "my-client", version: "1.0.0" }); +await mcpClient.connect(transport); + +// Use MCP prompts +const { messages } = await mcpClient.getPrompt({ name: "my-prompt" }); +const response = await anthropic.beta.messages.create({ + model: "claude-opus-4-8", + max_tokens: 1024, + messages: mcpMessages(messages) +}); +console.log(response.content); + +// Use MCP tools with toolRunner +const { tools } = await mcpClient.listTools(); +const finalMessage = await anthropic.beta.messages.toolRunner({ + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [{ role: "user", content: "Use the available tools" }], + tools: mcpTools(tools, mcpClient) +}); +console.log(finalMessage.content); + +// Use MCP resources as content +const resource = await mcpClient.readResource({ uri: "file:///path/to/doc.txt" }); +await anthropic.beta.messages.create({ + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [ + { + role: "user", + content: [ + mcpResourceToContent(resource), + { type: "text", text: "Summarize this document" } + ] + } + ] +}); + +// Upload MCP resources as files +const fileResource = await mcpClient.readResource({ uri: "file:///path/to/data.json" }); +await anthropic.beta.files.upload({ file: mcpResourceToFile(fileResource) }); +``` + +### MCP error handling + +The conversion functions throw `UnsupportedMCPValueError` if an MCP value isn't supported by the Claude API (for example, unsupported content type, unsupported MIME type, non-http/https resource link). + +## Message batches + +This SDK provides support for the [Message Batches API](/docs/en/build-with-claude/batch-processing) under the `client.messages.batches` namespace. + +### Creating a batch + +Message Batches takes an array of requests, where each object has a `custom_id` identifier, and the exact same request `params` as the standard Messages API: + +```typescript +const batch = await client.messages.batches.create({ + requests: [ + { + custom_id: "my-first-request", + params: { + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, world" }] + } + }, + { + custom_id: "my-second-request", + params: { + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [{ role: "user", content: "Hi again, friend" }] + } + } + ] +}); +``` + +### Getting results from a batch + +Once a Message Batch has been processed, indicated by `.processing_status === 'ended'`, you can access the results with `.batches.results()` + +```typescript +const results = await client.messages.batches.results(batch.id); +for await (const entry of results) { + if (entry.result.type === "succeeded") { + console.log(entry.result.message.content); + } +} +``` + +## File uploads + +Request parameters that correspond to file uploads can be passed in many different forms: + +* `File` (or an object with the same structure) +* a `fetch` `Response` (or an object with the same structure) +* an `fs.ReadStream` +* the return value of the `toFile` helper + +Set the content-type explicitly as the files API will not infer it for you: + +```typescript +import fs from "node:fs"; +import Anthropic, { toFile } from "@anthropic-ai/sdk"; + +const client = new Anthropic(); + +// If you have access to Node `fs`, use `fs.createReadStream()`: +await client.beta.files.upload({ + file: await toFile(fs.createReadStream("/path/to/file"), undefined, { + type: "application/json" + }) +}); + +// Or if you have the web `File` API you can pass a `File` instance: +await client.beta.files.upload({ + file: new File(["my bytes"], "file.txt", { type: "text/plain" }) +}); +// You can also pass a `fetch` `Response`: +await client.beta.files.upload({ + file: await fetch("https://somesite/file") +}); + +// Or a `Buffer` / `Uint8Array` +await client.beta.files.upload({ + file: await toFile(Buffer.from("my bytes"), "file", { type: "text/plain" }) +}); +await client.beta.files.upload({ + file: await toFile(new Uint8Array([0, 1, 2]), "file", { type: "text/plain" }) +}); +``` + +## Handling errors + +When the library is unable to connect to the API, or if the API returns a non-success status code (that is, 4xx or 5xx response), a subclass of `APIError` is thrown: + +```typescript +const message = await client.messages + .create({ + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" + }) + .catch(async (err) => { + if (err instanceof Anthropic.APIError) { + console.log(err.status); // 400 + console.log(err.name); // BadRequestError + console.log(err.headers); // {server: 'nginx', ...} + } else { + throw err; + } + }); +``` + +Error codes are as follows: + +| Status code | Error type | +| ----------- | -------------------------- | +| 400 | `BadRequestError` | +| 401 | `AuthenticationError` | +| 403 | `PermissionDeniedError` | +| 404 | `NotFoundError` | +| 409 | `ConflictError` | +| 422 | `UnprocessableEntityError` | +| 429 | `RateLimitError` | +| >=500 | `InternalServerError` | +| N/A | `APIConnectionError` | + +## Request IDs + +> For more information on debugging requests, see [Request ID](/docs/en/api/errors#request-id). + +All object responses in the SDK provide a `_request_id` property which is added from the `request-id` response header so that you can quickly log failing requests and report them back to Anthropic. + +```typescript +const message = await client.messages.create({ + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" +}); +console.log(message._request_id); // req_018EeWyXxfu5pfWkrYcMdjWG +``` + +## Retries + +Certain errors are automatically retried 2 times by default, with a short exponential backoff. Connection errors (for example, because of a network connectivity problem), 408 Request Timeout, 409 Conflict, 429 Rate Limit, and >=500 Internal errors are all retried by default. + +You can use the `maxRetries` option to configure or disable this: + +```typescript +// Configure the default for all requests: +const client = new Anthropic({ + maxRetries: 0 // default is 2 +}); + +// Or, configure per-request: +await client.messages.create( + { + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" + }, + { maxRetries: 5 } +); +``` + +## Timeouts + +By default requests time out after 10 minutes. However if you have specified a large `max_tokens` value and are *not* streaming, the default timeout will be calculated dynamically using the formula: + +```typescript +const minimum = 10 * 60; +const calculated = (60 * 60 * maxTokens) / 128_000; +return calculated < minimum ? minimum * 1000 : calculated * 1000; +``` + +which will result in a timeout up to 60 minutes, scaled by the `max_tokens` parameter, unless overridden at the request or client level. + +You can configure this with a `timeout` option: + +```typescript +// Configure the default for all requests: +const client = new Anthropic({ + timeout: 20 * 1000 // 20 seconds (default is 10 minutes) +}); + +// Override per-request: +await client.messages.create( + { + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" + }, + { timeout: 5 * 1000 } +); +``` + +On timeout, an `APIConnectionTimeoutError` is thrown. + +Note that requests that time out are [retried twice by default](#retries). + +## Long requests + + + Consider using the streaming [Messages API](#streaming-responses) for longer running requests. + + +Avoid setting a large `max_tokens` value without using streaming. Some networks may drop idle connections after a certain period of time, which can cause the request to fail or [timeout](#timeouts) without receiving a response from Anthropic. + +This SDK also throws an error if a non-streaming request is expected to be above roughly 10 minutes long. Passing `stream: true` or [overriding](#timeouts) the `timeout` option at the client or request level disables this error. + +An expected request latency longer than the [timeout](#timeouts) for a non-streaming request will result in the client terminating the connection and retrying without receiving a response. + +When supported by the `fetch` implementation, the SDK sets a [TCP socket keep-alive](https://tldp.org/HOWTO/TCP-Keepalive-HOWTO/overview.html) option to reduce the impact of idle connection timeouts on some networks. This can be [overridden](#configuring-proxies) by configuring a custom proxy. + +## Auto-pagination + +List methods in the Claude API are paginated. You can use the `for await ... of` syntax to iterate through items across all pages: + +```typescript +async function fetchAllMessageBatches() { + const allMessageBatches = []; + // Automatically fetches more pages as needed. + for await (const messageBatch of client.messages.batches.list({ limit: 20 })) { + allMessageBatches.push(messageBatch); + } + return allMessageBatches; +} +``` + +Alternatively, you can request a single page at a time: + +```typescript +let page = await client.messages.batches.list({ limit: 20 }); +for (const messageBatch of page.data) { + console.log(messageBatch); +} + +// Convenience methods are provided for manually paginating: +while (page.hasNextPage()) { + page = await page.getNextPage(); + // ... +} +``` + +## Default headers + +The SDK automatically sends the `anthropic-version` header set to `2023-06-01`. + +If you need to, you can override it by setting default headers on a per-request basis. + +Be aware that doing so may result in incorrect types and other unexpected or undefined behavior in the SDK. + +```typescript +const client = new Anthropic(); + +const message = await client.messages.create( + { + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" + }, + { headers: { "anthropic-version": "My-Custom-Value" } } +); +``` + +## Advanced usage + +### Accessing raw Response data (for example, headers) + +The "raw" `Response` returned by `fetch()` can be accessed through the `.asResponse()` method on the `APIPromise` type that all methods return. This method returns as soon as the headers for a successful response are received and does not consume the response body, so you are free to write custom parsing or streaming logic. + +You can also use the `.withResponse()` method to get the raw `Response` along with the parsed data. Unlike `.asResponse()` this method consumes the body, returning once it is parsed. + +```typescript +const client = new Anthropic(); + +const response = await client.messages + .create({ + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" + }) + .asResponse(); +console.log(response.headers.get("X-My-Header")); +console.log(response.statusText); // access the underlying Response object + +const { data: message, response: raw } = await client.messages + .create({ + max_tokens: 1024, + messages: [{ role: "user", content: "Hello, Claude" }], + model: "claude-opus-4-8" + }) + .withResponse(); +console.log(raw.headers.get("X-My-Header")); +console.log(message.content); +``` + +### Logging + + + All log messages are intended for debugging only. The format and content of log messages may change between releases. + + +#### Log levels + +You can configure the log level in two ways: + +1. Through the `ANTHROPIC_LOG` environment variable +2. Using the `logLevel` client option (overrides the environment variable if set) + +```typescript +const client = new Anthropic({ + logLevel: "debug" // Show all log messages +}); +``` + +Available log levels, from most to least verbose: + +* `'debug'` - Show debug messages, info, warnings, and errors +* `'info'` - Show info messages, warnings, and errors +* `'warn'` - Show warnings and errors (default) +* `'error'` - Show only errors +* `'off'` - Disable all logging + +At the `'debug'` level, all HTTP requests and responses are logged, including headers and bodies. Some authentication-related headers are redacted, but sensitive data in request and response bodies may still be visible. + +#### Custom logger + +By default, this library logs to `globalThis.console`. You can also provide a custom logger. Most logging libraries are supported, including [pino](https://www.npmjs.com/package/pino), [winston](https://www.npmjs.com/package/winston), [bunyan](https://www.npmjs.com/package/bunyan), [consola](https://www.npmjs.com/package/consola), [signale](https://www.npmjs.com/package/signale), and [@std/log](https://jsr.io/@std/log). If your logger doesn't work, open an issue. + +When providing a custom logger, the `logLevel` option still controls which messages are emitted; messages below the configured level will not be sent to your logger. + +```typescript +import pino from "pino"; + +const logger = pino(); + +const client = new Anthropic({ + logger: logger.child({ name: "Anthropic" }), + logLevel: "debug" // Send all messages to pino, allowing it to filter +}); +``` + +### Making custom/undocumented requests + +This library is typed for convenient access to the documented API. If you need to access undocumented endpoints, params, or response properties, the library can still be used. + +#### Undocumented endpoints + +To make requests to undocumented endpoints, you can use `client.get`, `client.post`, and other HTTP verbs. Options on the client, such as retries, are respected when making these requests. + +```typescript +await client.post("/some/path", { + body: { some_prop: "foo" }, + query: { some_query_arg: "bar" } +}); +``` + +#### Undocumented request parameters + +To make requests using undocumented parameters, you may use `// @ts-expect-error` on the undocumented parameter. This library doesn't validate at runtime that the request matches the type, so any extra values you send will be sent as-is. + +```typescript +client.messages.create({ + // ... + // @ts-expect-error baz is not yet public + baz: "undocumented option" +}); +``` + +For requests with the `GET` verb, any extra parameters will be in the query; all other requests will send the extra parameter in the body. + +If you want to explicitly send an extra argument, you can do so with the `query`, `body`, and `headers` request options. + +#### Undocumented response properties + +To access undocumented response properties, you may access the response object with `// @ts-expect-error` on the response object, or cast the response object to the requisite type. Like the request parameters, the SDK does not validate or strip extra properties from the response from the API. + +### Customizing the fetch client + +By default, this library expects a global `fetch` function is defined. + +If you want to use a different `fetch` function, you can either polyfill the global: + +```typescript +import fetch from "my-fetch"; + +globalThis.fetch = fetch; +``` + +Or pass it to the client: + +```typescript +import fetch from "my-fetch"; + +const client = new Anthropic({ fetch }); +``` + +### Fetch options + +If you want to set custom `fetch` options without overriding the `fetch` function, you can provide a `fetchOptions` object when creating the client or making a request. (Request-specific options override client options.) + +```typescript +const client = new Anthropic({ + fetchOptions: { + // `RequestInit` options + } +}); +``` + +### Configuring proxies + +To modify proxy behavior, you can provide custom `fetchOptions` that add runtime-specific proxy options to requests: + + + + ```typescript + import * as undici from "undici"; + + const proxyAgent = new undici.ProxyAgent("http://localhost:8888"); + const client = new Anthropic({ + fetchOptions: { + dispatcher: proxyAgent + } + }); + ``` + + + + ```typescript + const client = new Anthropic({ + fetchOptions: { + proxy: "http://localhost:8888" + } + }); + ``` + + + + ```typescript + import Anthropic from "npm:@anthropic-ai/sdk"; + + const httpClient = Deno.createHttpClient({ proxy: { url: "http://localhost:8888" } }); + const client = new Anthropic({ + fetchOptions: { + client: httpClient + } + }); + ``` + + + +## Beta features + +Beta features are available before general release to get early feedback and test new functionality. You can check the availability of all of Claude's capabilities and tools in the [build with Claude overview](/docs/en/build-with-claude/overview). + +You can access most beta API features through the beta property of the client. To enable a particular beta feature, you need to add the appropriate [beta header](/docs/en/api/beta-headers) to the `betas` field when creating a message. + +For example, to use the [Files API](/docs/en/build-with-claude/files): + +```typescript +const client = new Anthropic(); +const response = await client.beta.messages.create({ + model: "claude-opus-4-8", + max_tokens: 1024, + messages: [ + { + role: "user", + content: [ + { type: "text", text: "Please summarize this document for me." }, + { + type: "document", + source: { + type: "file", + file_id: "file_abc123" + } + } + ] + } + ], + betas: ["files-api-2025-04-14"] +}); +``` + +## Runtime support + + + Enabling the `dangerouslyAllowBrowser` option can be dangerous because it exposes your secret API credentials in the client-side code. Web browsers are inherently less secure than server environments, any user with access to the browser can potentially inspect, extract, and misuse these credentials. This could lead to unauthorized access using your credentials and potentially compromise sensitive data or functionality. + + **When might this not be dangerous?** + + In certain scenarios where enabling browser support might not pose significant risks: + + * **Internal tools:** If the application is used solely within a controlled internal environment where the users are trusted, the risk of credential exposure can be mitigated. + * **Development or debugging purpose:** Enabling this feature temporarily might be acceptable, provided the credentials are short-lived, aren't also used in production environments, or are frequently rotated. + + +## Platform integrations + + + For detailed platform setup guides with code examples, see: + + * [Amazon Bedrock](/docs/en/build-with-claude/claude-in-amazon-bedrock) + * [Amazon Bedrock (Opus 4.6 and earlier)](/docs/en/build-with-claude/claude-on-amazon-bedrock-legacy) + * [Claude Platform on AWS](/docs/en/build-with-claude/claude-platform-on-aws) + * [Google Cloud](/docs/en/build-with-claude/claude-on-vertex-ai) + * [Microsoft Foundry](/docs/en/build-with-claude/claude-in-microsoft-foundry) + + +The TypeScript SDK supports the following platforms: + +* **Agent Platform:** `npm install @anthropic-ai/vertex-sdk`: Provides `AnthropicVertex` client +* **Bedrock:** `npm install @anthropic-ai/bedrock-sdk`: Provides `AnthropicBedrockMantle` client, and `AnthropicBedrock` for the `bedrock-runtime` path +* **Claude Platform on AWS:** `npm install @anthropic-ai/aws-sdk`: Provides `AnthropicAws` client. Pass `workspaceId` to the constructor or set the `ANTHROPIC_AWS_WORKSPACE_ID` environment variable. Available in beta. +* **Foundry:** `npm install @anthropic-ai/foundry-sdk`: Provides `AnthropicFoundry` client + +Use `AnthropicBedrockMantle` for new projects; `AnthropicBedrock` remains for existing applications using the Bedrock `InvokeModel` API. + +## Semantic versioning + +This package generally follows [SemVer](https://semver.org/spec/v2.0.0.html) conventions, though certain backwards-incompatible changes may be released as minor versions: + +1. Changes that only affect static types, without breaking runtime behavior. +2. Changes to library internals which are technically public but not intended or documented for external use. +3. Changes that aren't expected to impact the vast majority of users in practice. + +Backwards-compatibility is taken seriously to ensure you can rely on a smooth upgrade experience. + +## Frequently asked questions + +See the [GitHub repository](https://github.com/anthropics/anthropic-sdk-typescript) for FAQs, issues, and community support. + +## Additional resources + +* [GitHub repository](https://github.com/anthropics/anthropic-sdk-typescript) +* [API reference](/docs/en/api/overview) +* [Streaming Messages](/docs/en/build-with-claude/streaming) +* [Tool use with Claude](/docs/en/agents-and-tools/tool-use/overview) diff --git a/content/en/docs/claude-code/accessibility.md b/content/en/docs/claude-code/accessibility.md index c3192f581..ce6e2dc59 100644 --- a/content/en/docs/claude-code/accessibility.md +++ b/content/en/docs/claude-code/accessibility.md @@ -20,10 +20,10 @@ Pick the method that matches how often you use a screen reader: * For one session: run `claude --ax-screen-reader`. * For sessions started from one shell: set the `CLAUDE_AX_SCREEN_READER` environment variable to `1`. In Bash or Zsh, run `export CLAUDE_AX_SCREEN_READER=1`; in PowerShell, run `$env:CLAUDE_AX_SCREEN_READER = "1"`. Add the line to your shell profile to cover every shell. -* For every session on the machine: add `"axScreenReader": true` to your user [settings file](/en/settings). This covers any terminal, including the VS Code integrated terminal. +* For every session on the machine: add `"axScreenReader": true` to your user [settings file](/docs/en/settings). This covers any terminal, including the VS Code integrated terminal. - The methods are listed in precedence order: the [`--ax-screen-reader`](/en/cli-reference#cli-flags) flag overrides the [`CLAUDE_AX_SCREEN_READER`](/en/env-vars) environment variable, which overrides the [`axScreenReader`](/en/settings#available-settings) setting. + The methods are listed in precedence order: the [`--ax-screen-reader`](/docs/en/cli-reference#cli-flags) flag overrides the [`CLAUDE_AX_SCREEN_READER`](/docs/en/env-vars) environment variable, which overrides the [`axScreenReader`](/docs/en/settings#available-settings) setting. If you use Claude Code over SSH, set the environment variable or setting on the remote machine where Claude Code runs. @@ -46,7 +46,7 @@ In screen reader mode, Claude Code writes flat text: Output accumulates in your terminal's scrollback, so you can re-read earlier turns with your screen reader's review commands or your terminal's search. -Screen reader mode renders as plain scrolling text, even if you've turned on [fullscreen rendering](/en/fullscreen) with the [`tui` setting](/en/settings#available-settings); the setting has no effect while the mode is active. Attached background sessions still render fullscreen; see [Known limitations](#known-limitations). +Screen reader mode renders as plain scrolling text, even if you've turned on [fullscreen rendering](/docs/en/fullscreen) with the [`tui` setting](/docs/en/settings#available-settings); the setting has no effect while the mode is active. Attached background sessions still render fullscreen; see [Known limitations](#known-limitations). Each message in the transcript starts with a label your screen reader announces, naming what it is: your messages, Claude's replies, tool activity, errors, and prompts. The labels are also searchable, so you can jump between sections of the transcript by searching your terminal's scrollback: @@ -58,11 +58,11 @@ Each message in the transcript starts with a label your screen reader announces, | `tool error:` | A tool that failed | | `error:` | An error in the conversation, such as a failed API request | | `Permission Required:` | A permission prompt waiting for your answer | -| `Cost:` | The session cost summary when Claude Code exits, if your account [shows costs](/en/costs) | +| `Cost:` | The session cost summary when Claude Code exits, if your account [shows costs](/docs/en/costs) | The terminal cursor follows the input caret, so a screen reader's read-current-line command answers "where am I" with the prompt you're editing. -{/* min-version: 2.1.210 */}Cycling [permission modes](/en/permission-modes) with `Shift+Tab` announces the mode you land on, such as `[plan mode on]` or `[accept edits on]`. Claude Code prints the announcement once and doesn't repeat it on later redraws. Requires Claude Code v2.1.210 or later. +{/* min-version: 2.1.210 */}Cycling [permission modes](/docs/en/permission-modes) with `Shift+Tab` announces the mode you land on, such as `[plan mode on]` or `[accept edits on]`. Claude Code prints the announcement once and doesn't repeat it on later redraws. Requires Claude Code v2.1.210 or later. ### Jump between turns @@ -92,25 +92,25 @@ In screen reader mode, Claude Code rings the terminal bell when it needs your at * a permission prompt appears * a tool that ran longer than 5 seconds finishes -The bell is your terminal's standard alert. To silence it, change the bell setting in your terminal application. The bell doesn't require screen reader mode: outside the mode, set [`preferredNotifChannel`](/en/settings#available-settings) to `"terminal_bell"` for similar alerts when Claude is waiting on you. See [Get a terminal bell or notification](/en/terminal-config#get-a-terminal-bell-or-notification). +The bell is your terminal's standard alert. To silence it, change the bell setting in your terminal application. The bell doesn't require screen reader mode: outside the mode, set [`preferredNotifChannel`](/docs/en/settings#available-settings) to `"terminal_bell"` for similar alerts when Claude is waiting on you. See [Get a terminal bell or notification](/docs/en/terminal-config#get-a-terminal-bell-or-notification). ## Accessibility settings beyond screen reader mode These options address accessibility needs outside of screen reader mode. All of them work alongside it. -* The `CLAUDE_CODE_ACCESSIBILITY` [environment variable](/en/env-vars) is for screen magnifiers. Set `CLAUDE_CODE_ACCESSIBILITY=1` to keep the native terminal cursor visible so that magnifiers, such as macOS Zoom, can track the cursor position. -* The `prefersReducedMotion` [setting](/en/settings#available-settings) reduces or disables spinners, shimmer, and other animations without changing the rest of the interface. -* The `theme` [setting](/en/settings#available-settings) selects the interface colors, including the colorblind-friendly `dark-daltonized` and `light-daltonized` themes. +* The `CLAUDE_CODE_ACCESSIBILITY` [environment variable](/docs/en/env-vars) is for screen magnifiers. Set `CLAUDE_CODE_ACCESSIBILITY=1` to keep the native terminal cursor visible so that magnifiers, such as macOS Zoom, can track the cursor position. +* The `prefersReducedMotion` [setting](/docs/en/settings#available-settings) reduces or disables spinners, shimmer, and other animations without changing the rest of the interface. +* The `theme` [setting](/docs/en/settings#available-settings) selects the interface colors, including the colorblind-friendly `dark-daltonized` and `light-daltonized` themes. ## Known limitations Some behaviors aren't adapted for screen reader mode: * Screen reader mode doesn't turn on automatically when a screen reader is running. -* Claude Code doesn't announce a permission mode change made in any way other than cycling with `Shift+Tab`, such as entering [plan mode](/en/permission-modes#analyze-before-you-edit-with-plan-mode) from a command. -* Attaching to a [background session](/en/agent-view) with `claude attach` or from agent view enters the terminal's alternate screen, which has no native scrollback. This is the [same behavior as other attached sessions](/en/fullscreen). To get back out, press Left Arrow on an empty prompt, or Ctrl+Z if a dialog has focus. +* Claude Code doesn't announce a permission mode change made in any way other than cycling with `Shift+Tab`, such as entering [plan mode](/docs/en/permission-modes#analyze-before-you-edit-with-plan-mode) from a command. +* Attaching to a [background session](/docs/en/agent-view) with `claude attach` or from agent view enters the terminal's alternate screen, which has no native scrollback. This is the [same behavior as other attached sessions](/docs/en/fullscreen). To get back out, press Left Arrow on an empty prompt, or Ctrl+Z if a dialog has focus. * Claude Code announces costs in the summary it prints at exit, not per turn. -* Screen reader mode doesn't change [non-interactive mode](/en/headless) with the `-p` flag. Non-interactive mode already writes plain text and remains an alternative for scripting. +* Screen reader mode doesn't change [non-interactive mode](/docs/en/headless) with the `-p` flag. Non-interactive mode already writes plain text and remains an alternative for scripting. ## Report an issue @@ -120,8 +120,8 @@ If something doesn't work with your screen reader, magnifier, or terminal, open These pages hold the full reference entries and related setup for what this page covers: -* [Settings](/en/settings#available-settings): the `axScreenReader`, `prefersReducedMotion`, `theme`, and `preferredNotifChannel` entries -* [Environment variables](/en/env-vars): the `CLAUDE_AX_SCREEN_READER` and `CLAUDE_CODE_ACCESSIBILITY` entries -* [CLI reference](/en/cli-reference#cli-flags): the `--ax-screen-reader` flag -* [Terminal configuration](/en/terminal-config): bells, notifications, and themes outside screen reader mode -* [Non-interactive mode](/en/headless): scripted `claude -p` runs, which write plain text without screen reader mode +* [Settings](/docs/en/settings#available-settings): the `axScreenReader`, `prefersReducedMotion`, `theme`, and `preferredNotifChannel` entries +* [Environment variables](/docs/en/env-vars): the `CLAUDE_AX_SCREEN_READER` and `CLAUDE_CODE_ACCESSIBILITY` entries +* [CLI reference](/docs/en/cli-reference#cli-flags): the `--ax-screen-reader` flag +* [Terminal configuration](/docs/en/terminal-config): bells, notifications, and themes outside screen reader mode +* [Non-interactive mode](/docs/en/headless): scripted `claude -p` runs, which write plain text without screen reader mode diff --git a/content/en/docs/claude-code/admin-setup.md b/content/en/docs/claude-code/admin-setup.md index 512bcc853..3125a2e0c 100644 --- a/content/en/docs/claude-code/admin-setup.md +++ b/content/en/docs/claude-code/admin-setup.md @@ -16,11 +16,11 @@ This page walks through the deployment decisions in order. Each row links to the | Decision | What you're choosing | Reference | | :---------------------------------------------------------------------- | :-------------------------------------------------- | :---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| [Choose your API provider](#choose-your-api-provider) | Where Claude Code authenticates and how it's billed | [Authentication](/en/authentication), [Amazon Bedrock](/en/amazon-bedrock), [Google Cloud's Agent Platform](/en/google-vertex-ai), [Microsoft Foundry](/en/microsoft-foundry) | -| [Decide how settings reach devices](#decide-how-settings-reach-devices) | How managed policy reaches developer machines | [Server-managed settings](/en/server-managed-settings), [Settings files](/en/settings#settings-files) | -| [Decide what to enforce](#decide-what-to-enforce) | Which tools, commands, and integrations are allowed | [Permissions](/en/permissions), [Sandboxing](/en/sandboxing) | -| [Set up usage visibility](#set-up-usage-visibility) | How you track spend and adoption | [Analytics](/en/analytics), [Monitoring](/en/monitoring-usage), [Costs](/en/costs) | -| [Review data handling](#review-data-handling) | Data retention and compliance posture | [Data usage](/en/data-usage), [Security](/en/security) | +| [Choose your API provider](#choose-your-api-provider) | Where Claude Code authenticates and how it's billed | [Authentication](/docs/en/authentication), [Amazon Bedrock](/docs/en/amazon-bedrock), [Google Cloud's Agent Platform](/docs/en/google-vertex-ai), [Microsoft Foundry](/docs/en/microsoft-foundry) | +| [Decide how settings reach devices](#decide-how-settings-reach-devices) | How managed policy reaches developer machines | [Server-managed settings](/docs/en/server-managed-settings), [Settings files](/docs/en/settings#settings-files) | +| [Decide what to enforce](#decide-what-to-enforce) | Which tools, commands, and integrations are allowed | [Permissions](/docs/en/permissions), [Sandboxing](/docs/en/sandboxing) | +| [Set up usage visibility](#set-up-usage-visibility) | How you track spend and adoption | [Analytics](/docs/en/analytics), [Monitoring](/docs/en/monitoring-usage), [Costs](/docs/en/costs) | +| [Review data handling](#review-data-handling) | Data retention and compliance posture | [Data usage](/docs/en/data-usage), [Security](/docs/en/security) | ## Choose your API provider @@ -34,76 +34,76 @@ Claude Code connects to Claude through one of several API providers. Your choice | Google Cloud's Agent Platform | You want to inherit existing GCP compliance controls and billing | | Microsoft Foundry | You want to inherit existing Azure compliance controls and billing | -Some Claude Code features require a claude.ai account. [Claude Code on the web](/en/claude-code-on-the-web), [Routines](/en/routines), [Code Review](/en/code-review), [Remote Control](/en/remote-control), and the [Chrome extension](/en/chrome) aren't available through Console API keys or cloud-provider credentials alone. If you deploy through Amazon Bedrock, Google Cloud's Agent Platform, or Microsoft Foundry, plan whether developers also need Claude for Teams or Enterprise seats. Each feature page lists its plan requirements. +Some Claude Code features require a claude.ai account. [Claude Code on the web](/docs/en/claude-code-on-the-web), [Routines](/docs/en/routines), [Code Review](/docs/en/code-review), [Remote Control](/docs/en/remote-control), and the [Chrome extension](/docs/en/chrome) aren't available through Console API keys or cloud-provider credentials alone. If you deploy through Amazon Bedrock, Google Cloud's Agent Platform, or Microsoft Foundry, plan whether developers also need Claude for Teams or Enterprise seats. Each feature page lists its plan requirements. -For the full provider comparison covering authentication, regions, and feature parity, see the [enterprise deployment overview](/en/third-party-integrations). Each provider's auth setup is in [Authentication](/en/authentication). +For the full provider comparison covering authentication, regions, and feature parity, see the [enterprise deployment overview](/docs/en/third-party-integrations). Each provider's auth setup is in [Authentication](/docs/en/authentication). -Proxy and firewall requirements in [Network configuration](/en/network-config) apply regardless of provider. If you want a single endpoint in front of multiple providers or centralized request logging, see [LLM gateway](/en/llm-gateway). +Proxy and firewall requirements in [Network configuration](/docs/en/network-config) apply regardless of provider. If you want a single endpoint in front of multiple providers or centralized request logging, see [LLM gateway](/docs/en/llm-gateway). ## Decide how settings reach devices -Managed settings define policy that takes precedence over local developer configuration. Claude Code checks the four sources below in priority order and applies the first one that returns a non-empty configuration. A small set of [cross-source lock keys](/en/settings#settings-precedence), such as the sandbox allowlist locks, is honored when any admin-controlled source sets them; when a [`policyHelper`](/en/settings#compute-managed-settings-with-a-policy-helper) is configured, its output is the only source these checks read. +Managed settings define policy that takes precedence over local developer configuration. Claude Code checks the four sources below in priority order and applies the first one that returns a non-empty configuration. A small set of [cross-source lock keys](/docs/en/settings#settings-precedence), such as the sandbox allowlist locks, is honored when any admin-controlled source sets them; when a [`policyHelper`](/docs/en/settings#compute-managed-settings-with-a-policy-helper) is configured, its output is the only source these checks read. | Mechanism | Delivery | Priority | Platforms | | :---------------------- | :---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :------- | :------------- | -| Server-managed | claude.ai admin console, or a self-hosted [Claude apps gateway](/en/claude-apps-gateway) for gateway sign-ins | Highest | All | +| Server-managed | claude.ai admin console, or a self-hosted [Claude apps gateway](/docs/en/claude-apps-gateway) for gateway sign-ins | Highest | All | | plist / registry policy | macOS: `com.anthropic.claudecode` plist
Windows: `HKLM\SOFTWARE\Policies\ClaudeCode` | High | macOS, Windows | | File-based managed | macOS: `/Library/Application Support/ClaudeCode/managed-settings.json`
Linux and WSL: `/etc/claude-code/managed-settings.json`
Windows: `C:\Program Files\ClaudeCode\managed-settings.json` | Medium | All | | Windows user registry | `HKCU\SOFTWARE\Policies\ClaudeCode` | Lowest | Windows only | -A configured [`policyHelper`](/en/settings#compute-managed-settings-with-a-policy-helper) preempts all four sources: its output becomes the only managed configuration for the run. See [Settings precedence](/en/settings#settings-precedence). +A configured [`policyHelper`](/docs/en/settings#compute-managed-settings-with-a-policy-helper) preempts all four sources: its output becomes the only managed configuration for the run. See [Settings precedence](/docs/en/settings#settings-precedence). -Server-managed settings reach devices at authentication time and refresh hourly during active sessions, with no endpoint infrastructure. Delivery through the claude.ai admin console requires a Claude for Teams or Enterprise plan. Deployments on Amazon Bedrock, Google Cloud's Agent Platform, or Microsoft Foundry can get the same remote delivery by running a [Claude apps gateway](/en/claude-apps-gateway), or use one of the file-based or OS-level mechanisms instead. +Server-managed settings reach devices at authentication time and refresh hourly during active sessions, with no endpoint infrastructure. Delivery through the claude.ai admin console requires a Claude for Teams or Enterprise plan. Deployments on Amazon Bedrock, Google Cloud's Agent Platform, or Microsoft Foundry can get the same remote delivery by running a [Claude apps gateway](/docs/en/claude-apps-gateway), or use one of the file-based or OS-level mechanisms instead. -If your organization mixes providers, configure [server-managed settings](/en/server-managed-settings) for claude.ai users plus a [file-based or plist/registry fallback](/en/settings#settings-files) so other users still receive managed policy. +If your organization mixes providers, configure [server-managed settings](/docs/en/server-managed-settings) for claude.ai users plus a [file-based or plist/registry fallback](/docs/en/settings#settings-files) so other users still receive managed policy. The plist and HKLM registry locations work with any provider and resist tampering because they require admin privileges to write. The Windows user registry at HKCU is writable without elevation, so treat it as a convenience default rather than an enforcement channel. -By default, WSL reads only the Linux file path at `/etc/claude-code`. To extend your Windows registry and `C:\Program Files\ClaudeCode` policy to WSL on the same machine, set [`wslInheritsWindowsSettings: true`](/en/settings#available-settings) in either of those admin-only Windows sources. +By default, WSL reads only the Linux file path at `/etc/claude-code`. To extend your Windows registry and `C:\Program Files\ClaudeCode` policy to WSL on the same machine, set [`wslInheritsWindowsSettings: true`](/docs/en/settings#available-settings) in either of those admin-only Windows sources. -Whichever mechanism you choose, managed values take precedence over user and project settings. Array settings such as `permissions.allow` and `permissions.deny` merge entries from all sources, so developers can extend managed lists but not remove from them. For [two exceptions](/en/settings#settings-precedence), `fallbackModel` and `availableModels`, the managed value replaces lower layers rather than merging. +Whichever mechanism you choose, managed values take precedence over user and project settings. Array settings such as `permissions.allow` and `permissions.deny` merge entries from all sources, so developers can extend managed lists but not remove from them. For [two exceptions](/docs/en/settings#settings-precedence), `fallbackModel` and `availableModels`, the managed value replaces lower layers rather than merging. -See [Server-managed settings](/en/server-managed-settings) and [Settings files and precedence](/en/settings#settings-files). +See [Server-managed settings](/docs/en/server-managed-settings) and [Settings files and precedence](/docs/en/settings#settings-files). ### WSL sessions in Claude Code Desktop -On Windows, [Claude Code Desktop can run Code sessions inside a WSL 2 distribution](/en/desktop-wsl). The session's Claude Code process runs inside the distribution, so it resolves managed settings through the WSL discovery path above: Windows-only sources don't reach it unless `wslInheritsWindowsSettings: true` is deployed. +On Windows, [Claude Code Desktop can run Code sessions inside a WSL 2 distribution](/docs/en/desktop-wsl). The session's Claude Code process runs inside the distribution, so it resolves managed settings through the WSL discovery path above: Windows-only sources don't reach it unless `wslInheritsWindowsSettings: true` is deployed. On devices where managed settings are present, Desktop WSL sessions are unavailable by default. If your organization wants to enable them, contact your Anthropic account team. When they're enabled: * Deploy `wslInheritsWindowsSettings: true` through the HKLM registry or the `C:\Program Files\ClaudeCode` file so WSL sessions inherit the same policy as host sessions. * Verify by running `/status` inside a WSL session: the `Setting sources` line should show `Enterprise managed settings` with the Windows source you deployed, `(HKLM)` or `(file)`. -Processes inside the WSL 2 utility VM aren't visible to Windows-side endpoint detection sensors. If you use CrowdStrike Falcon, enable the Falcon sensor for Linux on WSL 2 with the two exclusions CrowdStrike's WSL documentation requires, for the WSL virtual machine process and the VM disk image, so in-distro process and file activity is observable. Claude Code's [OpenTelemetry tool-execution telemetry](/en/monitoring-usage) is emitted identically for WSL and native sessions. +Processes inside the WSL 2 utility VM aren't visible to Windows-side endpoint detection sensors. If you use CrowdStrike Falcon, enable the Falcon sensor for Linux on WSL 2 with the two exclusions CrowdStrike's WSL documentation requires, for the WSL virtual machine process and the VM disk image, so in-distro process and file activity is observable. Claude Code's [OpenTelemetry tool-execution telemetry](/docs/en/monitoring-usage) is emitted identically for WSL and native sessions. ## Decide what to enforce Managed settings can lock down tools, sandbox execution, restrict MCP servers and plugin sources, and control which hooks run. Each row is a control surface with the setting keys that drive it. -| Control | What it does | Key settings | -| :------------------------------------------------------------------------------------- | :-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :----------------------------------------------------------------------------------------------------------- | -| [Permission rules](/en/permissions) | Allow, ask, or deny specific tools and commands | `permissions.allow`, `permissions.deny` | -| [Permission lockdown](/en/permissions#managed-only-settings) | Only managed permission rules apply; disable `--dangerously-skip-permissions` | `allowManagedPermissionRulesOnly`, `permissions.disableBypassPermissionsMode` | -| [Sandboxing](/en/sandboxing) | OS-level filesystem and network isolation with domain allowlists | `sandbox.enabled`, `sandbox.network.allowedDomains` | -| [Managed policy CLAUDE.md](/en/memory#deploy-organization-wide-claude-md) | Org-wide instructions loaded in every session, can't be excluded | File at the managed policy path | -| [MCP server control](/en/managed-mcp) | Restrict which MCP servers users can add or connect to, or deploy a fixed set | `allowedMcpServers`, `deniedMcpServers`, `allowManagedMcpServersOnly`, or a deployed `managed-mcp.json` file | -| [Plugin marketplace control](/en/plugin-marketplaces#managed-marketplace-restrictions) | Restrict which marketplace sources users can add and install from, reject the CLI flags that sideload plugins, agents, and MCP servers for a single run, and allowlist which marketplaces' plugins can be suggested | `strictKnownMarketplaces`, `blockedMarketplaces`, `disableSideloadFlags`, `pluginSuggestionMarketplaces` | -| [Customization lockdown](/en/settings#strictpluginonlycustomization) | Block skills, agents, hooks, and MCP servers from user and project sources, so they can only come from plugins or managed settings | `strictPluginOnlyCustomization` | -| [Hook restrictions](/en/settings#hook-configuration) | Only managed hooks load; restrict HTTP hook URLs | `allowManagedHooksOnly`, `allowedHttpHookUrls` | -| [Login enforcement](/en/settings#available-settings) | Restrict login to a specific method or Anthropic organization. The method restriction is enforced across the terminal, VS Code extension, Agent SDK, `claude setup-token`, and `/install-github-app`; the organization restriction covers the terminal, VS Code extension, and Agent SDK. {/* min-version: 2.1.212 */}Before v2.1.212, only terminal logins enforced either key. When set, sessions authenticated by `ANTHROPIC_API_KEY`, `ANTHROPIC_AUTH_TOKEN`, or `apiKeyHelper` are blocked at startup; cloud provider sessions aren't affected | `forceLoginMethod`, `forceLoginOrgUUID` | -| [Disable agent view](/en/agent-view#how-background-sessions-are-hosted) | Turn off `claude agents`, `--bg`, `/background`, and the on-demand supervisor | `disableAgentView` | -| [Configure the corporate launcher](/en/corporate-launcher) | Prefix the [background-agent supervisor](/en/agent-view#how-background-sessions-are-hosted), its workers, and the [other covered background processes](/en/corporate-launcher#what-the-launcher-covers) with a required corporate launcher instead of turning agent view off | `processWrapper` | -| [Model restrictions](/en/model-config#restrict-model-selection) | `availableModels` filters which models appear in the picker. Adding `enforceAvailableModels` also constrains the auto-selected default model. See [surface coverage](/en/model-config#surface-coverage) for how this setting reaches the CLI, web, and IDE | `availableModels`, `enforceAvailableModels` | -| [Version floor](/en/settings) | Prevent auto-update from installing below an org-wide minimum | `minimumVersion` | -| [Required version range](/en/settings) | Refuse to start at all when the running version is outside an org-approved range. Stronger than `minimumVersion`, which only blocks downgrades | `requiredMinimumVersion`, `requiredMaximumVersion` | - -Organizations whose members authenticate through claude.ai or the Anthropic API can also govern models without deploying settings: [organization model restrictions](/en/model-config#organization-model-restrictions) disable individual models, an [organization default model](/en/model-config#organization-default-model) sets which model new sessions start on, and [organization effort limits](/en/model-config#organization-effort-limits) cap effort levels per role. All three controls require a Claude Enterprise plan. Model restrictions and effort limits are enforced server-side; the default model is a starting point that users can change, unless the organization enforces it. Enforcement is available to a limited set of organizations; ask your Anthropic account team about availability. None of these controls reach sessions on Amazon Bedrock, Google Cloud's Agent Platform, Microsoft Foundry, or [Claude Platform on AWS](/en/claude-platform-on-aws); on those providers, use `availableModels` above for restrictions and the `model` key in managed settings for a default. - -[Claude Code on the web](/en/claude-code-on-the-web) has its own admin surface: on the Cloud environments page in admin settings, owners and admins create [organization-shared environments](/en/claude-code-on-the-web#organization-shared-environments) that set the [network access level](/en/claude-code-on-the-web#network-access), environment variables, and setup script for members' cloud sessions, and choose the organization's default environment. +| Control | What it does | Key settings | +| :------------------------------------------------------------------------------------- | :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :----------------------------------------------------------------------------------------------------------- | +| [Permission rules](/docs/en/permissions) | Allow, ask, or deny specific tools and commands | `permissions.allow`, `permissions.deny` | +| [Permission lockdown](/docs/en/permissions#managed-only-settings) | Only managed permission rules apply; disable `--dangerously-skip-permissions` | `allowManagedPermissionRulesOnly`, `permissions.disableBypassPermissionsMode` | +| [Sandboxing](/docs/en/sandboxing) | OS-level filesystem and network isolation with domain allowlists | `sandbox.enabled`, `sandbox.network.allowedDomains` | +| [Managed policy CLAUDE.md](/docs/en/memory#deploy-organization-wide-claude-md) | Org-wide instructions loaded in every session, can't be excluded | File at the managed policy path | +| [MCP server control](/docs/en/managed-mcp) | Restrict which MCP servers users can add or connect to, or deploy a fixed set | `allowedMcpServers`, `deniedMcpServers`, `allowManagedMcpServersOnly`, or a deployed `managed-mcp.json` file | +| [Plugin marketplace control](/docs/en/plugin-marketplaces#managed-marketplace-restrictions) | Restrict which marketplace sources users can add and install from, reject the CLI flags that sideload plugins, agents, and MCP servers for a single run, and allowlist which marketplaces' plugins can be suggested | `strictKnownMarketplaces`, `blockedMarketplaces`, `disableSideloadFlags`, `pluginSuggestionMarketplaces` | +| [Customization lockdown](/docs/en/settings#strictpluginonlycustomization) | Block skills, agents, hooks, and MCP servers from user and project sources, so they can only come from plugins or managed settings | `strictPluginOnlyCustomization` | +| [Hook restrictions](/docs/en/settings#hook-configuration) | Only managed hooks load; restrict HTTP hook URLs | `allowManagedHooksOnly`, `allowedHttpHookUrls` | +| [Login enforcement](/docs/en/settings#available-settings) | Restrict login to a specific method or Anthropic organization. The method restriction is enforced across the terminal, VS Code extension, Agent SDK, `claude setup-token`, and `/install-github-app`; the organization restriction covers the terminal, VS Code extension, and Agent SDK, except [gateway](/docs/en/claude-apps-gateway) sign-in, which doesn't authenticate against an Anthropic organization. {/* min-version: 2.1.212 */}Before v2.1.212, only terminal logins enforced either key. When set, sessions authenticated by `ANTHROPIC_API_KEY`, `ANTHROPIC_AUTH_TOKEN`, or `apiKeyHelper` are blocked at startup; cloud provider sessions aren't affected | `forceLoginMethod`, `forceLoginOrgUUID` | +| [Disable agent view](/docs/en/agent-view#how-background-sessions-are-hosted) | Turn off `claude agents`, `--bg`, `/background`, and the on-demand supervisor | `disableAgentView` | +| [Configure the corporate launcher](/docs/en/corporate-launcher) | Prefix the [background-agent supervisor](/docs/en/agent-view#how-background-sessions-are-hosted), its workers, and the [other covered background processes](/docs/en/corporate-launcher#what-the-launcher-covers) with a required corporate launcher instead of turning agent view off | `processWrapper` | +| [Model restrictions](/docs/en/model-config#restrict-model-selection) | `availableModels` filters which models appear in the picker. Adding `enforceAvailableModels` also constrains the auto-selected default model. See [surface coverage](/docs/en/model-config#surface-coverage) for how this setting reaches the CLI, web, and IDE | `availableModels`, `enforceAvailableModels` | +| [Version floor](/docs/en/settings) | Prevent auto-update from installing below an org-wide minimum | `minimumVersion` | +| [Required version range](/docs/en/settings) | Refuse to start at all when the running version is outside an org-approved range. Stronger than `minimumVersion`, which only blocks downgrades | `requiredMinimumVersion`, `requiredMaximumVersion` | + +Organizations whose members authenticate through claude.ai or the Anthropic API can also govern models without deploying settings: [organization model restrictions](/docs/en/model-config#organization-model-restrictions) disable individual models, an [organization default model](/docs/en/model-config#organization-default-model) sets which model new sessions start on, and [organization effort limits](/docs/en/model-config#organization-effort-limits) cap effort levels per role. All three controls require a Claude Enterprise plan. Model restrictions and effort limits are enforced server-side; the default model is a starting point that users can change, unless the organization enforces it. Enforcement is available to a limited set of organizations; ask your Anthropic account team about availability. None of these controls reach sessions on Amazon Bedrock, Google Cloud's Agent Platform, Microsoft Foundry, or [Claude Platform on AWS](/docs/en/claude-platform-on-aws); on those providers, use `availableModels` above for restrictions and the `model` key in managed settings for a default. + +[Claude Code on the web](/docs/en/claude-code-on-the-web) has its own admin surface: on the Cloud environments page in admin settings, owners and admins create [organization-shared environments](/docs/en/claude-code-on-the-web#organization-shared-environments) that set the [network access level](/docs/en/claude-code-on-the-web#network-access), environment variables, and setup script for members' cloud sessions, and choose the organization's default environment. Permission rules and sandboxing cover different layers. Denying WebFetch blocks Claude's fetch tool, but if Bash is allowed, `curl` and `wget` can still reach any URL. Sandboxing closes that gap with a network domain allowlist enforced at the OS level. -For the threat model these controls defend against, see [Security](/en/security). +For the threat model these controls defend against, see [Security](/docs/en/security). ## Set up usage visibility @@ -111,10 +111,10 @@ Choose monitoring based on what you need to report on. The dashboards, APIs, and | Capability | What you get | Availability | Where to start | | :--------------------- | :---------------------------------------------------------------------------------------------------------------------- | :--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :---------------------------------------------------- | -| Usage monitoring | OpenTelemetry export of sessions, tools, and tokens | All providers | [Monitoring usage](/en/monitoring-usage) | -| Analytics dashboard | Adoption and contribution metrics with a leaderboard on Teams / Enterprise; per-user usage and spend metrics on Console | Teams / Enterprise at [claude.ai/analytics](https://claude.ai/analytics/claude-code), Console at [platform.claude.com/claude-code](https://platform.claude.com/claude-code) | [Analytics](/en/analytics) | -| Programmatic reporting | Per-user usage and cost data over an API | [Enterprise Analytics API](https://platform.claude.com/docs/en/api/admin/analytics) for Enterprise, [Claude Code Analytics API](https://platform.claude.com/docs/en/build-with-claude/claude-code-analytics-api) for Console | [Costs](/en/costs#manage-costs-for-your-organization) | -| Spend controls | Spend limits and rate limits | Admin settings for Teams / Enterprise, workspace limits for Console; on third-party clouds, cloud budget controls or a [Claude apps gateway](/en/claude-apps-gateway) with per-user [spend limits](/en/claude-apps-gateway-spend-limits) | [Costs](/en/costs#manage-costs-for-your-organization) | +| Usage monitoring | OpenTelemetry export of sessions, tools, and tokens | All providers | [Monitoring usage](/docs/en/monitoring-usage) | +| Analytics dashboard | Adoption and contribution metrics with a leaderboard on Teams / Enterprise; per-user usage and spend metrics on Console | Teams / Enterprise at [claude.ai/analytics](https://claude.ai/analytics/claude-code), Console at [platform.claude.com/claude-code](https://platform.claude.com/claude-code) | [Analytics](/docs/en/analytics) | +| Programmatic reporting | Per-user usage and cost data over an API | [Enterprise Analytics API](https://platform.claude.com/docs/en/api/admin/analytics) for Enterprise, [Claude Code Analytics API](https://platform.claude.com/docs/en/build-with-claude/claude-code-analytics-api) for Console | [Costs](/docs/en/costs#manage-costs-for-your-organization) | +| Spend controls | Spend limits and rate limits | Admin settings for Teams / Enterprise, workspace limits for Console; on third-party clouds, cloud budget controls or a [Claude apps gateway](/docs/en/claude-apps-gateway) with per-user [spend limits](/docs/en/claude-apps-gateway-spend-limits) | [Costs](/docs/en/costs#manage-costs-for-your-organization) | On Teams and Enterprise, per-user usage and spend numbers come from the [spend report](https://support.claude.com/en/articles/12883420-view-usage-analytics-for-team-and-enterprise-plans) in your organization's analytics settings, not the analytics dashboard. Cloud providers expose spend through AWS Cost Explorer, GCP Billing, or Azure Cost Management. For planning enterprise budgets across Claude chat, Claude Code, and Cowork, see the [Claude Enterprise consumption guide](https://support.claude.com/en/articles/14782391-claude-enterprise-consumption-guide). @@ -124,23 +124,23 @@ On Team, Enterprise, Claude API, and cloud provider plans, Anthropic doesn't tra | Topic | What to know | Where to start | | :------------------------ | :--------------------------------------------------------------------------------------------------- | :--------------------------------------------- | -| Data usage policy | What Anthropic collects, how long it's retained, what's never used for training | [Data usage](/en/data-usage) | -| Zero Data Retention (ZDR) | Nothing stored after the request completes. Available to qualified accounts on Claude for Enterprise | [Zero data retention](/en/zero-data-retention) | -| Security architecture | Network model, encryption, authentication, audit trail | [Security](/en/security) | +| Data usage policy | What Anthropic collects, how long it's retained, what's never used for training | [Data usage](/docs/en/data-usage) | +| Zero Data Retention (ZDR) | Nothing stored after the request completes. Available to qualified accounts on Claude for Enterprise | [Zero data retention](/docs/en/zero-data-retention) | +| Security architecture | Network model, encryption, authentication, audit trail | [Security](/docs/en/security) | -If you need request-level audit logging or to route traffic by data sensitivity, place a gateway between developers and your provider: a self-hosted [Claude apps gateway](/en/claude-apps-gateway) records a per-request audit log with IdP identity, or use another [LLM gateway](/en/llm-gateway). For regulatory requirements and certifications, see [Legal and compliance](/en/legal-and-compliance). +If you need request-level audit logging or to route traffic by data sensitivity, place a gateway between developers and your provider: a self-hosted [Claude apps gateway](/docs/en/claude-apps-gateway) records a per-request audit log with IdP identity, or use another [LLM gateway](/docs/en/llm-gateway). For regulatory requirements and certifications, see [Legal and compliance](/docs/en/legal-and-compliance). ## Verify and onboard -After configuring managed settings, have a developer run `/status` inside Claude Code. On the **Status** tab, the `Setting sources` line shows `Enterprise managed settings` followed by the source in parentheses, one of `(remote)`, `(plist)`, `(HKLM)`, `(HKCU)`, or `(file)`. See [Verify active settings](/en/settings#verify-active-settings). +After configuring managed settings, have a developer run `/status` inside Claude Code. On the **Status** tab, the `Setting sources` line shows `Enterprise managed settings` followed by the source in parentheses, one of `(remote)`, `(plist)`, `(HKLM)`, `(HKCU)`, or `(file)`. See [Verify active settings](/docs/en/settings#verify-active-settings). Share these resources to help developers get started: -* [Quickstart](/en/quickstart): first-session walkthrough from install to working with a project -* [Common workflows](/en/common-workflows): patterns for everyday tasks like code review, refactoring, and debugging +* [Quickstart](/docs/en/quickstart): first-session walkthrough from install to working with a project +* [Common workflows](/docs/en/common-workflows): patterns for everyday tasks like code review, refactoring, and debugging * [Claude 101](https://anthropic.skilljar.com/claude-101) and [Claude Code in Action](https://anthropic.skilljar.com/claude-code-in-action): self-paced Anthropic Academy courses -For login issues, point developers to [authentication troubleshooting](/en/troubleshoot-install#login-and-authentication). The most common fixes are: +For login issues, point developers to [authentication troubleshooting](/docs/en/troubleshoot-install#login-and-authentication). The most common fixes are: * Run `/logout` then `/login` to switch accounts * Run `claude update` if the enterprise auth option is missing @@ -152,8 +152,8 @@ If a developer sees "You haven't been added to your organization yet," their sea With provider and delivery mechanism chosen, move on to detailed configuration: -* [Server-managed settings](/en/server-managed-settings): deliver managed policy from the Claude admin console -* [Settings reference](/en/settings): every setting key, file location, and precedence rule -* [Monorepos and large repos](/en/large-codebases): per-directory configuration patterns for organizations deploying into a monorepo -* [Amazon Bedrock](/en/amazon-bedrock), [Google Cloud's Agent Platform](/en/google-vertex-ai), [Microsoft Foundry](/en/microsoft-foundry): provider-specific deployment +* [Server-managed settings](/docs/en/server-managed-settings): deliver managed policy from the Claude admin console +* [Settings reference](/docs/en/settings): every setting key, file location, and precedence rule +* [Monorepos and large repos](/docs/en/large-codebases): per-directory configuration patterns for organizations deploying into a monorepo +* [Amazon Bedrock](/docs/en/amazon-bedrock), [Google Cloud's Agent Platform](/docs/en/google-vertex-ai), [Microsoft Foundry](/docs/en/microsoft-foundry): provider-specific deployment * [Claude Enterprise Administrator Guide](https://claude.com/resources/tutorials/claude-enterprise-administrator-guide): SSO, SCIM, seat management, and rollout playbook diff --git a/content/en/docs/claude-code/advisor.md b/content/en/docs/claude-code/advisor.md index a35b46d1c..957b8c27c 100644 --- a/content/en/docs/claude-code/advisor.md +++ b/content/en/docs/claude-code/advisor.md @@ -20,20 +20,20 @@ This page covers how to enable the advisor, which model pairings are accepted, w The advisor fits long, multi-step tasks where most turns are routine but plan quality determines the outcome. Examples include large refactors, debugging sessions where an error keeps recurring, and tasks you want independently checked before Claude declares them done. -It adds less value on short tasks where there is little to plan, or on work where every turn needs the strongest model. For those, [switch the main model](/en/model-config#setting-your-model) instead, or see [how the advisor compares with opusplan and subagents](#compare-with-related-features) for other ways to get a second opinion. +It adds less value on short tasks where there is little to plan, or on work where every turn needs the strongest model. For those, [switch the main model](/docs/en/model-config#setting-your-model) instead, or see [how the advisor compares with opusplan and subagents](#compare-with-related-features) for other ways to get a second opinion. ## Enable the advisor You can set the advisor model in three ways: * **`/advisor` command**: set or change the advisor mid-session and save it as your default -* **`advisorModel` setting**: configure a persistent default in your [settings file](/en/settings) +* **`advisorModel` setting**: configure a persistent default in your [settings file](/docs/en/settings) * **`--advisor` flag**: set the advisor for a single session at launch If any of these sets an advisor model, the advisor is enabled for sessions whose main model [supports it](#choose-an-advisor-model), and an `Advisor Tool (experimental) is on and may use more tokens · /advisor` notification appears after the session starts. To stop using it, see [Turn the advisor off](#turn-the-advisor-off). - {/* min-version: 2.1.210 */}Claude Code doesn't offer Fable 5 as the advisor. For organizations with [Fable 5 access](/en/model-config#work-with-fable-5), the `/advisor` picker lists it as a dimmed, unselectable row labeled `Fable 5 (temporarily unavailable)`, and Claude Code rejects `/advisor fable` and `--advisor fable`. Fable 5 as the main model isn't affected. + {/* min-version: 2.1.210 */}Claude Code doesn't offer Fable 5 as the advisor. For organizations with [Fable 5 access](/docs/en/model-config#work-with-fable-5), the `/advisor` picker lists it as a dimmed, unselectable row labeled `Fable 5 (temporarily unavailable)`, and Claude Code rejects `/advisor fable` and `--advisor fable`. Fable 5 as the main model isn't affected. A remotely configured rollout controls when Fable 5 returns as an advisor option. @@ -48,7 +48,7 @@ Run `/advisor` without arguments to open a picker listing the available advisor The command confirms with `Advisor set to` followed by the advisor model name. Your selection is saved to `advisorModel` in your user settings and persists across sessions. -If your organization's [`availableModels`](/en/model-config#restrict-model-selection) allowlist excludes the saved advisor model, the advisor is not invoked until you pick an allowed model with `/advisor`. If your current main model does not support the advisor, the selection is still saved and activates when you switch to a [compatible main model](#choose-an-advisor-model) with [`/model`](/en/model-config#setting-your-model). +If your organization's [`availableModels`](/docs/en/model-config#restrict-model-selection) allowlist excludes the saved advisor model, the advisor is not invoked until you pick an allowed model with `/advisor`. If your current main model does not support the advisor, the selection is still saved and activates when you switch to a [compatible main model](#choose-an-advisor-model) with [`/model`](/docs/en/model-config#setting-your-model). ### Set `advisorModel` in settings @@ -68,7 +68,7 @@ To set the advisor for a single session without changing your saved setting, lau claude --advisor opus ``` -The flag takes precedence over the `advisorModel` setting for that session, and isn't listed in `claude --help`. It exits with an error if the session's main model does not support the advisor, or if the requested advisor model is excluded by your organization's [`availableModels`](/en/model-config#restrict-model-selection) allowlist. +The flag takes precedence over the `advisorModel` setting for that session, and isn't listed in `claude --help`. It exits with an error if the session's main model does not support the advisor, or if the requested advisor model is excluded by your organization's [`availableModels`](/docs/en/model-config#restrict-model-selection) allowlist. ## Choose an advisor model @@ -125,13 +125,13 @@ The advisor always receives the full conversation, and Claude controls the timin Each advisor call sends the conversation to the advisor model, so it consumes tokens at the advisor model's rates in addition to your main model's usage. With API billing, advisor tokens are charged at the advisor model's input and output rates. On subscription plans, advisor usage counts toward your plan's usage limits. -Claude calls the advisor at decision points rather than on every turn, so pairing a faster main model with a stronger advisor typically costs less than running the stronger model throughout. Advisor usage counts toward the session totals shown by [`/usage`](/en/costs#track-your-costs). +Claude calls the advisor at decision points rather than on every turn, so pairing a faster main model with a stronger advisor typically costs less than running the stronger model throughout. Advisor usage counts toward the session totals shown by [`/usage`](/docs/en/costs#track-your-costs). For how advisor tokens are reported in API responses, see [Usage and billing](https://platform.claude.com/docs/en/agents-and-tools/tool-use/advisor-tool#usage-and-billing) in the Claude API documentation. ## Impact on prompt caching -Enabling or disabling the advisor mid-session does not invalidate your main model's [prompt cache](/en/prompt-caching). Unlike [changing model or effort level](/en/prompt-caching#actions-that-invalidate-the-cache), toggling `/advisor` keeps the cached prefix intact, and the advisor's returned guidance is cached as part of the transcript on later turns. +Enabling or disabling the advisor mid-session does not invalidate your main model's [prompt cache](/docs/en/prompt-caching). Unlike [changing model or effort level](/docs/en/prompt-caching#actions-that-invalidate-the-cache), toggling `/advisor` keeps the cached prefix intact, and the advisor's returned guidance is cached as part of the transcript on later turns. The advisor model's own read of the conversation is not cached. Each advisor call processes the full transcript anew, with no reuse between calls. @@ -139,7 +139,7 @@ The advisor model's own read of the conversation is not cached. Each advisor cal The advisor tool requires all of the following: -* **Anthropic API only**: the advisor is a server-executed tool. It is not available on Amazon Bedrock, Claude Platform on AWS, Google Cloud's Agent Platform, or Microsoft Foundry. Through an [LLM gateway](/en/llm-gateway) configured with `ANTHROPIC_BASE_URL`, availability depends on whether the gateway forwards the request intact to the Anthropic API. +* **Anthropic API only**: the advisor is a server-executed tool. It is not available on Amazon Bedrock, Claude Platform on AWS, Google Cloud's Agent Platform, or Microsoft Foundry. Through an [LLM gateway](/docs/en/llm-gateway) configured with `ANTHROPIC_BASE_URL`, availability depends on whether the gateway forwards the request intact to the Anthropic API. * **Supported main model**: Opus 4.6 or later, Sonnet 4.6 or later, or Haiku 4.5. {/* min-version: 2.1.170 */}Fable 5 also qualifies on Claude Code v2.1.170 or later, but a Fable 5 main [accepts only a Fable advisor](#choose-an-advisor-model) and Fable [isn't offered as the advisor](#enable-the-advisor), so a Fable 5 session runs without one until the rollout returns it as an option. ## Turn the advisor off @@ -150,7 +150,7 @@ To stop using the advisor and clear your saved `advisorModel`, run `/advisor off /advisor off ``` -To disable the advisor tool entirely, set `CLAUDE_CODE_DISABLE_ADVISOR_TOOL=1`. The `/advisor` command becomes unavailable and any configured `advisorModel` is ignored. The `--advisor` flag is accepted but has no effect; existing scripts that pass it continue to work without errors. See [Environment variables](/en/env-vars). +To disable the advisor tool entirely, set `CLAUDE_CODE_DISABLE_ADVISOR_TOOL=1`. The `/advisor` command becomes unavailable and any configured `advisorModel` is ignored. The `--advisor` flag is accepted but has no effect; existing scripts that pass it continue to work without errors. See [Environment variables](/docs/en/env-vars). ## Compare with related features @@ -159,13 +159,13 @@ The advisor is one of several ways to combine model strengths. Pick based on whe | Approach | When the stronger model runs | How it starts | | ----------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------- | | Advisor tool | At decision points mid-task | Claude calls it when it needs guidance | -| [`opusplan`](/en/model-config#opusplan-model-setting) | During plan mode when [allowed by `availableModels`](/en/model-config#restrict-model-selection), then switches to Sonnet for execution | You enter plan mode | -| [Subagents](/en/sub-agents#choose-a-model) with `model` set | For the entire delegated subtask | Claude delegates, or you invoke the subagent | -| [`/model`](/en/model-config#setting-your-model) | For all subsequent turns | You switch models | +| [`opusplan`](/docs/en/model-config#opusplan-model-setting) | During plan mode when [allowed by `availableModels`](/docs/en/model-config#restrict-model-selection), then switches to Sonnet for execution | You enter plan mode | +| [Subagents](/docs/en/sub-agents#choose-a-model) with `model` set | For the entire delegated subtask | Claude delegates, or you invoke the subagent | +| [`/model`](/docs/en/model-config#setting-your-model) | For all subsequent turns | You switch models | ## See also -* [Model configuration](/en/model-config): switch models, set effort levels, and use `opusplan` -* [Manage costs effectively](/en/costs): track token usage across models +* [Model configuration](/docs/en/model-config): switch models, set effort levels, and use `opusplan` +* [Manage costs effectively](/docs/en/costs): track token usage across models * [Advisor tool in the Claude API](https://platform.claude.com/docs/en/agents-and-tools/tool-use/advisor-tool): understand the underlying server tool, or use it directly from the Messages API * [The advisor strategy](https://claude.com/blog/the-advisor-strategy): why pairing a fast main model with a stronger advisor works diff --git a/content/en/docs/claude-code/agent-sdk/agent-loop.md b/content/en/docs/claude-code/agent-sdk/agent-loop.md index ae07eb013..f054fdf12 100644 --- a/content/en/docs/claude-code/agent-sdk/agent-loop.md +++ b/content/en/docs/claude-code/agent-sdk/agent-loop.md @@ -8,7 +8,7 @@ The Agent SDK lets you embed Claude Code's autonomous agent loop in your own applications. The SDK is a standalone package that gives you programmatic control over tools, permissions, cost limits, and output. You don't need the Claude Code CLI installed to use it. -When you start an agent, the SDK runs the same [execution loop that powers Claude Code](/en/how-claude-code-works#the-agentic-loop): Claude evaluates your prompt, calls tools to take action, receives the results, and repeats until the task is complete. This page explains what happens inside that loop so you can build, debug, and optimize your agents effectively. +When you start an agent, the SDK runs the same [execution loop that powers Claude Code](/docs/en/how-claude-code-works#the-agentic-loop): Claude evaluates your prompt, calls tools to take action, receives the results, and repeats until the task is complete. This page explains what happens inside that loop so you can build, debug, and optimize your agents effectively. ## The loop at a glance @@ -18,7 +18,7 @@ Every agent session follows the same cycle: 1. **Receive prompt.** Claude receives your prompt, along with the system prompt, tool definitions, and conversation history. The SDK yields a [`SystemMessage`](#message-types) with subtype `"init"` containing session metadata. 2. **Evaluate and respond.** Claude evaluates the current state and determines how to proceed. It may respond with text, request one or more tool calls, or both. The SDK yields an [`AssistantMessage`](#message-types) containing the text and any tool call requests. -3. **Execute tools.** The SDK runs each requested tool and collects the results. Each set of tool results feeds back to Claude for the next decision. You can use [hooks](/en/agent-sdk/hooks) to intercept, modify, or block tool calls before they run. +3. **Execute tools.** The SDK runs each requested tool and collects the results. Each set of tool results feeds back to Claude for the next decision. You can use [hooks](/docs/en/agent-sdk/hooks) to intercept, modify, or block tool calls before they run. 4. **Repeat.** Steps 2 and 3 repeat as a cycle. Each full cycle is one turn. Claude continues calling tools and processing results until it produces a response with no tool calls. 5. **Return result.** The SDK yields a final [`AssistantMessage`](#message-types) with the text response (no tool calls), followed by a [`ResultMessage`](#message-types) with the final text, token usage, cost, and session ID. @@ -49,18 +49,18 @@ As the loop runs, the SDK yields a stream of messages. Each message carries a ty * **`SystemMessage`:** session lifecycle events. The `subtype` field distinguishes them: - * `"init"`: session metadata for the run. When a `SessionStart` or `Setup` hook runs during session startup, its [hook lifecycle messages](/en/agent-sdk/typescript#sdkhookstartedmessage) arrive before the `init` message + * `"init"`: session metadata for the run. When a `SessionStart` or `Setup` hook runs during session startup, its [hook lifecycle messages](/docs/en/agent-sdk/typescript#sdkhookstartedmessage) arrive before the `init` message * `"compact_boundary"`: fires after [compaction](#automatic-compaction) * `"informational"`: plain-text status banners from the loop * `"worker_shutting_down"`: the loop will end after the current turn because the host is exiting or Remote Control disconnected - In TypeScript, each subtype other than `"init"` is its own type in the [`SDKMessage` union](/en/agent-sdk/typescript#sdkmessage) rather than a subtype of `SDKSystemMessage`. + In TypeScript, each subtype other than `"init"` is its own type in the [`SDKMessage` union](/docs/en/agent-sdk/typescript#sdkmessage) rather than a subtype of `SDKSystemMessage`. * **`AssistantMessage`:** emitted after each Claude response, including the final text-only one. Contains text content blocks and tool call blocks from that turn. * **`UserMessage`:** emitted after each tool execution with the tool result content sent back to Claude. Also emitted for any user inputs you stream mid-loop. -* **`StreamEvent`:** only emitted when partial messages are enabled. Contains raw API streaming events (text deltas, tool input chunks). See [Stream responses](/en/agent-sdk/streaming-output). +* **`StreamEvent`:** only emitted when partial messages are enabled. Contains raw API streaming events (text deltas, tool input chunks). See [Stream responses](/docs/en/agent-sdk/streaming-output). * **`ResultMessage`:** marks the end of the agent loop. Contains the final text result, token usage, cost, and session ID. Check the `subtype` field to determine whether the task succeeded or hit a limit. A small number of trailing system events, such as `prompt_suggestion`, can arrive after it, so iterate the stream to completion rather than breaking on the result. See [Handle the result](#handle-the-result). -These five types cover the full agent loop lifecycle. Both SDKs also yield observability events such as rate-limit status and task notifications that are not required to drive the loop. See the [Python message types reference](/en/agent-sdk/python#message-types) and [TypeScript message types reference](/en/agent-sdk/typescript#message-types) for the complete lists. +These five types cover the full agent loop lifecycle. Both SDKs also yield observability events such as rate-limit status and task notifications that are not required to drive the loop. See the [Python message types reference](/docs/en/agent-sdk/python#message-types) and [TypeScript message types reference](/docs/en/agent-sdk/typescript#message-types) for the complete lists. ### Handle messages @@ -68,7 +68,7 @@ Which messages you handle depends on what you're building: * **Final results only:** handle `ResultMessage` to get the output, cost, and whether the task succeeded or hit a limit. * **Progress updates:** handle `AssistantMessage` to see what Claude is doing each turn, including which tools it called. -* **Live streaming:** enable partial messages (`include_partial_messages` in Python, `includePartialMessages` in TypeScript) to get `StreamEvent` messages in real time. See [Stream responses in real-time](/en/agent-sdk/streaming-output). +* **Live streaming:** enable partial messages (`include_partial_messages` in Python, `includePartialMessages` in TypeScript) to get `StreamEvent` messages in real time. See [Stream responses in real-time](/docs/en/agent-sdk/streaming-output). How you check message types depends on the SDK: @@ -147,19 +147,19 @@ The SDK includes the same tools that power Claude Code: Beyond built-in tools, you can: -* **Connect external services** with [MCP servers](/en/agent-sdk/mcp) (databases, browsers, APIs) -* **Define custom tools** with [custom tool handlers](/en/agent-sdk/custom-tools) -* **Load project skills** via [setting sources](/en/agent-sdk/claude-code-features) for reusable workflows +* **Connect external services** with [MCP servers](/docs/en/agent-sdk/mcp) (databases, browsers, APIs) +* **Define custom tools** with [custom tool handlers](/docs/en/agent-sdk/custom-tools) +* **Load project skills** via [setting sources](/docs/en/agent-sdk/claude-code-features) for reusable workflows ### Tool permissions Claude determines which tools to call based on the task, but you control whether those calls are allowed to execute. You can auto-approve specific tools, block others entirely, or require approval for everything. Three options work together to determine what runs: * **`allowed_tools` / `allowedTools`** auto-approves listed tools. A read-only agent with `["Read", "Glob", "Grep"]` in its allowed tools list runs those tools without prompting. Tools not listed are still available but require permission. -* **`disallowed_tools` / `disallowedTools`** blocks listed tools, regardless of other settings. See [Permissions](/en/agent-sdk/permissions) for the order that rules are checked before a tool runs. +* **`disallowed_tools` / `disallowedTools`** blocks listed tools, regardless of other settings. See [Permissions](/docs/en/agent-sdk/permissions) for the order that rules are checked before a tool runs. * **`permission_mode` / `permissionMode`** controls what happens to tools that aren't covered by allow or deny rules. See [Permission mode](#permission-mode) for available modes. -You can also scope individual tools with rules like `"Bash(npm *)"` to allow only specific commands. See [Permissions](/en/agent-sdk/permissions) for the full rule syntax. +You can also scope individual tools with rules like `"Bash(npm *)"` to allow only specific commands. See [Permissions](/docs/en/agent-sdk/permissions) for the full rule syntax. When a tool is denied, Claude receives a rejection message as the tool result and typically attempts a different approach or reports that it couldn't proceed. @@ -167,11 +167,11 @@ When a tool is denied, Claude receives a rejection message as the tool result an When Claude requests multiple tool calls in a single turn, both SDKs can run them concurrently or sequentially depending on the tool. Read-only tools (like `Read`, `Glob`, `Grep`, and MCP tools marked as read-only) can run concurrently. Tools that modify state (like `Edit`, `Write`, and `Bash`) run sequentially to avoid conflicts. -Custom tools default to sequential execution. To enable parallel execution for a custom tool, set `readOnlyHint` in its annotations. Both the [TypeScript](/en/agent-sdk/typescript#tool) and [Python](/en/agent-sdk/python#tool) SDKs use this field name from the MCP SDK. +Custom tools default to sequential execution. To enable parallel execution for a custom tool, set `readOnlyHint` in its annotations. Both the [TypeScript](/docs/en/agent-sdk/typescript#tool) and [Python](/docs/en/agent-sdk/python#tool) SDKs use this field name from the MCP SDK. ## Control how the loop runs -You can limit how many turns the loop takes, how much it costs, how deeply Claude reasons, and whether tools require approval before running. All of these are fields on [`ClaudeAgentOptions`](/en/agent-sdk/python#claudeagentoptions) (Python) / [`Options`](/en/agent-sdk/typescript#options) (TypeScript). +You can limit how many turns the loop takes, how much it costs, how deeply Claude reasons, and whether tools require approval before running. All of these are fields on [`ClaudeAgentOptions`](/docs/en/agent-sdk/python#claudeagentoptions) (Python) / [`Options`](/docs/en/agent-sdk/typescript#options) (TypeScript). ### Turns and budget @@ -180,9 +180,9 @@ You can limit how many turns the loop takes, how much it costs, how deeply Claud | Max turns (`max_turns` / `maxTurns`) | Maximum tool-use round trips | No limit | | Max budget (`max_budget_usd` / `maxBudgetUsd`) | Maximum cost before stopping | No limit | -When either limit is hit, the SDK returns a `ResultMessage` with a corresponding error subtype (`error_max_turns` or `error_max_budget_usd`). See [Handle the result](#handle-the-result) for how to check these subtypes and [`ClaudeAgentOptions`](/en/agent-sdk/python#claudeagentoptions) / [`Options`](/en/agent-sdk/typescript#options) for syntax. +When either limit is hit, the SDK returns a `ResultMessage` with a corresponding error subtype (`error_max_turns` or `error_max_budget_usd`). See [Handle the result](#handle-the-result) for how to check these subtypes and [`ClaudeAgentOptions`](/docs/en/agent-sdk/python#claudeagentoptions) / [`Options`](/docs/en/agent-sdk/typescript#options) for syntax. -With [streaming input](/en/agent-sdk/streaming-vs-single-mode), a message you send while a turn is still running stays queued when that turn ends at the max-turns limit, and it starts its own turn with its own max-turns limit. Before v2.1.205, a message that arrived on the turn's final iteration could be consumed into the ending turn and lost without ever reaching the model. +With [streaming input](/docs/en/agent-sdk/streaming-vs-single-mode), a message you send while a turn is still running stays queued when that turn ends at the max-turns limit, and it starts its own turn with its own max-turns limit. Before v2.1.205, a message that arrived on the turn's final iteration could be consumed into the ending turn and lost without ever reaching the model. ### Effort level @@ -202,7 +202,7 @@ If you don't set `effort`, both SDKs leave the parameter unset and defer to the `effort` trades latency and token cost for reasoning depth within each response. [Extended thinking](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) is a separate feature that produces visible chain-of-thought blocks in the output. They are independent: you can set `effort: "low"` with extended thinking enabled, or `effort: "max"` without it. -Use lower effort for agents doing simple, well-scoped tasks (like listing files or running a single grep) to reduce cost and latency. Set `effort` in the top-level `query()` options for the whole session, or per subagent with the `effort` field on [`AgentDefinition`](/en/agent-sdk/subagents#agentdefinition-configuration) to override the session level. +Use lower effort for agents doing simple, well-scoped tasks (like listing files or running a single grep) to reduce cost and latency. Set `effort` in the top-level `query()` options for the whole session, or per subagent with the `effort` field on [`AgentDefinition`](/docs/en/agent-sdk/subagents#agentdefinition-configuration) to override the session level. ### Permission mode @@ -213,11 +213,11 @@ The permission mode option (`permission_mode` in Python, `permissionMode` in Typ | `"default"` | Tools not covered by allow rules trigger your approval callback; no callback means deny | | `"acceptEdits"` | Auto-approves file edits and common filesystem commands (`mkdir`, `touch`, `mv`, `cp`, etc.); other Bash commands follow default rules | | `"plan"` | Claude explores and plans without editing your source files; file edits are never auto-approved and prompt through your `canUseTool` callback | -| `"dontAsk"` | Never prompts. Tools pre-approved by [permission rules](/en/settings#permission-settings) run; everything else is denied. `AskUserQuestion`, connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools), and MCP tools marked [`requiresUserInteraction`](/en/mcp#require-approval-for-a-specific-tool) are denied even if you've allowed them | -| `"auto"` | Uses a model classifier to approve or deny each tool call. See [Auto mode](/en/permission-modes#eliminate-prompts-with-auto-mode) for availability and behavior | -| `"bypassPermissions"` | Runs all allowed tools without asking, except tools matched by an explicit [`ask` rule](/en/settings#permission-settings), connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools), and tools that require user interaction; see [How permissions are evaluated](/en/agent-sdk/permissions#how-permissions-are-evaluated) for the precedence order. Cannot be used when running as root on Unix. Use only in isolated environments where the agent's actions cannot affect systems you care about | +| `"dontAsk"` | Never prompts. Tools pre-approved by [permission rules](/docs/en/settings#permission-settings) run; everything else is denied. `AskUserQuestion`, connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools), and MCP tools marked [`requiresUserInteraction`](/docs/en/mcp#require-approval-for-a-specific-tool) are denied even if you've allowed them | +| `"auto"` | Uses a model classifier to approve or deny permission prompts. See [Auto mode](/docs/en/permission-modes#eliminate-prompts-with-auto-mode) for availability and behavior | +| `"bypassPermissions"` | Runs all allowed tools without asking, except tools matched by an explicit [`ask` rule](/docs/en/settings#permission-settings), connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools), and tools that require user interaction; see [How permissions are evaluated](/docs/en/agent-sdk/permissions#how-permissions-are-evaluated) for the precedence order. Cannot be used when running as root on Unix. Use only in isolated environments where the agent's actions cannot affect systems you care about | -For interactive applications, use `"default"` with a tool approval callback to surface approval prompts. For autonomous agents on a dev machine, `"acceptEdits"` auto-approves file edits and common filesystem commands (`mkdir`, `touch`, `mv`, `cp`, etc.) while still gating other `Bash` commands behind allow rules. Reserve `"bypassPermissions"` for CI, containers, or other isolated environments. See [Permissions](/en/agent-sdk/permissions) for full details. +For interactive applications, use `"default"` with a tool approval callback to surface approval prompts. For autonomous agents on a dev machine, `"acceptEdits"` auto-approves file edits and common filesystem commands (`mkdir`, `touch`, `mv`, `cp`, etc.) while still gating other `Bash` commands behind allow rules. Reserve `"bypassPermissions"` for CI, containers, or other isolated environments. See [Permissions](/docs/en/agent-sdk/permissions) for full details. ### Model @@ -225,7 +225,7 @@ If you don't set `model`, the SDK uses Claude Code's default, which depends on y ## The context window -The context window is the total amount of information available to Claude during a session. It does not reset between turns within a session. Everything accumulates: the system prompt, tool definitions, conversation history, tool inputs, and tool outputs. Content that stays the same across turns (system prompt, tool definitions, CLAUDE.md) is automatically [prompt cached](https://platform.claude.com/docs/en/build-with-claude/prompt-caching), which reduces cost and latency for repeated prefixes. For how a custom system prompt or `append` text affects cache reuse across sessions, see [Modifying system prompts](/en/agent-sdk/modifying-system-prompts#improve-prompt-caching-across-users-and-machines). +The context window is the total amount of information available to Claude during a session. It does not reset between turns within a session. Everything accumulates: the system prompt, tool definitions, conversation history, tool inputs, and tool outputs. Content that stays the same across turns (system prompt, tool definitions, CLAUDE.md) is automatically [prompt cached](https://platform.claude.com/docs/en/build-with-claude/prompt-caching), which reduces cost and latency for repeated prefixes. For how a custom system prompt or `append` text affects cache reuse across sessions, see [Modifying system prompts](/docs/en/agent-sdk/modifying-system-prompts#improve-prompt-caching-across-users-and-machines). ### What consumes context @@ -234,8 +234,8 @@ Here's how each component affects context in the SDK: | Source | When it loads | Impact | | :----------------------- | :------------------------------------------------------------------------ | :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | | **System prompt** | Every request | Small fixed cost, always present | -| **CLAUDE.md files** | Session start, via [`settingSources`](/en/agent-sdk/claude-code-features) | Full content in every request (but prompt-cached, so only the first request pays full cost) | -| **Tool definitions** | Every request; MCP schemas deferred by default | Built-in tool schemas load every request. [Tool search](/en/agent-sdk/mcp#mcp-tool-search) defers MCP tool schemas by default, falling back to upfront loading on Google Cloud's Agent Platform or a non-first-party `ANTHROPIC_BASE_URL`. See [Configure tool search](/en/agent-sdk/tool-search#configure-tool-search) for the full matrix | +| **CLAUDE.md files** | Session start, via [`settingSources`](/docs/en/agent-sdk/claude-code-features) | Full content in every request (but prompt-cached, so only the first request pays full cost) | +| **Tool definitions** | Every request; MCP schemas deferred by default | Built-in tool schemas load every request. [Tool search](/docs/en/agent-sdk/mcp#mcp-tool-search) defers MCP tool schemas by default, falling back to upfront loading on Google Cloud's Agent Platform or a non-first-party `ANTHROPIC_BASE_URL`. See [Configure tool search](/docs/en/agent-sdk/tool-search#configure-tool-search) for the full matrix | | **Conversation history** | Accumulates over turns | Grows with each turn: prompts, responses, tool inputs, tool outputs | | **Skill descriptions** | Session start, via setting sources | Short summaries; full content loads only when invoked | @@ -245,13 +245,13 @@ Large tool outputs consume significant context. Reading a big file or running a When the context window approaches its limit, the SDK automatically compacts the conversation: it summarizes older history to free space, keeping your most recent exchanges and key decisions intact. The SDK emits a message with `type: "system"` and `subtype: "compact_boundary"` in the stream when this happens (in Python this is a `SystemMessage`; in TypeScript it is a separate `SDKCompactBoundaryMessage` type). -Compaction replaces older messages with a summary, so specific instructions from early in the conversation may not be preserved. Persistent rules belong in CLAUDE.md (loaded via [`settingSources`](/en/agent-sdk/claude-code-features)) rather than in the initial prompt, because CLAUDE.md content is re-injected on every request. +Compaction replaces older messages with a summary, so specific instructions from early in the conversation may not be preserved. Persistent rules belong in CLAUDE.md (loaded via [`settingSources`](/docs/en/agent-sdk/claude-code-features)) rather than in the initial prompt, because CLAUDE.md content is re-injected on every request. You can customize compaction behavior in several ways: * **Summarization instructions in CLAUDE.md:** The compactor reads your CLAUDE.md like any other context, so you can include a section telling it what to preserve when summarizing. The section header is free-form (not a magic string); the compactor matches on intent. -* **`PreCompact` hook:** Run custom logic before compaction occurs, for example to archive the full transcript. The hook receives a `trigger` field (`manual` or `auto`). See [hooks](/en/agent-sdk/hooks). -* **Manual compaction:** Send `/compact` as a prompt string to trigger compaction on demand. Commands sent this way are SDK inputs, not CLI-only shortcuts. See [commands in the SDK](/en/agent-sdk/slash-commands). +* **`PreCompact` hook:** Run custom logic before compaction occurs, for example to archive the full transcript. The hook receives a `trigger` field (`manual` or `auto`). See [hooks](/docs/en/agent-sdk/hooks). +* **Manual compaction:** Send `/compact` as a prompt string to trigger compaction on demand. Commands sent this way are SDK inputs, not CLI-only shortcuts. See [commands in the SDK](/docs/en/agent-sdk/slash-commands). Add a section to your project's CLAUDE.md telling the compactor what to preserve. The header name isn't special; use any clear label. @@ -271,12 +271,12 @@ You can customize compaction behavior in several ways: A few strategies for long-running agents: -* **Use subagents for subtasks.** Each subagent starts with a fresh conversation (no prior message history, though it does load its own system prompt and project-level context like CLAUDE.md). It does not see the parent's turns, and only its final response returns to the parent as a tool result. The main agent's context grows by that summary, not by the full subtask transcript. See [What subagents inherit](/en/agent-sdk/subagents#what-subagents-inherit) for details. -* **Be selective with tools.** Every tool definition takes context space. Use the `tools` field on [`AgentDefinition`](/en/agent-sdk/subagents#agentdefinition-configuration) to scope subagents to the minimum set they need. -* **Watch MCP server costs.** [MCP tool search](/en/agent-sdk/mcp#mcp-tool-search) defers MCP tool schemas by default and loads them on demand. When tool search is off, on Google Cloud's Agent Platform, or behind a non-first-party `ANTHROPIC_BASE_URL`, each MCP server adds all its tool schemas to every request, so a few servers with many tools can consume significant context before the agent does any work. +* **Use subagents for subtasks.** Each subagent starts with a fresh conversation (no prior message history, though it does load its own system prompt and project-level context like CLAUDE.md). It does not see the parent's turns, and only its final response returns to the parent as a tool result. The main agent's context grows by that summary, not by the full subtask transcript. See [What subagents inherit](/docs/en/agent-sdk/subagents#what-subagents-inherit) for details. +* **Be selective with tools.** Every tool definition takes context space. Use the `tools` field on [`AgentDefinition`](/docs/en/agent-sdk/subagents#agentdefinition-configuration) to scope subagents to the minimum set they need. +* **Watch MCP server costs.** [MCP tool search](/docs/en/agent-sdk/mcp#mcp-tool-search) defers MCP tool schemas by default and loads them on demand. When tool search is off, on Google Cloud's Agent Platform, or behind a non-first-party `ANTHROPIC_BASE_URL`, each MCP server adds all its tool schemas to every request, so a few servers with many tools can consume significant context before the agent does any work. * **Use lower effort for routine tasks.** Set [effort](#effort-level) to `"low"` for agents that only need to read files or list directories. This reduces token usage and cost. -For a detailed breakdown of per-feature context costs, see [Understand context costs](/en/features-overview#understand-context-costs). +For a detailed breakdown of per-feature context costs, see [Understand context costs](/docs/en/features-overview#understand-context-costs). ## Sessions and continuity @@ -284,10 +284,10 @@ Each interaction with the SDK creates or continues a session. Capture the sessio When you resume, the full context from previous turns is restored: files that were read, analysis that was performed, and actions that were taken. You can also fork a session to branch into a different approach without modifying the original. -See [Session management](/en/agent-sdk/sessions) for the full guide on resume, continue, and fork patterns. To resume sessions across stateless containers or serverless hosts, pass a [`session_store` / `sessionStore` adapter](/en/agent-sdk/session-storage) so transcripts are mirrored to your own backend and any host can resume them. The Claude Code subprocess still writes to local disk first; point `CLAUDE_CONFIG_DIR` at a temp directory in `options.env` if the local copy needs to be ephemeral. +See [Session management](/docs/en/agent-sdk/sessions) for the full guide on resume, continue, and fork patterns. To resume sessions across stateless containers or serverless hosts, pass a [`session_store` / `sessionStore` adapter](/docs/en/agent-sdk/session-storage) so transcripts are mirrored to your own backend and any host can resume them. The Claude Code subprocess still writes to local disk first; point `CLAUDE_CONFIG_DIR` at a temp directory in `options.env` if the local copy needs to be ephemeral. - In Python, `ClaudeSDKClient` handles session IDs automatically across multiple calls. See the [Python SDK reference](/en/agent-sdk/python#choosing-between-query-and-claudesdkclient) for details. + In Python, `ClaudeSDKClient` handles session IDs automatically across multiple calls. See the [Python SDK reference](/docs/en/agent-sdk/python#choosing-between-query-and-claudesdkclient) for details. ## Handle the result @@ -302,7 +302,7 @@ When the loop ends, the `ResultMessage` tells you what happened and gives you th | `error_during_execution` | An error interrupted the loop (for example, an API failure or cancelled request) | No | | `error_max_structured_output_retries` | No valid structured output was produced within the configured retry limit: every attempt failed validation, or a model fallback retracted the completed output with no successful retry | No | -The `result` field (the final text output) is only present on the `success` variant, so always check the subtype before reading it. All result subtypes carry `total_cost_usd`, `usage`, `num_turns`, and `session_id` so you can track cost and resume even after errors. In Python, `total_cost_usd` and `usage` are typed as optional and may be `None` on some error paths, so guard before formatting them. See [Tracking costs and usage](/en/agent-sdk/cost-tracking) for details on interpreting the `usage` fields. +The `result` field (the final text output) is only present on the `success` variant, so always check the subtype before reading it. All result subtypes carry `total_cost_usd`, `usage`, `num_turns`, and `session_id` so you can track cost and resume even after errors. In Python, `total_cost_usd` and `usage` are typed as optional and may be `None` on some error paths, so guard before formatting them. See [Tracking costs and usage](/docs/en/agent-sdk/cost-tracking) for details on interpreting the `usage` fields. When a query ends on an error result: @@ -311,11 +311,11 @@ The `result` field (the final text output) is only present on the `success` vari * A streaming input session stays alive, and you can keep sending messages. -The result also includes a `stop_reason` field (`string | null` in TypeScript, `str | None` in Python) indicating why the model stopped generating on its final turn. Common values are `end_turn` (model finished normally), `max_tokens` (hit the output token limit), and `refusal` (the model declined the request). On error result subtypes, `stop_reason` carries the value from the last assistant response before the loop ended. To detect refusals, check `stop_reason === "refusal"` (TypeScript) or `stop_reason == "refusal"` (Python). See [`SDKResultMessage`](/en/agent-sdk/typescript#sdkresultmessage) (TypeScript) or [`ResultMessage`](/en/agent-sdk/python#resultmessage) (Python) for the full type. +The result also includes a `stop_reason` field (`string | null` in TypeScript, `str | None` in Python) indicating why the model stopped generating on its final turn. Common values are `end_turn` (model finished normally), `max_tokens` (hit the output token limit), and `refusal` (the model declined the request). On error result subtypes, `stop_reason` carries the value from the last assistant response before the loop ended. To detect refusals, check `stop_reason === "refusal"` (TypeScript) or `stop_reason == "refusal"` (Python). See [`SDKResultMessage`](/docs/en/agent-sdk/typescript#sdkresultmessage) (TypeScript) or [`ResultMessage`](/docs/en/agent-sdk/python#resultmessage) (Python) for the full type. ## Hooks -[Hooks](/en/agent-sdk/hooks) are callbacks that fire at specific points in the loop: before a tool runs, after it returns, when the agent finishes, and so on. Some commonly used hooks are: +[Hooks](/docs/en/agent-sdk/hooks) are callbacks that fire at specific points in the loop: before a tool runs, after it returns, when the agent finishes, and so on. Some commonly used hooks are: | Hook | When it fires | Common uses | | :------------------------------- | :---------------------------------- | :----------------------------------------- | @@ -328,7 +328,7 @@ The result also includes a `stop_reason` field (`string | null` in TypeScript, ` Hooks run in your application process, not inside the agent's context window, so they don't consume context. Hooks can also short-circuit the loop: a `PreToolUse` hook that rejects a tool call prevents it from executing, and Claude receives the rejection message instead. -Both SDKs support all the events above. The TypeScript SDK includes additional events that Python does not yet support. See [Control execution with hooks](/en/agent-sdk/hooks) for the complete event list, per-SDK availability, and the full callback API. +Both SDKs support all the events above. The TypeScript SDK includes additional events that Python does not yet support. See [Control execution with hooks](/docs/en/agent-sdk/hooks) for the complete event list, per-SDK availability, and the full callback API. ## Put it all together @@ -436,11 +436,11 @@ Because a single-shot `query()` call raises after yielding an error result, the Now that you understand the loop, here's where to go depending on what you're building: -* **Haven't run an agent yet?** Start with the [quickstart](/en/agent-sdk/quickstart) to get the SDK installed and see a full example running end to end. -* **Ready to hook into your project?** [Load CLAUDE.md, skills, and filesystem hooks](/en/agent-sdk/claude-code-features) so the agent follows your project conventions automatically. -* **Building an interactive UI?** Enable [streaming](/en/agent-sdk/streaming-output) to show live text and tool calls as the loop runs. -* **Need tighter control over what the agent can do?** Lock down tool access with [permissions](/en/agent-sdk/permissions), and use [hooks](/en/agent-sdk/hooks) to audit, block, or transform tool calls before they execute. -* **Running long or expensive tasks?** Offload isolated work to [subagents](/en/agent-sdk/subagents) to keep your main context lean. -* **Deploying as a service?** See [Hosting the Agent SDK](/en/agent-sdk/hosting) for container and serverless guidance, and [Session storage](/en/agent-sdk/session-storage) to persist sessions to your own backend. +* **Haven't run an agent yet?** Start with the [quickstart](/docs/en/agent-sdk/quickstart) to get the SDK installed and see a full example running end to end. +* **Ready to hook into your project?** [Load CLAUDE.md, skills, and filesystem hooks](/docs/en/agent-sdk/claude-code-features) so the agent follows your project conventions automatically. +* **Building an interactive UI?** Enable [streaming](/docs/en/agent-sdk/streaming-output) to show live text and tool calls as the loop runs. +* **Need tighter control over what the agent can do?** Lock down tool access with [permissions](/docs/en/agent-sdk/permissions), and use [hooks](/docs/en/agent-sdk/hooks) to audit, block, or transform tool calls before they execute. +* **Running long or expensive tasks?** Offload isolated work to [subagents](/docs/en/agent-sdk/subagents) to keep your main context lean. +* **Deploying as a service?** See [Hosting the Agent SDK](/docs/en/agent-sdk/hosting) for container and serverless guidance, and [Session storage](/docs/en/agent-sdk/session-storage) to persist sessions to your own backend. -For the broader conceptual picture of the agentic loop (not SDK-specific), see [How Claude Code works](/en/how-claude-code-works). For a practical guide to designing loops in Claude Code, from turn-based to goal-based and proactive loops, see [Loop engineering: getting started with loops](https://claude.com/blog/getting-started-with-loops) on the blog. +For the broader conceptual picture of the agentic loop (not SDK-specific), see [How Claude Code works](/docs/en/how-claude-code-works). For a practical guide to designing loops in Claude Code, from turn-based to goal-based and proactive loops, see [Loop engineering: getting started with loops](https://claude.com/blog/getting-started-with-loops) on the blog. diff --git a/content/en/docs/claude-code/agent-sdk/claude-code-features.md b/content/en/docs/claude-code/agent-sdk/claude-code-features.md index 7b2d1ae3c..6b9234792 100644 --- a/content/en/docs/claude-code/agent-sdk/claude-code-features.md +++ b/content/en/docs/claude-code/agent-sdk/claude-code-features.md @@ -10,11 +10,11 @@ The Agent SDK is built on the same foundation as Claude Code, which means your S When you omit `settingSources`, `query()` reads the same filesystem settings as the Claude Code CLI: user, project, and local settings, CLAUDE.md files, and `.claude/` skills, agents, and commands. To run without these, pass `settingSources: []`, which limits the agent to what you configure programmatically. Managed policy settings and the global `~/.claude.json` config are read regardless of this option. See [What settingSources does not control](#what-settingsources-does-not-control). -For a conceptual overview of what each feature does and when to use it, see [Extend Claude Code](/en/features-overview). +For a conceptual overview of what each feature does and when to use it, see [Extend Claude Code](/docs/en/features-overview). ## Control filesystem settings with settingSources -The setting sources option ([`setting_sources`](/en/agent-sdk/python#claudeagentoptions) in Python, [`settingSources`](/en/agent-sdk/typescript#settingsource) in TypeScript) controls which filesystem-based settings the SDK loads. Pass an explicit list to opt in to specific sources, or pass an empty array to disable user, project, and local settings. +The setting sources option ([`setting_sources`](/docs/en/agent-sdk/python#claudeagentoptions) in Python, [`settingSources`](/docs/en/agent-sdk/typescript#settingsource) in TypeScript) controls which filesystem-based settings the SDK loads. Pass an explicit list to opt in to specific sources, or pass an empty array to disable user, project, and local settings. This example loads both user-level and project-level settings by setting `settingSources` to `["user", "project"]`: @@ -73,7 +73,7 @@ This example loads both user-level and project-level settings by setting `settin When this runs, the assistant's response prints to stdout, followed by a final result line once the run completes. -Each source loads settings from a specific location, where `` is the working directory you pass via the `cwd` option, or the process's current directory if unset. For the full type definition, see [`SettingSource`](/en/agent-sdk/typescript#settingsource) (TypeScript) or [`SettingSource`](/en/agent-sdk/python#settingsource) (Python). +Each source loads settings from a specific location, where `` is the working directory you pass via the `cwd` option, or the process's current directory if unset. For the full type definition, see [`SettingSource`](/docs/en/agent-sdk/typescript#settingsource) (TypeScript) or [`SettingSource`](/docs/en/agent-sdk/python#settingsource) (Python). | Source | What it loads | Location | | :---------- | :---------------------------------------------------------------------------------------------- | :---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | @@ -91,13 +91,13 @@ The `cwd` option determines where the SDK looks for project-level inputs. CLAUDE | Input | Behavior | To disable | | :----------------------------------------------------------------- | :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | :--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| Managed policy settings | Endpoint-managed policy, such as an MDM plist, registry policy, or managed settings file, loads from the host. [Server-managed settings](/en/server-managed-settings) are fetched on an [eligible configuration](/en/server-managed-settings#platform-availability) when the session authenticates with an organization OAuth login or a directly configured API key | Endpoint policy: remove the managed settings file, plist, or registry policy from the host. Server-managed settings: controlled by your org admin; cannot be disabled from the SDK | +| Managed policy settings | Endpoint-managed policy, such as an MDM plist, registry policy, or managed settings file, loads from the host. [Server-managed settings](/docs/en/server-managed-settings) are fetched on an [eligible configuration](/docs/en/server-managed-settings#platform-availability) when the session authenticates with an organization OAuth login or a directly configured API key | Endpoint policy: remove the managed settings file, plist, or registry policy from the host. Server-managed settings: controlled by your org admin; cannot be disabled from the SDK | | `~/.claude.json` global config | Always read | Relocate with `CLAUDE_CONFIG_DIR` in `env` | | Auto memory at `~/.claude/projects//memory/` | Loaded into the system prompt at session start. The agent writes new memories there with the standard `Write` and `Edit` tools rather than a dedicated memory tool, so those tools must be enabled for the agent to save memories | Set `autoMemoryEnabled: false` in settings, or `CLAUDE_CODE_DISABLE_AUTO_MEMORY=1` in `env` | -| [claude.ai MCP connectors](/en/mcp#use-mcp-servers-from-claude-ai) | Loaded when the session authenticates with your claude.ai login. Not loaded when `CLAUDE_CODE_OAUTH_TOKEN` holds a token from [`claude setup-token`](/en/authentication#generate-a-long-lived-token), which can only make model requests. Passing `mcpServers: {}` does not suppress the connectors | Set `strictMcpConfig: true`, [`disableClaudeAiConnectors: true`](/en/mcp#disable-claude-ai-connectors) in settings, or `ENABLE_CLAUDEAI_MCP_SERVERS=false` in `env` | +| [claude.ai MCP connectors](/docs/en/mcp#use-mcp-servers-from-claude-ai) | Loaded when the session authenticates with your claude.ai login. Not loaded when `CLAUDE_CODE_OAUTH_TOKEN` holds a token from [`claude setup-token`](/docs/en/authentication#generate-a-long-lived-token), which can only make model requests. Passing `mcpServers: {}` does not suppress the connectors | Set `strictMcpConfig: true`, [`disableClaudeAiConnectors: true`](/docs/en/mcp#disable-claude-ai-connectors) in settings, or `ENABLE_CLAUDEAI_MCP_SERVERS=false` in `env` | - Do not rely on default `query()` options for multi-tenant isolation. Because the inputs above are read regardless of `settingSources`, an SDK process can pick up host-level configuration and per-directory memory. For multi-tenant deployments, run each tenant in its own filesystem and set `settingSources: []` plus `CLAUDE_CODE_DISABLE_AUTO_MEMORY=1` in `env`. [Server-managed settings](/en/server-managed-settings) are fetched when the process authenticates with an organization credential; filesystem isolation does not remove them. See [Secure deployment](/en/agent-sdk/secure-deployment). + Do not rely on default `query()` options for multi-tenant isolation. Because the inputs above are read regardless of `settingSources`, an SDK process can pick up host-level configuration and per-directory memory. For multi-tenant deployments, run each tenant in its own filesystem and set `settingSources: []` plus `CLAUDE_CODE_DISABLE_AUTO_MEMORY=1` in `env`. [Server-managed settings](/docs/en/server-managed-settings) are fetched when the process authenticates with an organization credential; filesystem isolation does not remove them. See [Secure deployment](/docs/en/agent-sdk/secure-deployment). ## Project instructions (CLAUDE.md and rules) @@ -119,10 +119,10 @@ The `cwd` option determines where the SDK looks for project-level inputs. CLAUDE All levels are additive: if both project and user CLAUDE.md files exist, the agent sees both. There is no hard precedence rule between levels; if instructions conflict, the outcome depends on how Claude interprets them. Write non-conflicting rules, or state precedence explicitly in the more specific file ("These project instructions override any conflicting user-level defaults"). - You can also inject context directly via `systemPrompt` without using CLAUDE.md files. See [Modify system prompts](/en/agent-sdk/modifying-system-prompts). Use CLAUDE.md when you want the same context shared between interactive Claude Code sessions and your SDK agents. + You can also inject context directly via `systemPrompt` without using CLAUDE.md files. See [Modify system prompts](/docs/en/agent-sdk/modifying-system-prompts). Use CLAUDE.md when you want the same context shared between interactive Claude Code sessions and your SDK agents. -For how to structure and organize CLAUDE.md content, see [Manage Claude's memory](/en/memory). +For how to structure and organize CLAUDE.md content, see [Manage Claude's memory](/docs/en/memory). ## Skills @@ -175,21 +175,21 @@ Skills are discovered from the filesystem through `settingSources`. When the `sk - Skills must be created as filesystem artifacts (`.claude/skills//SKILL.md`). The SDK does not have a programmatic API for registering skills. See [Agent Skills in the SDK](/en/agent-sdk/skills) for full details. + Skills must be created as filesystem artifacts (`.claude/skills//SKILL.md`). The SDK does not have a programmatic API for registering skills. See [Agent Skills in the SDK](/docs/en/agent-sdk/skills) for full details. -For more on creating and using skills, see [Agent Skills in the SDK](/en/agent-sdk/skills). +For more on creating and using skills, see [Agent Skills in the SDK](/docs/en/agent-sdk/skills). ## Hooks The SDK supports two ways to define hooks, and they run side by side: -* **Filesystem hooks:** shell commands defined in `settings.json`, loaded when `settingSources` includes the relevant source. These are the same hooks you'd configure for [interactive Claude Code sessions](/en/hooks-guide). -* **Programmatic hooks:** callback functions passed directly to `query()`. These run in your application process and can return structured decisions. See [Control execution with hooks](/en/agent-sdk/hooks). +* **Filesystem hooks:** shell commands defined in `settings.json`, loaded when `settingSources` includes the relevant source. These are the same hooks you'd configure for [interactive Claude Code sessions](/docs/en/hooks-guide). +* **Programmatic hooks:** callback functions passed directly to `query()`. These run in your application process and can return structured decisions. See [Control execution with hooks](/docs/en/agent-sdk/hooks). Both types execute during the same hook lifecycle. If you already have hooks in your project's `.claude/settings.json` and you set `settingSources: ["project"]`, those hooks run automatically in the SDK with no extra configuration. -Hook callbacks receive the tool input and return a decision dict. Returning `{}` means allow the tool to proceed. To block execution, return a `hookSpecificOutput` object with `permissionDecision: "deny"` and a `permissionDecisionReason`. The reason is sent to Claude as the tool result. The top-level `decision` and `reason` fields are deprecated for `PreToolUse`. See the [hooks guide](/en/agent-sdk/hooks) for the full callback signature and return types. +Hook callbacks receive the tool input and return a decision dict. Returning `{}` means allow the tool to proceed. To block execution, return a `hookSpecificOutput` object with `permissionDecision: "deny"` and a `permissionDecisionReason`. The reason is sent to Claude as the tool result. The top-level `decision` and `reason` fields are deprecated for `PreToolUse`. See the [hooks guide](/docs/en/agent-sdk/hooks) for the full callback signature and return types. ```python Python theme={null} @@ -282,10 +282,10 @@ Hook callbacks receive the tool input and return a decision dict. Returning `{}` | **Programmatic** (callbacks in `query()`) | Application-specific logic, structured decisions, and in-process integration. These also fire inside subagents. The hook input, the callback's first argument, carries `agent_id` and `agent_type` fields that identify which agent fired the hook. | - The TypeScript SDK supports additional hook events beyond Python, including `SessionStart`, `SessionEnd`, `TeammateIdle`, and `TaskCompleted`. See the [hooks guide](/en/agent-sdk/hooks) for the full event compatibility table. + The TypeScript SDK supports additional hook events beyond Python, including `SessionStart`, `SessionEnd`, `TeammateIdle`, and `TaskCompleted`. See the [hooks guide](/docs/en/agent-sdk/hooks) for the full event compatibility table. -For full details on programmatic hooks, see [Control execution with hooks](/en/agent-sdk/hooks). For filesystem hook syntax, see [Hooks](/en/hooks). +For full details on programmatic hooks, see [Control execution with hooks](/docs/en/agent-sdk/hooks). For filesystem hook syntax, see [Hooks](/docs/en/hooks). ## Choose the right feature @@ -293,25 +293,25 @@ The Agent SDK gives you access to several ways to extend your agent's behavior. | You want to... | Use | SDK surface | | :------------------------------------------------------------------------------------------------ | :-------------------------------------------- | :------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| Set project conventions your agent always follows | [CLAUDE.md](/en/memory) | `settingSources: ["project"]` loads it automatically | -| Give the agent reference material it loads when relevant | [Skills](/en/agent-sdk/skills) | `settingSources` + `skills` option | -| Run a reusable workflow (deploy, review, release) | [User-invocable skills](/en/agent-sdk/skills) | `settingSources` + `skills` option | -| Delegate an isolated subtask to a fresh context (research, review) | [Subagents](/en/agent-sdk/subagents) | `agents` parameter + `allowedTools: ["Agent"]` | -| Coordinate multiple Claude Code instances with shared task lists and direct inter-agent messaging | [Agent teams](/en/agent-teams) | Not directly configured via SDK options. Agent teams are a CLI feature where one session acts as the team lead, coordinating work across independent teammates | -| Run deterministic logic on tool calls (audit, block, transform) | [Hooks](/en/agent-sdk/hooks) | `hooks` parameter with callbacks, or shell scripts loaded via `settingSources` | -| Give Claude structured tool access to an external service | [MCP](/en/agent-sdk/mcp) | `mcpServers` parameter | +| Set project conventions your agent always follows | [CLAUDE.md](/docs/en/memory) | `settingSources: ["project"]` loads it automatically | +| Give the agent reference material it loads when relevant | [Skills](/docs/en/agent-sdk/skills) | `settingSources` + `skills` option | +| Run a reusable workflow (deploy, review, release) | [User-invocable skills](/docs/en/agent-sdk/skills) | `settingSources` + `skills` option | +| Delegate an isolated subtask to a fresh context (research, review) | [Subagents](/docs/en/agent-sdk/subagents) | `agents` parameter + `allowedTools: ["Agent"]` | +| Coordinate multiple Claude Code instances with shared task lists and direct inter-agent messaging | [Agent teams](/docs/en/agent-teams) | Not directly configured via SDK options. Agent teams are a CLI feature where one session acts as the team lead, coordinating work across independent teammates | +| Run deterministic logic on tool calls (audit, block, transform) | [Hooks](/docs/en/agent-sdk/hooks) | `hooks` parameter with callbacks, or shell scripts loaded via `settingSources` | +| Give Claude structured tool access to an external service | [MCP](/docs/en/agent-sdk/mcp) | `mcpServers` parameter | - **Subagents versus agent teams:** Subagents are ephemeral and isolated: fresh conversation, one task, summary returned to parent. Agent teams coordinate multiple independent Claude Code instances that share a task list and message each other directly. Agent teams are a CLI feature. See [What subagents inherit](/en/agent-sdk/subagents#what-subagents-inherit) and the [agent teams comparison](/en/agent-teams#compare-with-subagents) for details. + **Subagents versus agent teams:** Subagents are ephemeral and isolated: fresh conversation, one task, summary returned to parent. Agent teams coordinate multiple independent Claude Code instances that share a task list and message each other directly. Agent teams are a CLI feature. See [What subagents inherit](/docs/en/agent-sdk/subagents#what-subagents-inherit) and the [agent teams comparison](/docs/en/agent-teams#compare-with-subagents) for details. -Every feature you enable adds to your agent's context window. For per-feature costs and how these features layer together, see [Extend Claude Code](/en/features-overview#understand-context-costs). +Every feature you enable adds to your agent's context window. For per-feature costs and how these features layer together, see [Extend Claude Code](/docs/en/features-overview#understand-context-costs). ## Related resources -* [Extend Claude Code](/en/features-overview): Conceptual overview of all extension features, with comparison tables and context cost analysis -* [Skills in the SDK](/en/agent-sdk/skills): Full guide to using skills programmatically -* [Subagents](/en/agent-sdk/subagents): Define and invoke subagents for isolated subtasks -* [Hooks](/en/agent-sdk/hooks): Intercept and control agent behavior at key execution points -* [Permissions](/en/agent-sdk/permissions): Control tool access with modes, rules, and callbacks -* [System prompts](/en/agent-sdk/modifying-system-prompts): Inject context without CLAUDE.md files +* [Extend Claude Code](/docs/en/features-overview): Conceptual overview of all extension features, with comparison tables and context cost analysis +* [Skills in the SDK](/docs/en/agent-sdk/skills): Full guide to using skills programmatically +* [Subagents](/docs/en/agent-sdk/subagents): Define and invoke subagents for isolated subtasks +* [Hooks](/docs/en/agent-sdk/hooks): Intercept and control agent behavior at key execution points +* [Permissions](/docs/en/agent-sdk/permissions): Control tool access with modes, rules, and callbacks +* [System prompts](/docs/en/agent-sdk/modifying-system-prompts): Inject context without CLAUDE.md files diff --git a/content/en/docs/claude-code/agent-sdk/cost-tracking.md b/content/en/docs/claude-code/agent-sdk/cost-tracking.md index e0c21315a..4a32467d0 100644 --- a/content/en/docs/claude-code/agent-sdk/cost-tracking.md +++ b/content/en/docs/claude-code/agent-sdk/cost-tracking.md @@ -8,7 +8,7 @@ The Claude Agent SDK provides detailed token usage information for each interaction with Claude. This guide explains how to properly track usage and understand cost reporting, especially when dealing with parallel tool uses and multi-step conversations. -For complete API documentation, see the [TypeScript SDK reference](/en/agent-sdk/typescript) and [Python SDK reference](/en/agent-sdk/python). +For complete API documentation, see the [TypeScript SDK reference](/docs/en/agent-sdk/typescript) and [Python SDK reference](/docs/en/agent-sdk/python). The `total_cost_usd` and `costUSD` fields are client-side estimates, not authoritative billing data. The SDK computes them locally from a price table bundled at build time, so they can drift from what you are actually billed when: @@ -31,7 +31,7 @@ Both SDKs use the same underlying cost model and expose the same granularity. Th Cost tracking depends on understanding how the SDK scopes usage data: -* **`query()` call:** one invocation of the SDK's `query()` function. A single call can involve multiple steps (Claude responds, uses tools, gets results, responds again). Each call produces one [`result`](/en/agent-sdk/typescript#sdkresultmessage) message at the end. +* **`query()` call:** one invocation of the SDK's `query()` function. A single call can involve multiple steps (Claude responds, uses tools, gets results, responds again). Each call produces one [`result`](/docs/en/agent-sdk/typescript#sdkresultmessage) message at the end. * **Step:** a single request/response cycle within a `query()` call. Each step produces assistant messages with token usage. * **Session:** a series of `query()` calls linked by a session ID (using the `resume` option). Each `query()` call within a session reports its own cost independently. @@ -45,15 +45,15 @@ The following diagram shows the message stream from a single `query()` call, wit - When the `query()` call completes, the SDK emits a result message with `total_cost_usd` and cumulative `usage`. This is available in both TypeScript ([`SDKResultMessage`](/en/agent-sdk/typescript#sdkresultmessage)) and Python ([`ResultMessage`](/en/agent-sdk/python#resultmessage)). If you make multiple `query()` calls (for example, in a multi-turn session), each result only reflects the cost of that individual call. If you only need the estimated total, you can ignore the per-step usage and read this single value. + When the `query()` call completes, the SDK emits a result message with `total_cost_usd` and cumulative `usage`. This is available in both TypeScript ([`SDKResultMessage`](/docs/en/agent-sdk/typescript#sdkresultmessage)) and Python ([`ResultMessage`](/docs/en/agent-sdk/python#resultmessage)). If you make multiple `query()` calls (for example, in a multi-turn session), each result only reflects the cost of that individual call. If you only need the estimated total, you can ignore the per-step usage and read this single value. ## Get the total cost of a query -The result message ([TypeScript](/en/agent-sdk/typescript#sdkresultmessage), [Python](/en/agent-sdk/python#resultmessage)) marks the end of the agent loop for a `query()` call. It includes `total_cost_usd`, the cumulative estimated cost across all steps in that call. This works for both success and error results. If you use sessions to make multiple `query()` calls, each result only reflects the cost of that individual call. +The result message ([TypeScript](/docs/en/agent-sdk/typescript#sdkresultmessage), [Python](/docs/en/agent-sdk/python#resultmessage)) marks the end of the agent loop for a `query()` call. It includes `total_cost_usd`, the cumulative estimated cost across all steps in that call. This works for both success and error results. If you use sessions to make multiple `query()` calls, each result only reflects the cost of that individual call. -The three result-level fields differ in what they count when the agent spawns [subagents](/en/agent-sdk/subagents). Use `modelUsage`, or `model_usage` in Python, for whole-tree token accounting; the `usage` field undercounts as soon as nesting occurs. +The three result-level fields differ in what they count when the agent spawns [subagents](/docs/en/agent-sdk/subagents). Use `modelUsage`, or `model_usage` in Python, for whole-tree token accounting; the `usage` field undercounts as soon as nesting occurs. | Field | Subagent activity | | ---------------------------- | ------------------------------------------------------------------------------------------------- | @@ -106,7 +106,7 @@ The following examples iterate over the message stream from a `query()` call and ## Track per-step and per-model usage -The examples in this section use TypeScript field names. In Python, the equivalent fields are [`AssistantMessage.usage`](/en/agent-sdk/python#assistantmessage) and `AssistantMessage.message_id` for per-step usage, and [`ResultMessage.model_usage`](/en/agent-sdk/python#resultmessage) for per-model breakdowns. +The examples in this section use TypeScript field names. In Python, the equivalent fields are [`AssistantMessage.usage`](/docs/en/agent-sdk/python#assistantmessage) and `AssistantMessage.message_id` for per-step usage, and [`ResultMessage.model_usage`](/docs/en/agent-sdk/python#resultmessage) for per-model breakdowns. ### Track per-step usage @@ -151,7 +151,7 @@ console.log(`Output tokens: ${totalOutputTokens}`); ### Break down usage per model -The result message includes [`modelUsage`](/en/agent-sdk/typescript#modelusage), a map of model name to per-model token counts and cost. This is useful when you run multiple models (for example, Haiku for subagents and Opus for the main agent) and want to see where tokens are going. +The result message includes [`modelUsage`](/docs/en/agent-sdk/typescript#modelusage), a map of model name to per-model token counts and cost. This is useful when you run multiple models (for example, Haiku for subagents and Opus for the main agent) and want to see where tokens are going. The following example runs a query and prints the cost and token breakdown for each model used: @@ -274,15 +274,15 @@ The Agent SDK automatically uses [prompt caching](https://platform.claude.com/do * `cache_creation_input_tokens`: tokens used to create new cache entries (charged at a higher rate than standard input tokens). * `cache_read_input_tokens`: tokens read from existing cache entries (charged at a reduced rate). -Track these separately from `input_tokens` to understand caching savings. In TypeScript, these fields are typed on the [`Usage`](/en/agent-sdk/typescript#usage) object. In Python, they appear as keys in the [`ResultMessage.usage`](/en/agent-sdk/python#resultmessage) dict (for example, `message.usage.get("cache_read_input_tokens", 0)`). +Track these separately from `input_tokens` to understand caching savings. In TypeScript, these fields are typed on the [`Usage`](/docs/en/agent-sdk/typescript#usage) object. In Python, they appear as keys in the [`ResultMessage.usage`](/docs/en/agent-sdk/python#resultmessage) dict (for example, `message.usage.get("cache_read_input_tokens", 0)`). ### Extend the prompt cache TTL to one hour Cache entries written by the SDK use a 5-minute TTL by default when you authenticate with an API key or run on Amazon Bedrock, Google Cloud's Agent Platform, or Microsoft Foundry. If your workload runs many short sessions against the same system prompt and context with gaps longer than 5 minutes between them, the cache expires between sessions and each new session pays full input price. -To request a 1-hour TTL on cache writes, set the [`ENABLE_PROMPT_CACHING_1H`](/en/env-vars) environment variable. You can export it in your shell or container environment, or pass it through `options.env`. +To request a 1-hour TTL on cache writes, set the [`ENABLE_PROMPT_CACHING_1H`](/docs/en/env-vars) environment variable. You can export it in your shell or container environment, or pass it through `options.env`. -The following example enables 1-hour TTL for an agent running on Amazon Bedrock. Because it sets `CLAUDE_CODE_USE_BEDROCK`, it requires working AWS credentials for [Amazon Bedrock](/en/amazon-bedrock); without them the query fails. +The following example enables 1-hour TTL for an agent running on Amazon Bedrock. Because it sets `CLAUDE_CODE_USE_BEDROCK`, it requires working AWS credentials for [Amazon Bedrock](/docs/en/amazon-bedrock); without them the query fails. ```python Python theme={null} @@ -326,6 +326,6 @@ Cache writes with a 1-hour TTL are billed at a higher rate than 5-minute writes, ## Related documentation -* [TypeScript SDK Reference](/en/agent-sdk/typescript) - Complete API documentation -* [SDK Overview](/en/agent-sdk/overview) - Getting started with the SDK -* [SDK Permissions](/en/agent-sdk/permissions) - Managing tool permissions +* [TypeScript SDK Reference](/docs/en/agent-sdk/typescript) - Complete API documentation +* [SDK Overview](/docs/en/agent-sdk/overview) - Getting started with the SDK +* [SDK Permissions](/docs/en/agent-sdk/permissions) - Managing tool permissions diff --git a/content/en/docs/claude-code/agent-sdk/custom-tools.md b/content/en/docs/claude-code/agent-sdk/custom-tools.md index 1c659e6f6..47d55ef29 100644 --- a/content/en/docs/claude-code/agent-sdk/custom-tools.md +++ b/content/en/docs/claude-code/agent-sdk/custom-tools.md @@ -14,7 +14,7 @@ This guide covers how to define tools with input schemas and handlers, bundle th | If you want to... | Do this | | :------------------------------------------- | :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| Define a tool | Use [`@tool`](/en/agent-sdk/python#tool) (Python) or [`tool()`](/en/agent-sdk/typescript#tool) (TypeScript) with a name, description, schema, and handler. See [Create a custom tool](#create-a-custom-tool). | +| Define a tool | Use [`@tool`](/docs/en/agent-sdk/python#tool) (Python) or [`tool()`](/docs/en/agent-sdk/typescript#tool) (TypeScript) with a name, description, schema, and handler. See [Create a custom tool](#create-a-custom-tool). | | Register a tool with Claude | Wrap in `create_sdk_mcp_server` / `createSdkMcpServer` and pass to `mcpServers` in `query()`. See [Call a custom tool](#call-a-custom-tool). | | Pre-approve a tool | Add to your allowed tools. See [Configure allowed tools](#configure-allowed-tools). | | Remove a built-in tool from Claude's context | Pass a `tools` array listing only the built-ins you want. See [Configure allowed tools](#configure-allowed-tools). | @@ -22,11 +22,11 @@ This guide covers how to define tools with input schemas and handlers, bundle th | Control the error message Claude reads | Return `isError: true` to compose the message instead of surfacing the raw exception. See [Handle errors](#handle-errors). | | Return images or files | Use `image` or `resource` blocks in the content array. See [Return images and resources](#return-images-and-resources). | | Return a machine-readable JSON result | Set `structuredContent` on the result. See [Return structured data](#return-structured-data). | -| Scale to many tools | Use [tool search](/en/agent-sdk/tool-search) to load tools on demand. | +| Scale to many tools | Use [tool search](/docs/en/agent-sdk/tool-search) to load tools on demand. | ## Create a custom tool -A tool is defined by four parts, passed as arguments to the [`tool()`](/en/agent-sdk/typescript#tool) helper in TypeScript or the [`@tool`](/en/agent-sdk/python#tool) decorator in Python: +A tool is defined by four parts, passed as arguments to the [`tool()`](/docs/en/agent-sdk/typescript#tool) helper in TypeScript or the [`@tool`](/docs/en/agent-sdk/python#tool) decorator in Python: * **Name:** a unique identifier Claude uses to call the tool. * **Description:** what the tool does. Claude reads this to decide when to call it. @@ -36,7 +36,7 @@ A tool is defined by four parts, passed as arguments to the [`tool()`](/en/agent * `structuredContent` (optional): a JSON object holding the result as machine-readable data, returned alongside `content`. See [Return structured data](#return-structured-data). * `isError` (optional): set to `true` to signal a tool failure so Claude can react to it. See [Handle errors](#handle-errors). -After defining a tool, wrap it in a server with [`createSdkMcpServer`](/en/agent-sdk/typescript#createsdkmcpserver) (TypeScript) or [`create_sdk_mcp_server`](/en/agent-sdk/python#create_sdk_mcp_server) (Python). The server runs in-process inside your application, not as a separate process. +After defining a tool, wrap it in a server with [`createSdkMcpServer`](/docs/en/agent-sdk/typescript#createsdkmcpserver) (TypeScript) or [`create_sdk_mcp_server`](/docs/en/agent-sdk/python#create_sdk_mcp_server) (Python). The server runs in-process inside your application, not as a separate process. ### Weather tool example @@ -122,7 +122,7 @@ This example defines a `get_temperature` tool and wraps it in an MCP server. It ``` -See the [`tool()`](/en/agent-sdk/typescript#tool) TypeScript reference or the [`@tool`](/en/agent-sdk/python#tool) Python reference for full parameter details, including JSON Schema input formats and return value structure. +See the [`tool()`](/docs/en/agent-sdk/typescript#tool) TypeScript reference or the [`@tool`](/docs/en/agent-sdk/python#tool) Python reference for full parameter details, including JSON Schema input formats and return value structure. To make a parameter optional: in TypeScript, add `.default()` to the Zod field. In Python, the dict schema treats every key as required, so leave the parameter out of the schema, mention it in the description string, and read it with `args.get()` in the handler. The [`get_precipitation_chance` tool below](#add-more-tools) shows both patterns. @@ -176,11 +176,13 @@ These snippets reuse the `weatherServer` from the [example above](#weather-tool- ``` +Combine this snippet with the tool and server definitions from the [weather tool example](#weather-tool-example) in one file, then run it with `python weather.py` for Python or `npx tsx weather.ts` for TypeScript. Claude calls `get_temperature` and the script prints a one-line answer with the current temperature in San Francisco. + ### Add more tools A server holds as many tools as you list in its `tools` array. With more than one tool on a server, you can list each one in `allowedTools` individually or use the wildcard `mcp__weather__*` to cover every tool the server exposes. -The example below adds a second tool, `get_precipitation_chance`, to the `weatherServer` from the [weather tool example](#weather-tool-example) and rebuilds it with both tools in the array. +The example below defines a second tool, `get_precipitation_chance`, and replaces the `weatherServer` definition from the [weather tool example](#weather-tool-example) with one that lists both tools in the array. ```python Python theme={null} @@ -263,7 +265,7 @@ The example below adds a second tool, `get_precipitation_chance`, to the `weathe ``` -Every tool in this array consumes context window space on every turn. If you're defining dozens of tools, see [tool search](/en/agent-sdk/tool-search) to load them on demand instead. +[Tool search](/docs/en/agent-sdk/tool-search) is on by default and defers SDK MCP tools: Claude sees each tool's name in a compact list and loads its full schema on demand. With tool search disabled, every tool in this array consumes context window space on every turn. In TypeScript, pass `alwaysLoad: true` in the `extras` argument of [`tool()`](/docs/en/agent-sdk/typescript#tool) or in the options of [`createSdkMcpServer()`](/docs/en/agent-sdk/typescript#createsdkmcpserver) to keep a tool's full schema in the initial prompt. ### Add tool annotations @@ -311,7 +313,7 @@ This example adds `readOnlyHint` to the `get_temperature` tool from the [weather ``` -See `ToolAnnotations` in the [TypeScript](/en/agent-sdk/typescript#toolannotations) or [Python](/en/agent-sdk/python#toolannotations) reference. +See `ToolAnnotations` in the [TypeScript](/docs/en/agent-sdk/typescript#toolannotations) or [Python](/docs/en/agent-sdk/python#toolannotations) reference. ## Control tool access @@ -332,10 +334,10 @@ The `tools` option and the allowed/disallowed lists affect two layers: availabil | :------------------------ | :----------- | :-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `tools: ["Read", "Grep"]` | Availability | Only the listed built-ins are in Claude's context. Unlisted built-ins are removed. MCP tools are unaffected. | | `tools: []` | Availability | All built-ins are removed. Claude can only use your MCP tools. | -| allowed tools | Permission | Listed tools run without a permission prompt. Unlisted tools remain available; calls go through the [permission flow](/en/agent-sdk/permissions). | +| allowed tools | Permission | Listed tools run without a permission prompt. Unlisted tools remain available; calls go through the [permission flow](/docs/en/agent-sdk/permissions). | | disallowed tools | Both | A bare tool name such as `"Bash"` removes the tool from Claude's context, the same as omitting it from `tools`. A scoped rule such as `"Bash(rm *)"` leaves the tool in context and denies only matching calls. | -To remove a built-in entirely, omit it from `tools` or list its bare name in `disallowedTools` (Python: `disallowed_tools`); both keep the tool out of context so Claude never attempts it. A scoped `disallowedTools` rule blocks matching calls but leaves the tool visible, so Claude may waste a turn trying it. See [Configure permissions](/en/agent-sdk/permissions) for the full evaluation order. +To remove a built-in entirely, omit it from `tools` or list its bare name in `disallowedTools` (Python: `disallowed_tools`); both keep the tool out of context so Claude never attempts it. A scoped `disallowedTools` rule blocks matching calls but leaves the tool visible, so Claude may waste a turn trying it. See [Configure permissions](/docs/en/agent-sdk/permissions) for the full evaluation order. ## Handle errors @@ -355,6 +357,7 @@ The example below catches two kinds of failures inside the handler and composes import json import httpx from typing import Any + from claude_agent_sdk import tool from claude_agent_sdk import tool @@ -465,6 +468,7 @@ An image block carries the image bytes inline, encoded as base64. There is no UR ```python Python theme={null} import base64 import httpx + from claude_agent_sdk import tool from claude_agent_sdk import tool @@ -591,7 +595,7 @@ return { ``` - The Python `@tool` decorator forwards only `content` and `is_error` from the handler's return dict. To return `structuredContent` from Python, run a [standalone MCP server](/en/agent-sdk/mcp) instead of an in-process SDK server. + The Python `@tool` decorator forwards only `content` and `is_error` from the handler's return dict. To return `structuredContent` from Python, run a [standalone MCP server](/docs/en/agent-sdk/mcp) instead of an in-process SDK server. ## Example: unit converter @@ -762,6 +766,8 @@ It demonstrates two patterns: Once the server is defined, pass it to `query` the same way as the weather example. This example sends three different prompts in a loop to show the same tool handling different unit types. For each response, it inspects `AssistantMessage` objects (which contain the tool calls Claude made during that turn) and prints each `ToolUseBlock` before printing the final `ResultMessage` text. This lets you see when Claude is using the tool versus answering from its own knowledge. +Because [tool search](/docs/en/agent-sdk/tool-search) is on by default, the output may also include a `ToolSearch` call as Claude loads the deferred tool schema. + ```python Python theme={null} import asyncio @@ -849,13 +855,13 @@ Custom tools wrap async functions in a standard interface. You can mix the patte From here: -* If your server grows to dozens of tools, see [tool search](/en/agent-sdk/tool-search) to defer loading them until Claude needs them. -* To connect to external MCP servers (filesystem, GitHub, Slack) instead of building your own, see [Connect MCP servers](/en/agent-sdk/mcp). -* To control which tools run automatically versus requiring approval, see [Configure permissions](/en/agent-sdk/permissions). +* If your server grows to dozens of tools, see [tool search](/docs/en/agent-sdk/tool-search) to defer loading them until Claude needs them. +* To connect to external MCP servers (filesystem, GitHub, Slack) instead of building your own, see [Connect MCP servers](/docs/en/agent-sdk/mcp). +* To control which tools run automatically versus requiring approval, see [Configure permissions](/docs/en/agent-sdk/permissions). ## Related documentation -* [TypeScript SDK Reference](/en/agent-sdk/typescript) -* [Python SDK Reference](/en/agent-sdk/python) +* [TypeScript SDK Reference](/docs/en/agent-sdk/typescript) +* [Python SDK Reference](/docs/en/agent-sdk/python) * [MCP Documentation](https://modelcontextprotocol.io) -* [SDK Overview](/en/agent-sdk/overview) +* [SDK Overview](/docs/en/agent-sdk/overview) diff --git a/content/en/docs/claude-code/agent-sdk/file-checkpointing.md b/content/en/docs/claude-code/agent-sdk/file-checkpointing.md index a2befd723..a38c614b5 100644 --- a/content/en/docs/claude-code/agent-sdk/file-checkpointing.md +++ b/content/en/docs/claude-code/agent-sdk/file-checkpointing.md @@ -257,7 +257,7 @@ The following example shows the complete flow: enable checkpointing, capture the ``` - If you capture the session ID and checkpoint ID, you can also rewind from the CLI. This command requires the `claude` executable, which comes from [installing Claude Code](/en/setup) and is not installed by the SDK package. The SDK enables checkpointing for you, but when you run `claude -p` directly you must set the `CLAUDE_CODE_ENABLE_SDK_FILE_CHECKPOINTING` environment variable: + If you capture the session ID and checkpoint ID, you can also rewind from the CLI. This command requires the `claude` executable, which comes from [installing Claude Code](/docs/en/setup) and is not installed by the SDK package. The SDK enables checkpointing for you, but when you run `claude -p` directly you must set the `CLAUDE_CODE_ENABLE_SDK_FILE_CHECKPOINTING` environment variable: ```bash theme={null} CLAUDE_CODE_ENABLE_SDK_FILE_CHECKPOINTING=true claude -p --resume --rewind-files @@ -486,7 +486,7 @@ This pattern stores all checkpoint UUIDs in an array with metadata. After the se This complete example creates a small utility file, has the agent add documentation comments, shows you the changes, then asks if you want to rewind. -Before you begin, make sure you have the [Claude Agent SDK installed](/en/agent-sdk/quickstart). +Before you begin, make sure you have the [Claude Agent SDK installed](/docs/en/agent-sdk/quickstart). @@ -816,7 +816,7 @@ This error occurs when you call `rewindFiles()` or `rewind_files()` after you've ## Next steps -* **[Sessions](/en/agent-sdk/sessions)**: learn how to resume sessions, which is required for rewinding after the stream completes. Covers session IDs, resuming conversations, and session forking. -* **[Permissions](/en/agent-sdk/permissions)**: configure which tools Claude can use and how file modifications are approved. Useful if you want more control over when edits happen. -* **[TypeScript SDK reference](/en/agent-sdk/typescript)**: complete API reference including all options for `query()` and the `rewindFiles()` method. -* **[Python SDK reference](/en/agent-sdk/python)**: complete API reference including all options for `ClaudeAgentOptions` and the `rewind_files()` method. +* **[Sessions](/docs/en/agent-sdk/sessions)**: learn how to resume sessions, which is required for rewinding after the stream completes. Covers session IDs, resuming conversations, and session forking. +* **[Permissions](/docs/en/agent-sdk/permissions)**: configure which tools Claude can use and how file modifications are approved. Useful if you want more control over when edits happen. +* **[TypeScript SDK reference](/docs/en/agent-sdk/typescript)**: complete API reference including all options for `query()` and the `rewindFiles()` method. +* **[Python SDK reference](/docs/en/agent-sdk/python)**: complete API reference including all options for `ClaudeAgentOptions` and the `rewind_files()` method. diff --git a/content/en/docs/claude-code/agent-sdk/hooks.md b/content/en/docs/claude-code/agent-sdk/hooks.md index b7275a084..1ad176e86 100644 --- a/content/en/docs/claude-code/agent-sdk/hooks.md +++ b/content/en/docs/claude-code/agent-sdk/hooks.md @@ -24,7 +24,7 @@ This guide covers how hooks work and how to configure them, with examples for co - The SDK checks for hooks registered for that event type. This includes callback hooks you pass in `options.hooks` and shell command hooks from settings files when the corresponding [`settingSources`](/en/agent-sdk/typescript#settingsource) or [`setting_sources`](/en/agent-sdk/python#settingsource) entry is enabled, which it is for default `query()` options. + The SDK checks for hooks registered for that event type. This includes callback hooks you pass in `options.hooks` and shell command hooks from settings files when the corresponding [`settingSources`](/docs/en/agent-sdk/typescript#settingsource) or [`setting_sources`](/docs/en/agent-sdk/python#settingsource) entry is enabled, which it is for default `query()` options. @@ -153,7 +153,7 @@ The SDK provides hooks for different stages of agent execution. Some hooks are a | `PostToolUseFailure` | Yes | Yes | Tool execution failure | Handle or log tool errors | | `PostToolBatch` | No | Yes | A full batch of tool calls resolves, once per batch before the next model call | Inject conventions once for the whole batch | | `UserPromptSubmit` | Yes | Yes | User prompt submission | Inject additional context into prompts | -| [`UserPromptExpansion`](/en/hooks#userpromptexpansion) | No | Yes | A user-typed command, or an MCP prompt, expands into a prompt before it reaches Claude. Doesn't fire when Claude invokes a skill itself | Block a command from direct invocation or add context when a skill is typed | +| [`UserPromptExpansion`](/docs/en/hooks#userpromptexpansion) | No | Yes | A user-typed command, or an MCP prompt, expands into a prompt before it reaches Claude. Doesn't fire when Claude invokes a skill itself | Block a command from direct invocation or add context when a skill is typed | | `MessageDisplay` | No | Yes | An assistant message with text completes, once per message with the full message text | Redact or reformat the displayed text without changing the transcript | | `Stop` | Yes | Yes | Agent execution stop | Save session state before exit | | `StopFailure` | No | Yes | The turn ends with an API error instead of a normal stop | Log failures or send alerts | @@ -216,9 +216,9 @@ The `hooks` option is a dictionary in Python or an object in TypeScript, where: ### Matchers -Use matchers to filter when your callbacks fire. The `matcher` field matches against a different value depending on the hook event type. For example, tool-based hooks match against the tool name, while `Notification` hooks match against the notification type. See the [Claude Code hooks reference](/en/hooks#matcher-patterns) for the full list of matcher values for each event type. +Use matchers to filter when your callbacks fire. The `matcher` field matches against a different value depending on the hook event type. For example, tool-based hooks match against the tool name, while `Notification` hooks match against the notification type. See the [Claude Code hooks reference](/docs/en/hooks#matcher-patterns) for the full list of matcher values for each event type. -SDK matchers follow the same rules as [matchers in settings files](/en/hooks#matcher-patterns). A matcher containing only letters, digits, `_`, `-`, spaces, `,`, and `|` is compared as an exact string, with alternatives separated by `|` or `,` and optional surrounding whitespace, so `Write|Edit` and `Write, Edit` each match exactly those two tools and `code-reviewer` matches only that agent type. A matcher of `*`, an empty string, or omitting the matcher entirely matches every occurrence of the event. +SDK matchers follow the same rules as [matchers in settings files](/docs/en/hooks#matcher-patterns). A matcher containing only letters, digits, `_`, `-`, spaces, `,`, and `|` is compared as an exact string, with alternatives separated by `|` or `,` and optional surrounding whitespace, so `Write|Edit` and `Write, Edit` each match exactly those two tools and `code-reviewer` matches only that agent type. A matcher of `*`, an empty string, or omitting the matcher entirely matches every occurrence of the event. A matcher containing any other character is evaluated as an unanchored regular expression, so `^mcp__` matches every MCP tool and `Edit.*` matches both `Edit` and `NotebookEdit`. Wrap a regular expression in `^` and `$` when you need a whole-string match. @@ -226,11 +226,11 @@ A matcher like `mcp__memory` or `mcp__brave-search` contains only exact-match ch Hyphens in the exact-match set require a Claude Code runtime of v2.1.195 or later. On earlier versions a hyphenated name like `code-reviewer` is evaluated as an unanchored regular expression and must be anchored as `^code-reviewer$` to match exactly. -`StopFailure` and `FileChanged` use a narrower exact-match set of letters, digits, `_`, and `|` only. A hyphen, space, or comma in a matcher for those two events keeps it on the regular-expression path, and only `|` separates alternatives, so write `rate_limit|overloaded`, not `rate_limit, overloaded`. `FileChanged` additionally uses its matcher to build the watch list of literal filenames; see [FileChanged in the hooks reference](/en/hooks#filechanged). +`StopFailure` and `FileChanged` use a narrower exact-match set of letters, digits, `_`, and `|` only. A hyphen, space, or comma in a matcher for those two events keeps it on the regular-expression path, and only `|` separates alternatives, so write `rate_limit|overloaded`, not `rate_limit, overloaded`. `FileChanged` additionally uses its matcher to build the watch list of literal filenames; see [FileChanged in the hooks reference](/docs/en/hooks#filechanged). | Option | Type | Default | Description | | --------- | ---------------- | ----------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `matcher` | `string` | `undefined` | Pattern matched against the event's filter field, following the comparison rules above. For tool hooks, this is the tool name. Built-in tools include `Bash`, `Read`, `Write`, `Edit`, `Glob`, `Grep`, `WebFetch`, `Agent`, and others (see [Tool Input Types](/en/agent-sdk/typescript#tool-input-types) for the full list). MCP tools use the pattern `mcp____`. | +| `matcher` | `string` | `undefined` | Pattern matched against the event's filter field, following the comparison rules above. For tool hooks, this is the tool name. Built-in tools include `Bash`, `Read`, `Write`, `Edit`, `Glob`, `Grep`, `WebFetch`, `Agent`, and others (see [Tool Input Types](/docs/en/agent-sdk/typescript#tool-input-types) for the full list). MCP tools use the pattern `mcp____`. | | `hooks` | `HookCallback[]` | - | Required. Array of callback functions to execute when the pattern matches | | `timeout` | `number` | `undefined` | Timeout in seconds. When omitted, the per-event default applies: 10 minutes for most events, 30 seconds for `UserPromptSubmit`. A few events run with shorter limits, such as 10 seconds for `MessageDisplay` | @@ -239,7 +239,7 @@ Use the `matcher` pattern to target specific tools whenever possible. A matcher For tool-based hooks, matchers only filter by tool name, not by file paths or other arguments. To filter by file path, check `tool_input.file_path` inside your callback. - **Discovering tool names:** See [Tool Input Types](/en/agent-sdk/typescript#tool-input-types) for the full list of built-in tool names, or add a hook without a matcher to log all tool calls your session makes. + **Discovering tool names:** See [Tool Input Types](/docs/en/agent-sdk/typescript#tool-input-types) for the full list of built-in tool names, or add a hook without a matcher to log all tool calls your session makes. **MCP tool naming:** MCP tools always start with `mcp__` followed by the server name and action: `mcp____`. For example, if you configure a server named `playwright`, its tools are named `mcp__playwright__browser_screenshot`, `mcp__playwright__browser_click`, and so on. The server name comes from the key you use in the `mcpServers` configuration. @@ -250,7 +250,7 @@ For tool-based hooks, matchers only filter by tool name, not by file paths or ot Every hook callback receives three arguments: -* **Input data:** a typed object containing event details. Each hook type has its own input shape. For example, `PreToolUseHookInput` includes `tool_name` and `tool_input`, while `NotificationHookInput` includes `message`. See the full type definitions in the [TypeScript](/en/agent-sdk/typescript#hookinput) and [Python](/en/agent-sdk/python#hookinput) SDK references. +* **Input data:** a typed object containing event details. Each hook type has its own input shape. For example, `PreToolUseHookInput` includes `tool_name` and `tool_input`, while `NotificationHookInput` includes `message`. See the full type definitions in the [TypeScript](/docs/en/agent-sdk/typescript#hookinput) and [Python](/docs/en/agent-sdk/python#hookinput) SDK references. * All hook inputs share `session_id`, `cwd`, and `hook_event_name`. * `agent_id` and `agent_type` are populated when the hook fires inside a subagent. In TypeScript, these are on the base hook input and available to all hook types. In Python, they are optional fields on `PreToolUse`, `PostToolUse`, `PostToolUseFailure`, and `PermissionRequest`, and required fields on `SubagentStart` and `SubagentStop`. * **Tool use ID** (`str | None` / `string | undefined`): correlates `PreToolUse` and `PostToolUse` events for the same tool call. @@ -261,9 +261,9 @@ Every hook callback receives three arguments: Your callback returns an object with two categories of fields: * **Top-level fields** work the same on every event: `systemMessage` shows a message to the user, and `continue` (`continue_` in Python) determines whether the agent keeps running after this hook. -* **`hookSpecificOutput`** controls the current operation. The fields inside depend on the hook event type. For `PreToolUse` hooks, this is where you set `permissionDecision` (`"allow"`, `"deny"`, `"ask"`, or `"defer"`), `permissionDecisionReason`, and `updatedInput`. Returning `"defer"` ends the query so you can [resume it later](/en/hooks#defer-a-tool-call-for-later). For `PostToolUse` hooks, you can set `additionalContext` to append information to the tool result. To replace the tool's output before Claude sees it, set `updatedToolOutput`, which works for any tool in both SDKs. The older `updatedMCPToolOutput` field replaces MCP tool output only and is deprecated. +* **`hookSpecificOutput`** controls the current operation. The fields inside depend on the hook event type. For `PreToolUse` hooks, this is where you set `permissionDecision` (`"allow"`, `"deny"`, `"ask"`, or `"defer"`), `permissionDecisionReason`, and `updatedInput`. Returning `"defer"` ends the query so you can [resume it later](/docs/en/hooks#defer-a-tool-call-for-later). For `PostToolUse` hooks, you can set `additionalContext` to append information to the tool result. To replace the tool's output before Claude sees it, set `updatedToolOutput`, which works for any tool in both SDKs. The older `updatedMCPToolOutput` field replaces MCP tool output only and is deprecated. -Return `{}` to allow the operation without changes. SDK callback hooks use the same JSON output format as [Claude Code shell command hooks](/en/hooks#json-output), which documents every field and event-specific option. For the SDK type definitions, see the [TypeScript](/en/agent-sdk/typescript#synchookjsonoutput) and [Python](/en/agent-sdk/python#synchookjsonoutput) SDK references. +Return `{}` to allow the operation without changes. SDK callback hooks use the same JSON output format as [Claude Code shell command hooks](/docs/en/hooks#json-output), which documents every field and event-specific option. For the SDK type definitions, see the [TypeScript](/docs/en/agent-sdk/typescript#synchookjsonoutput) and [Python](/docs/en/agent-sdk/python#synchookjsonoutput) SDK references. When multiple hooks or permission rules apply, `deny` takes priority over `defer`, which takes priority over `ask`, which takes priority over `allow`. If any hook returns `deny`, the operation is blocked regardless of other hooks. @@ -524,7 +524,7 @@ Use multi-tool matchers to share one callback across related tools. This example ### Track subagent activity -Use `SubagentStop` hooks to monitor when subagents finish their work. See the full input type in the [TypeScript](/en/agent-sdk/typescript#hookinput) and [Python](/en/agent-sdk/python#hookinput) SDK references. This example logs a summary each time a subagent completes: +Use `SubagentStop` hooks to monitor when subagents finish their work. See the full input type in the [TypeScript](/docs/en/agent-sdk/typescript#hookinput) and [Python](/docs/en/agent-sdk/python#hookinput) SDK references. This example logs a summary each time a subagent completes: ```python Python theme={null} @@ -768,8 +768,8 @@ This example forwards every notification to a Slack channel. It requires a [Slac * Verify the hook event name is correct and case-sensitive (`PreToolUse`, not `preToolUse`) * Check that your matcher pattern matches the tool name exactly * Ensure the hook is under the correct event type in `options.hooks` -* For non-tool hooks that support matchers, like `Notification` and `SubagentStop`, matchers match against different fields, and `Stop` ignores matchers entirely (see [matcher patterns](/en/hooks#matcher-patterns)) -* Hooks may not fire when the agent hits the [`max_turns`](/en/agent-sdk/python#claudeagentoptions) limit because the session ends before hooks can execute +* For non-tool hooks that support matchers, like `Notification` and `SubagentStop`, matchers match against different fields, and `Stop` ignores matchers entirely (see [matcher patterns](/docs/en/hooks#matcher-patterns)) +* Hooks may not fire when the agent hits the [`max_turns`](/docs/en/agent-sdk/python#claudeagentoptions) limit because the session ends before hooks can execute ### Matcher not filtering as expected @@ -791,7 +791,7 @@ const myHook: HookCallback = async (input, toolUseID, { signal }) => { * Increase the `timeout` value in the `HookMatcher` configuration * Use the `AbortSignal` from the third callback argument to handle cancellation gracefully in TypeScript -{/* min-version: 2.1.208 */}A `UserPromptSubmit` or [`UserPromptExpansion`](/en/hooks#userpromptexpansion) callback that exceeds its timeout blocks that prompt with a timeout message and the session continues. Interrupting the query while a callback is pending cancels the pending tool call. Before v2.1.208, a callback timeout on those events ended the query with `error_during_execution`, and an interrupt during a pending `PreToolUse` callback could let the tool call proceed. +{/* min-version: 2.1.208 */}A `UserPromptSubmit` or [`UserPromptExpansion`](/docs/en/hooks#userpromptexpansion) callback that exceeds its timeout blocks that prompt with a timeout message and the session continues. Interrupting the query while a callback is pending cancels the pending tool call. Before v2.1.208, a callback timeout on those events ended the query with `error_during_execution`, and an interrupt during a pending `PreToolUse` callback could let the tool call proceed. {/* min-version: 2.1.210 */}A `PreToolUse` callback that exceeds its timeout blocks the tool call, and Claude receives an error result naming the timeout. If another `PreToolUse` hook returned an explicit deny, Claude receives that denial instead. Before v2.1.210, Claude Code reported the timeout to Claude as if the user had rejected the tool call, so an unattended session stopped and waited for input. @@ -821,7 +821,7 @@ const myHook: HookCallback = async (input, toolUseID, { signal }) => { ### Session hooks not available in Python -`SessionStart` and `SessionEnd` can be registered as SDK callback hooks in TypeScript, but aren't available in the Python SDK because its `HookEvent` type omits them. In Python, they are only available as [shell command hooks](/en/hooks#hook-events) defined in settings files such as `.claude/settings.json`. To load shell command hooks from your SDK application, include the appropriate setting source with [`setting_sources`](/en/agent-sdk/python#settingsource) or [`settingSources`](/en/agent-sdk/typescript#settingsource): +`SessionStart` and `SessionEnd` can be registered as SDK callback hooks in TypeScript, but aren't available in the Python SDK because its `HookEvent` type omits them. In Python, they are only available as [shell command hooks](/docs/en/hooks#hook-events) defined in settings files such as `.claude/settings.json`. To load shell command hooks from your SDK application, include the appropriate setting source with [`setting_sources`](/docs/en/agent-sdk/python#settingsource) or [`settingSources`](/docs/en/agent-sdk/typescript#settingsource): ```python Python theme={null} @@ -853,15 +853,15 @@ A `UserPromptSubmit` hook that spawns subagents can create infinite loops if tho ### systemMessage not appearing in output -The `systemMessage` field shows a message to the user, not the model. By default the SDK surfaces hook output in the message stream only for `SessionStart` and `Setup` hooks, so a message from any other hook event doesn't appear unless you set `includeHookEvents` (`include_hook_events` in Python). To pass context to the model instead, return [`additionalContext`](/en/hooks#add-context-for-claude). +The `systemMessage` field shows a message to the user, not the model. By default the SDK surfaces hook output in the message stream only for `SessionStart` and `Setup` hooks, so a message from any other hook event doesn't appear unless you set `includeHookEvents` (`include_hook_events` in Python). To pass context to the model instead, return [`additionalContext`](/docs/en/hooks#add-context-for-claude). If you need to surface hook decisions to your application reliably, log them separately or use a dedicated output channel. ## Related resources -* [Claude Code hooks reference](/en/hooks): full JSON input/output schemas, event documentation, and matcher patterns -* [Claude Code hooks guide](/en/hooks-guide): shell command hook examples and walkthroughs -* [TypeScript SDK reference](/en/agent-sdk/typescript): hook types, input/output definitions, and configuration options -* [Python SDK reference](/en/agent-sdk/python): hook types, input/output definitions, and configuration options -* [Permissions](/en/agent-sdk/permissions): control what your agent can do -* [Custom tools](/en/agent-sdk/custom-tools): build tools to extend agent capabilities +* [Claude Code hooks reference](/docs/en/hooks): full JSON input/output schemas, event documentation, and matcher patterns +* [Claude Code hooks guide](/docs/en/hooks-guide): shell command hook examples and walkthroughs +* [TypeScript SDK reference](/docs/en/agent-sdk/typescript): hook types, input/output definitions, and configuration options +* [Python SDK reference](/docs/en/agent-sdk/python): hook types, input/output definitions, and configuration options +* [Permissions](/docs/en/agent-sdk/permissions): control what your agent can do +* [Custom tools](/docs/en/agent-sdk/custom-tools): build tools to extend agent capabilities diff --git a/content/en/docs/claude-code/agent-sdk/hosting.md b/content/en/docs/claude-code/agent-sdk/hosting.md index c70404a65..2851397e2 100644 --- a/content/en/docs/claude-code/agent-sdk/hosting.md +++ b/content/en/docs/claude-code/agent-sdk/hosting.md @@ -13,7 +13,7 @@ This page covers self-hosting on your own infrastructure: understand [the subpro If you do not need infrastructure control, custom isolation, or your own data plane, consider [Managed Agents](https://platform.claude.com/docs/en/managed-agents/overview) instead: a hosted REST API where Anthropic runs the agent and the sandbox, so your application sends events and streams back results with no hosting infrastructure to operate. - For security hardening beyond basic sandboxing, including network controls, credential management, and isolation options, see [Secure Deployment](/en/agent-sdk/secure-deployment). + For security hardening beyond basic sandboxing, including network controls, credential management, and isolation options, see [Secure Deployment](/docs/en/agent-sdk/secure-deployment). ## The subprocess model @@ -44,9 +44,9 @@ Three kinds of agent state live on the container's filesystem by default. None o | `CLAUDE.md` memory files | `~/.claude/CLAUDE.md` for the user tier and the session's working directory for the project tier | | Working-directory artifacts | The session's working directory | -To persist transcripts across hosts, configure a [`SessionStore` adapter](/en/agent-sdk/session-storage). Memory files and other working-directory artifacts need their own storage strategy, such as a mounted volume or an object-store sync. +To persist transcripts across hosts, configure a [`SessionStore` adapter](/docs/en/agent-sdk/session-storage). Memory files and other working-directory artifacts need their own storage strategy, such as a mounted volume or an object-store sync. -For how sessions, resumption, and forking work at the API level, see [Sessions](/en/agent-sdk/sessions). +For how sessions, resumption, and forking work at the API level, see [Sessions](/docs/en/agent-sdk/sessions). ## Choose a session pattern @@ -75,11 +75,11 @@ Run persistent container instances, often hosting multiple SDK processes per con Example workloads include an email agent that triages and responds to incoming mail, a site builder that hosts a per-user editable site through container ports, and a chat bot that handles continuous traffic from a platform like Slack. -The container exposes an HTTP or WebSocket endpoint and maps each active session to a long-lived query and the subprocess behind it. In TypeScript, use [`streamInput()`](/en/agent-sdk/typescript#query-object) to add turns to an active session and [`startup()`](/en/agent-sdk/typescript#startup) to pre-warm subprocesses ahead of incoming traffic. In Python, use [`ClaudeSDKClient`](/en/agent-sdk/python#claudesdkclient) to hold a session open across turns. Size the container so it can hold the maximum number of concurrent sessions in memory. +The container exposes an HTTP or WebSocket endpoint and maps each active session to a long-lived query and the subprocess behind it. In TypeScript, use [`streamInput()`](/docs/en/agent-sdk/typescript#query-object) to add turns to an active session and [`startup()`](/docs/en/agent-sdk/typescript#startup) to pre-warm subprocesses ahead of incoming traffic. In Python, use [`ClaudeSDKClient`](/docs/en/agent-sdk/python#claudesdkclient) to hold a session open across turns. Size the container so it can hold the maximum number of concurrent sessions in memory. ### Hybrid sessions -Ephemeral containers that hydrate from a [`SessionStore`](/en/agent-sdk/session-storage) on startup and persist updates back. Best for sessions that span many interactions but sit idle between them. The container spins down during idle periods and spins back up when the user returns. +Ephemeral containers that hydrate from a [`SessionStore`](/docs/en/agent-sdk/session-storage) on startup and persist updates back. Best for sessions that span many interactions but sit idle between them. The container spins down during idle periods and spins back up when the user returns. Example workloads include a personal project manager with intermittent check-ins, deep research that pauses and resumes over hours, and a customer support agent that loads ticket history across interactions. @@ -127,7 +127,7 @@ The pattern hinges on resuming a session by ID with a shared store attached: ``` -See [Session storage](/en/agent-sdk/session-storage) for the full `SessionStore` interface and reference adapters. +See [Session storage](/docs/en/agent-sdk/session-storage) for the full `SessionStore` interface and reference adapters. ### Multi-agent container @@ -158,7 +158,7 @@ Providers to evaluate: * [Fly Machines](https://fly.io/docs/machines/) * [Vercel Sandbox](https://vercel.com/docs/functions/sandbox) -For self-hosted options such as Docker, gVisor, and Firecracker, and detailed isolation configuration, see [Isolation Technologies](/en/agent-sdk/secure-deployment#isolation-technologies). +For self-hosted options such as Docker, gVisor, and Firecracker, and detailed isolation configuration, see [Isolation Technologies](/docs/en/agent-sdk/secure-deployment#isolation-technologies). ### Runtime dependencies @@ -175,7 +175,7 @@ The bundled binary is pinned to the SDK package version, so updating the SDK is ### Network -The SDK needs outbound HTTPS to `api.anthropic.com`, or to your provider's regional endpoint when running on Amazon Bedrock or Google Cloud's Agent Platform. If your agents use [MCP servers](/en/agent-sdk/mcp) or external tools, they need outbound access to those endpoints as well. For production, route outbound traffic through an egress proxy that enforces domain allowlists, injects credentials, and logs requests. See [Secure Deployment](/en/agent-sdk/secure-deployment) for the full pattern. +The SDK needs outbound HTTPS to `api.anthropic.com`, or to your provider's regional endpoint when running on Amazon Bedrock or Google Cloud's Agent Platform. If your agents use [MCP servers](/docs/en/agent-sdk/mcp) or external tools, they need outbound access to those endpoints as well. For production, route outbound traffic through an egress proxy that enforces domain allowlists, injects credentials, and logs requests. See [Secure Deployment](/docs/en/agent-sdk/secure-deployment) for the full pattern. For inbound traffic, expose an HTTP or WebSocket port on the container. Your application handles client requests on that port and calls the SDK internally; the subprocess itself does not listen on the network. @@ -185,7 +185,7 @@ Work through these decisions before shipping a self-hosted agent. ### Session and state persistence -Default local disk is lost on restart, scale-down, or a move to a different node. For any session a user expects to resume, mirror the transcript to durable storage with a [`SessionStore` adapter](/en/agent-sdk/session-storage). See [Reference implementations](/en/agent-sdk/session-storage#reference-implementations) for S3, Redis, and Postgres adapters and a conformance suite for your own. +Default local disk is lost on restart, scale-down, or a move to a different node. For any session a user expects to resume, mirror the transcript to durable storage with a [`SessionStore` adapter](/docs/en/agent-sdk/session-storage). See [Reference implementations](/docs/en/agent-sdk/session-storage#reference-implementations) for S3, Redis, and Postgres adapters and a conformance suite for your own. Three things to know about how `SessionStore` behaves: @@ -209,13 +209,13 @@ OTEL_EXPORTER_OTLP_PROTOCOL=http/protobuf OTEL_EXPORTER_OTLP_ENDPOINT=http://collector.example.com:4318 ``` -Prompt text and tool inputs are not included in exports by default. See [Control sensitive data in exports](/en/agent-sdk/observability#control-sensitive-data-in-exports) for the opt-in flags, and [Observability](/en/agent-sdk/observability) for the full signal catalog. +Prompt text and tool inputs are not included in exports by default. See [Control sensitive data in exports](/docs/en/agent-sdk/observability#control-sensitive-data-in-exports) for the opt-in flags, and [Observability](/docs/en/agent-sdk/observability) for the full signal catalog. ### Auth and secrets Three auth concerns matter at hosting time: -* **Anthropic API**: the subprocess reads `ANTHROPIC_API_KEY` from its environment. Supply it from your secret manager, or set `ANTHROPIC_BASE_URL` to route model calls through a proxy that injects the key outside the container. See [Credential management](/en/agent-sdk/secure-deployment#credential-management) for the proxy pattern and the [SDK overview](/en/agent-sdk/overview#get-started) for supported authentication methods. +* **Anthropic API**: the subprocess reads `ANTHROPIC_API_KEY` from its environment. Supply it from your secret manager, or set `ANTHROPIC_BASE_URL` to route model calls through a proxy that injects the key outside the container. See [Credential management](/docs/en/agent-sdk/secure-deployment#credential-management) for the proxy pattern and the [SDK overview](/docs/en/agent-sdk/overview#get-started) for supported authentication methods. * **Inbound**: put authentication at a gateway in front of the agent container. The agent should receive pre-authenticated requests and should not be the component that validates user tokens. * **Outbound tools**: keep tool credentials out of the agent environment. Route outbound calls through a proxy that injects API keys after the request leaves the container. The agent makes the call; the proxy adds the credential. @@ -233,11 +233,11 @@ Measure the per-session ceiling by running a representative session to your targ Horizontal-scale routing depends on your pattern. For long-running sessions, where containers hold many sessions, run a pool of containers behind a load balancer and pin each session to one container using consistent hashing on `sessionId`. A pinned session keeps hitting the same container, and therefore the same running subprocess, until it is evicted or the container restarts. -Large fanouts of concurrent [subagents](/en/agent-sdk/subagents) from a single session can hit API rate limits. Break the work into smaller batches rather than issuing one wide dispatch. +Large fanouts of concurrent [subagents](/docs/en/agent-sdk/subagents) from a single session can hit API rate limits. Break the work into smaller batches rather than issuing one wide dispatch. ### Cost -Anthropic token cost typically dominates container infrastructure cost by an order of magnitude or more. A minimally provisioned container runs roughly \$0.05 per hour, while a single long agent session can spend dollars in tokens. See [Cost tracking](/en/agent-sdk/cost-tracking) for per-session token accounting. +Anthropic token cost typically dominates container infrastructure cost by an order of magnitude or more. A minimally provisioned container runs roughly \$0.05 per hour, while a single long agent session can spend dollars in tokens. See [Cost tracking](/docs/en/agent-sdk/cost-tracking) for per-session token accounting. ### Multi-tenant isolation @@ -246,7 +246,7 @@ Default SDK behavior reads settings and `CLAUDE.md` memory files from the filesy To isolate tenants inside a shared container: * Pass `settingSources: []` in TypeScript or `setting_sources=[]` in Python so no filesystem settings load. -* Set `CLAUDE_CODE_DISABLE_AUTO_MEMORY=1` in `env`. [Auto memory](/en/memory#auto-memory) at `~/.claude/projects//memory/` loads into the system prompt regardless of `settingSources`. See [What settingSources does not control](/en/agent-sdk/claude-code-features#what-settingsources-does-not-control) for the other inputs that load unconditionally. +* Set `CLAUDE_CODE_DISABLE_AUTO_MEMORY=1` in `env`. [Auto memory](/docs/en/memory#auto-memory) at `~/.claude/projects//memory/` loads into the system prompt regardless of `settingSources`. See [What settingSources does not control](/docs/en/agent-sdk/claude-code-features#what-settingsources-does-not-control) for the other inputs that load unconditionally. * Point `CLAUDE_CONFIG_DIR` at a per-tenant directory so tenants do not share the `~/.claude.json` global config. * Use a per-tenant working directory. Pass `cwd` explicitly on every `query()` call. * Apply per-tenant egress rules at your proxy, such as distinct outbound IPs, credentials, or domain allowlists, so a compromised tenant cannot exfiltrate data via another tenant's outbound policy. @@ -305,7 +305,7 @@ The example below applies the four SDK-level options together. Construct `tenant ``` -For per-tenant network controls, see [Secure Deployment](/en/agent-sdk/secure-deployment). +For per-tenant network controls, see [Secure Deployment](/docs/en/agent-sdk/secure-deployment). ## Known limitations @@ -316,12 +316,12 @@ Plan around these in your deployment design. | No top-level session timeout | A session does not time out on its own. Set `maxTurns` in `Options` to bound how many tool-use round trips the agent takes before stopping. | | Memory growth over long sessions | Cap session length or recycle subprocesses periodically. See [Scaling and concurrency](#scaling-and-concurrency). | | Large parallel-subagent fanouts can hit rate limits | Break work into smaller batches rather than issuing one wide dispatch. | -| No per-subagent wall-clock deadline | Cap each [subagent](/en/agent-sdk/subagents) with `maxTurns` in its `AgentDefinition`. For background subagents only, `CLAUDE_ASYNC_AGENT_STALL_TIMEOUT_MS` sets a stall watchdog that fires when a `run_in_background` subagent stops producing output; it is not a total-runtime deadline. | +| No per-subagent wall-clock deadline | Cap each [subagent](/docs/en/agent-sdk/subagents) with `maxTurns` in its `AgentDefinition`. For background subagents only, `CLAUDE_ASYNC_AGENT_STALL_TIMEOUT_MS` sets a stall watchdog that fires when a `run_in_background` subagent stops producing output; it is not a total-runtime deadline. | ## Next steps * [Hosting cookbook](https://github.com/anthropics/claude-cookbooks/blob/main/claude_agent_sdk/07_Hosting_the_agent.ipynb): notebook walkthrough with [deployable code](https://github.com/anthropics/claude-cookbooks/tree/main/claude_agent_sdk/hosting) for Docker, Modal, and Kubernetes. -* [Session storage](/en/agent-sdk/session-storage): persist transcripts across hosts with a `SessionStore` adapter. -* [Observability](/en/agent-sdk/observability): export OTEL traces, metrics, and logs to your collector. -* [Secure deployment](/en/agent-sdk/secure-deployment): network controls, credential management, and isolation hardening. -* [Cost tracking](/en/agent-sdk/cost-tracking): per-session token and cost accounting. +* [Session storage](/docs/en/agent-sdk/session-storage): persist transcripts across hosts with a `SessionStore` adapter. +* [Observability](/docs/en/agent-sdk/observability): export OTEL traces, metrics, and logs to your collector. +* [Secure deployment](/docs/en/agent-sdk/secure-deployment): network controls, credential management, and isolation hardening. +* [Cost tracking](/docs/en/agent-sdk/cost-tracking): per-session token and cost accounting. diff --git a/content/en/docs/claude-code/agent-sdk/mcp.md b/content/en/docs/claude-code/agent-sdk/mcp.md index 64b2faed8..9995cec65 100644 --- a/content/en/docs/claude-code/agent-sdk/mcp.md +++ b/content/en/docs/claude-code/agent-sdk/mcp.md @@ -11,7 +11,7 @@ The [Model Context Protocol (MCP)](https://modelcontextprotocol.io/docs/getting- MCP servers can run as local processes, connect over HTTP, or execute directly within your SDK application. - This page covers MCP configuration for the Agent SDK. To add MCP servers to the Claude Code CLI so they load in every project, see [MCP installation scopes](/en/mcp#mcp-installation-scopes). + This page covers MCP configuration for the Agent SDK. To add MCP servers to the Claude Code CLI so they load in every project, see [MCP installation scopes](/docs/en/mcp#mcp-installation-scopes). ## Quickstart @@ -148,9 +148,9 @@ Create a `.mcp.json` file at your project root. The file is picked up when the ` Servers you pass in `options.mcpServers` start connecting as soon as the query starts. Connection is non-blocking by default: the first turn begins without waiting, and each server's tools become available once its connection completes. {/* min-version: 2.1.142 */}Before Claude Code v2.1.142, startup blocked on the connection batch for up to 5 seconds. -To restore a bounded startup wait for every server, set the [`MCP_CONNECTION_NONBLOCKING`](/en/env-vars) environment variable to `0`. The wait is capped at 5 seconds by [`MCP_CONNECT_TIMEOUT_MS`](/en/env-vars), and servers still pending at that deadline keep connecting in the background. +To restore a bounded startup wait for every server, set the [`MCP_CONNECTION_NONBLOCKING`](/docs/en/env-vars) environment variable to `0`. The wait is capped at 5 seconds by [`MCP_CONNECT_TIMEOUT_MS`](/docs/en/env-vars), and servers still pending at that deadline keep connecting in the background. -To make one server's tools available before the first turn, set `alwaysLoad: true` on its config. Startup then waits for that server to connect, capped at the same 5-second startup deadline, while other servers keep connecting in the background. The `alwaysLoad` field requires Claude Code v2.1.121 or later. See [Exempt a server from deferral](/en/mcp#exempt-a-server-from-deferral) for the `alwaysLoad` field's effect on tool search. +To make one server's tools available before the first turn, set `alwaysLoad: true` on its config. Startup then waits for that server to connect, capped at the same 5-second startup deadline, while other servers keep connecting in the background. The `alwaysLoad` field requires Claude Code v2.1.121 or later. See [Exempt a server from deferral](/docs/en/mcp#exempt-a-server-from-deferral) for the `alwaysLoad` field's effect on tool search. The `system` message with subtype `init` reports each server's status at the moment it's emitted. A server that's still connecting has status `pending`. Check for status `failed` or `needs-auth` when you want to detect servers that won't be usable, rather than treating every status other than `connected` as a failure; see [Error handling](#error-handling) for the full status check. @@ -199,14 +199,14 @@ Use `allowedTools` to auto-approve specific MCP tools so Claude can use them wit Wildcards (`*`) let you allow all tools from a server without listing each one individually. - **Prefer `allowedTools` over permission modes for MCP access.** `permissionMode: "acceptEdits"` does not auto-approve MCP tools (only file edits and filesystem Bash commands). `permissionMode: "bypassPermissions"` does auto-approve MCP tools but also disables most other safety prompts, which is broader than necessary; see [How permissions are evaluated](/en/agent-sdk/permissions#how-permissions-are-evaluated) for the prompts that remain. A wildcard in `allowedTools` grants exactly the MCP server you want and nothing more. See [Permission modes](/en/agent-sdk/permissions#permission-modes) for a full comparison. + **Prefer `allowedTools` over permission modes for MCP access.** `permissionMode: "acceptEdits"` does not auto-approve MCP tools (only file edits and filesystem Bash commands). `permissionMode: "bypassPermissions"` does auto-approve MCP tools but also disables most other safety prompts, which is broader than necessary; see [How permissions are evaluated](/docs/en/agent-sdk/permissions#how-permissions-are-evaluated) for the prompts that remain. A wildcard in `allowedTools` grants exactly the MCP server you want and nothing more. See [Permission modes](/docs/en/agent-sdk/permissions#permission-modes) for a full comparison. ### Discover available tools To see what tools an MCP server provides, check the server's documentation or inspect the `tools` array in the `system` init message. MCP tool names start with `mcp__`. -MCP servers connect in the background by default, so the init message arrives before they finish: the `tools` array lists only built-in tools and `mcp_servers` shows a `pending` status for each server. Set the [`MCP_CONNECTION_NONBLOCKING`](/en/env-vars) environment variable to `0` to wait up to 5 seconds for servers to connect before the init message is sent; servers that connect in time list their `mcp__` tools there, and slower ones keep connecting in the background: +MCP servers connect in the background by default, so the init message arrives before they finish: the `tools` array lists only built-in tools and `mcp_servers` shows a `pending` status for each server. Set the [`MCP_CONNECTION_NONBLOCKING`](/docs/en/env-vars) environment variable to `0` to wait up to 5 seconds for servers to connect before the init message is sent; servers that connect in time list their `mcp__` tools there, and slower ones keep connecting in the background: ```bash theme={null} export MCP_CONNECTION_NONBLOCKING=0 @@ -376,15 +376,15 @@ For the streamable HTTP transport, use `"type": "http"` instead. In `.mcp.json` ### SDK MCP servers -Define custom tools directly in your application code instead of running a separate server process. See the [custom tools guide](/en/agent-sdk/custom-tools) for implementation details. +Define custom tools directly in your application code instead of running a separate server process. See the [custom tools guide](/docs/en/agent-sdk/custom-tools) for implementation details. -{/* min-version: 2.1.210 */}An SDK MCP server registered by an [`initialize` control request](/en/agent-sdk/typescript#sdkcontrolinitializeresponse) begins connecting as soon as Claude Code processes the request. +{/* min-version: 2.1.210 */}An SDK MCP server registered by an [`initialize` control request](/docs/en/agent-sdk/typescript#sdkcontrolinitializeresponse) begins connecting as soon as Claude Code processes the request. ## MCP tool search When you have many MCP tools configured, tool definitions can consume a significant portion of your context window. Tool search solves this by withholding tool definitions from context and loading only the ones Claude needs for each turn. -Tool search is enabled by default. See [Tool search](/en/agent-sdk/tool-search) for configuration options, best practices, and using tool search with custom SDK tools. +Tool search is enabled by default. See [Tool search](/docs/en/agent-sdk/tool-search) for configuration options, best practices, and using tool search with custom SDK tools. ## Authentication @@ -510,7 +510,7 @@ For a complete working example of a remote server authenticated with headers, se ### OAuth2 authentication -The [MCP specification supports OAuth 2.1](https://modelcontextprotocol.io/specification/2025-03-26/basic/authorization) for authorization. The SDK doesn't open a browser or run an interactive OAuth flow. When a configured server returns an authorization challenge and no stored token is available, the agent run continues without that server's tools, and the server reports status `needs-auth`. Because servers connect in the background by default, the `mcp_servers` array of the [system init message](/en/agent-sdk/typescript#sdksystemmessage) may still show `pending` for that server. To confirm whether a server needs credentials, poll `mcpServerStatus()` in the TypeScript SDK or [`get_mcp_status()`](/en/agent-sdk/python#methods) in Python, or set `MCP_CONNECTION_NONBLOCKING=0` to wait for connections before the init message. +The [MCP specification supports OAuth 2.1](https://modelcontextprotocol.io/specification/2025-03-26/basic/authorization) for authorization. The SDK doesn't open a browser or run an interactive OAuth flow. When a configured server returns an authorization challenge and no stored token is available, the agent run continues without that server's tools, and the server reports status `needs-auth`. Because servers connect in the background by default, the `mcp_servers` array of the [system init message](/docs/en/agent-sdk/typescript#sdksystemmessage) may still show `pending` for that server. To confirm whether a server needs credentials, poll `mcpServerStatus()` in the TypeScript SDK or [`get_mcp_status()`](/docs/en/agent-sdk/python#methods) in Python, or set `MCP_CONNECTION_NONBLOCKING=0` to wait for connections before the init message. To supply credentials, complete the OAuth flow in your own application and pass the resulting access token in the server's `headers`: @@ -839,7 +839,7 @@ Check the `init` message to see which servers failed to connect: ``` -A `"pending"` status means the server is still connecting, not that it failed. To get updated statuses later in the session, call the query's `mcpServerStatus()` method in the TypeScript SDK, or [`ClaudeSDKClient.get_mcp_status()`](/en/agent-sdk/python#methods) in Python. +A `"pending"` status means the server is still connecting, not that it failed. To get updated statuses later in the session, call the query's `mcpServerStatus()` method in the TypeScript SDK, or [`ClaudeSDKClient.get_mcp_status()`](/docs/en/agent-sdk/python#methods) in Python. Common causes: @@ -876,7 +876,7 @@ If Claude sees tools but doesn't use them, check that you've granted permission ### Connection timeouts -MCP server connections time out after 30 seconds by default. If your server takes longer to start, the connection fails. Raise the limit with the [`MCP_TIMEOUT`](/en/env-vars) environment variable, in milliseconds. For servers that need more startup time, also consider: +MCP server connections time out after 30 seconds by default. If your server takes longer to start, the connection fails. Raise the limit with the [`MCP_TIMEOUT`](/docs/en/env-vars) environment variable, in milliseconds. For servers that need more startup time, also consider: * Using a lighter-weight server if available * Pre-warming the server before starting your agent @@ -884,13 +884,13 @@ MCP server connections time out after 30 seconds by default. If your server take ### Tool output exceeds maximum allowed tokens -The SDK applies the same MCP output limit as Claude Code. When a tool result is larger than 25,000 tokens, the full output is saved to a file and the tool result is replaced with an error message that names the file path, so the agent can read the output back in portions. Raise the limit with the [`MAX_MCP_OUTPUT_TOKENS`](/en/env-vars) environment variable. See [MCP output limits and warnings](/en/mcp#mcp-output-limits-and-warnings) for the full behavior, including how a server can declare a higher per-tool limit. +The SDK applies the same MCP output limit as Claude Code. When a tool result is larger than 25,000 tokens, the full output is saved to a file and the tool result is replaced with an error message that names the file path, so the agent can read the output back in portions. Raise the limit with the [`MAX_MCP_OUTPUT_TOKENS`](/docs/en/env-vars) environment variable. See [MCP output limits and warnings](/docs/en/mcp#mcp-output-limits-and-warnings) for the full behavior, including how a server can declare a higher per-tool limit. ## Related resources -* **[Custom tools guide](/en/agent-sdk/custom-tools)**: Build your own MCP server that runs in-process with your SDK application -* **[Permissions](/en/agent-sdk/permissions)**: Control which MCP tools your agent can use with `allowedTools` and `disallowedTools` -* **[MCP output limits and warnings](/en/mcp#mcp-output-limits-and-warnings)**: How the SDK handles tool results that exceed `MAX_MCP_OUTPUT_TOKENS`, including the persist-to-disk fallback and the `anthropic/maxResultSizeChars` per-tool annotation -* **[TypeScript SDK reference](/en/agent-sdk/typescript)**: Full API reference including MCP configuration options -* **[Python SDK reference](/en/agent-sdk/python)**: Full API reference including MCP configuration options +* **[Custom tools guide](/docs/en/agent-sdk/custom-tools)**: Build your own MCP server that runs in-process with your SDK application +* **[Permissions](/docs/en/agent-sdk/permissions)**: Control which MCP tools your agent can use with `allowedTools` and `disallowedTools` +* **[MCP output limits and warnings](/docs/en/mcp#mcp-output-limits-and-warnings)**: How the SDK handles tool results that exceed `MAX_MCP_OUTPUT_TOKENS`, including the persist-to-disk fallback and the `anthropic/maxResultSizeChars` per-tool annotation +* **[TypeScript SDK reference](/docs/en/agent-sdk/typescript)**: Full API reference including MCP configuration options +* **[Python SDK reference](/docs/en/agent-sdk/python)**: Full API reference including MCP configuration options * **[MCP server directory](https://github.com/modelcontextprotocol/servers)**: Browse available MCP servers for databases, APIs, and more diff --git a/content/en/docs/claude-code/agent-sdk/migration-guide.md b/content/en/docs/claude-code/agent-sdk/migration-guide.md index a98b4cd05..13bfdb2ce 100644 --- a/content/en/docs/claude-code/agent-sdk/migration-guide.md +++ b/content/en/docs/claude-code/agent-sdk/migration-guide.md @@ -19,7 +19,7 @@ The Claude Code SDK has been renamed to the **Claude Agent SDK** and its documen | **Documentation Location** | Claude Code docs | API Guide → Agent SDK section | - **Documentation Changes:** The Agent SDK documentation has moved from the Claude Code docs to the API Guide under a dedicated [Agent SDK](/en/agent-sdk/overview) section. The Claude Code docs now focus on the CLI tool and automation features. + **Documentation Changes:** The Agent SDK documentation has moved from the Claude Code docs to the API Guide under a dedicated [Agent SDK](/docs/en/agent-sdk/overview) section. The Claude Code docs now focus on the CLI tool and automation features. ## Migration Steps @@ -274,7 +274,7 @@ To run isolated from filesystem settings, pass an empty array: Isolation is especially important for CI/CD pipelines, deployed applications, test environments, and multi-tenant systems where local customizations should not leak in. - SDK v0.1.0 briefly defaulted to no settings loaded; this was reverted in subsequent releases. Python SDK 0.1.59 and earlier treated an empty list the same as omitting the option, so upgrade before relying on `setting_sources=[]`. See [What settingSources does not control](/en/agent-sdk/claude-code-features#what-settingsources-does-not-control) for inputs that are read even when `settingSources` is `[]`. + SDK v0.1.0 briefly defaulted to no settings loaded; this was reverted in subsequent releases. Python SDK 0.1.59 and earlier treated an empty list the same as omitting the option, so upgrade before relying on `setting_sources=[]`. See [What settingSources does not control](/docs/en/agent-sdk/claude-code-features#what-settingsources-does-not-control) for inputs that are read even when `settingSources` is `[]`. ## Why the Rename? @@ -303,7 +303,7 @@ If you encounter any issues during migration: ## Next Steps -* Explore the [Agent SDK Overview](/en/agent-sdk/overview) to learn about available features -* Check out the [TypeScript SDK Reference](/en/agent-sdk/typescript) for detailed API documentation -* Review the [Python SDK Reference](/en/agent-sdk/python) for Python-specific documentation -* Learn about [Custom Tools](/en/agent-sdk/custom-tools) and [MCP Integration](/en/agent-sdk/mcp) +* Explore the [Agent SDK Overview](/docs/en/agent-sdk/overview) to learn about available features +* Check out the [TypeScript SDK Reference](/docs/en/agent-sdk/typescript) for detailed API documentation +* Review the [Python SDK Reference](/docs/en/agent-sdk/python) for Python-specific documentation +* Learn about [Custom Tools](/docs/en/agent-sdk/custom-tools) and [MCP Integration](/docs/en/agent-sdk/mcp) diff --git a/content/en/docs/claude-code/agent-sdk/modifying-system-prompts.md b/content/en/docs/claude-code/agent-sdk/modifying-system-prompts.md index 05cde2586..dec1a43d4 100644 --- a/content/en/docs/claude-code/agent-sdk/modifying-system-prompts.md +++ b/content/en/docs/claude-code/agent-sdk/modifying-system-prompts.md @@ -45,11 +45,11 @@ The [comparison table](#compare-the-four-approaches) shows what each customizati ## Customize agent behavior -Output styles, `append`, and a custom prompt string each change the system prompt directly. CLAUDE.md takes a different path: the SDK reads it and injects its content into the conversation as project context, not into the system prompt, so it shapes behavior alongside whichever system prompt you choose. [Skills](/en/agent-sdk/skills), [hooks](/en/agent-sdk/hooks), and [permissions](/en/agent-sdk/permissions) also shape behavior outside the system prompt and are covered on their own pages. +Output styles, `append`, and a custom prompt string each change the system prompt directly. CLAUDE.md takes a different path: the SDK reads it and injects its content into the conversation as project context, not into the system prompt, so it shapes behavior alongside whichever system prompt you choose. [Skills](/docs/en/agent-sdk/skills), [hooks](/docs/en/agent-sdk/hooks), and [permissions](/docs/en/agent-sdk/permissions) also shape behavior outside the system prompt and are covered on their own pages. ### CLAUDE.md files for project-level instructions -CLAUDE.md files give Claude persistent project context and instructions. The SDK injects their content into the conversation, not into the system prompt, so they work with any system prompt configuration. For what to put in CLAUDE.md, where to place it, and how to write effective instructions, see [How Claude remembers your project](/en/memory). This section covers what's specific to the SDK: how CLAUDE.md loads. +CLAUDE.md files give Claude persistent project context and instructions. The SDK injects their content into the conversation, not into the system prompt, so they work with any system prompt configuration. For what to put in CLAUDE.md, where to place it, and how to write effective instructions, see [How Claude remembers your project](/docs/en/memory). This section covers what's specific to the SDK: how CLAUDE.md loads. The SDK reads CLAUDE.md when the matching setting source is enabled: `'project'` loads `CLAUDE.md` or `.claude/CLAUDE.md` from the working directory, and `'user'` loads `~/.claude/CLAUDE.md`. Default `query()` options enable both sources, so CLAUDE.md loads automatically. If you set `settingSources` in TypeScript or `setting_sources` in Python explicitly, include the sources you need. CLAUDE.md loading is controlled by setting sources, not by the `claude_code` preset. @@ -108,7 +108,7 @@ Output styles are saved configurations that modify Claude's system prompt. They' #### Create an output style -An output style is a markdown file with [frontmatter](/en/output-styles#frontmatter) for metadata, followed by the prompt content. Save it to `~/.claude/output-styles/` for a user-level style available in every project, or `.claude/output-styles/` in your repository for a project-level style you can commit and share with your team. +An output style is a markdown file with [frontmatter](/docs/en/output-styles#frontmatter) for metadata, followed by the prompt content. Save it to `~/.claude/output-styles/` for a user-level style available in every project, or `.claude/output-styles/` in your repository for a project-level style you can commit and share with your team. By default, a custom output style replaces the `claude_code` preset's software engineering instructions with your own. To keep them and layer your instructions on top, set `keep-coding-instructions: true` in the frontmatter. Keep them when your agent is still doing software engineering work. Leave them out when you're replacing the role entirely. @@ -245,7 +245,7 @@ The following example pairs a shared `append` block with `excludeDynamicSections **Tradeoffs:** the working directory, the git-repo flag, the platform, the active shell, the OS version, and auto-memory paths still reach Claude, but as part of the first user message rather than the system prompt. Instructions in the user message carry marginally less weight than the same text in the system prompt, so Claude may rely on them less strongly when reasoning about the current directory or auto-memory paths. Enable this option when cross-session cache reuse matters more than maximally authoritative environment context. -For the equivalent flag in non-interactive CLI mode, see [`--exclude-dynamic-system-prompt-sections`](/en/cli-reference). +For the equivalent flag in non-interactive CLI mode, see [`--exclude-dynamic-system-prompt-sections`](/docs/en/cli-reference). ### Custom system prompts @@ -323,7 +323,7 @@ The four customization methods differ in where they live, how they're shared, an ### When to use CLAUDE.md -Use CLAUDE.md for instructions that should apply to every session in a project, regardless of which system prompt the session uses: coding standards, common commands, architecture context, and team conventions. CLAUDE.md is committed to your repository, so it stays in sync with the code it describes. See [When to add to CLAUDE.md](/en/memory#when-to-add-to-claude-md) for full guidance. +Use CLAUDE.md for instructions that should apply to every session in a project, regardless of which system prompt the session uses: coding standards, common commands, architecture context, and team conventions. CLAUDE.md is committed to your repository, so it stays in sync with the code it describes. See [When to add to CLAUDE.md](/docs/en/memory#when-to-add-to-claude-md) for full guidance. CLAUDE.md files load when the `project` setting source is enabled, which it is for default `query()` options. If you set `settingSources` in TypeScript or `setting_sources` in Python explicitly, include `'project'` to keep loading project-level CLAUDE.md. @@ -431,8 +431,8 @@ The example below assumes a Code Reviewer output style is already active. The `a ## See also -* [Output styles](/en/output-styles): create, manage, and share output styles for the CLI, including the file format and storage locations -* [How Claude remembers your project](/en/memory): what to put in CLAUDE.md, where to place it, and how to write effective project instructions -* [TypeScript SDK reference](/en/agent-sdk/typescript): the full `Options` type, including `systemPrompt`, `settingSources`, and `settings` -* [Python SDK reference](/en/agent-sdk/python): the full `ClaudeAgentOptions` type, including `system_prompt` and `setting_sources` -* [Settings](/en/settings): the `settings.json` reference, including where output styles and other configuration are stored +* [Output styles](/docs/en/output-styles): create, manage, and share output styles for the CLI, including the file format and storage locations +* [How Claude remembers your project](/docs/en/memory): what to put in CLAUDE.md, where to place it, and how to write effective project instructions +* [TypeScript SDK reference](/docs/en/agent-sdk/typescript): the full `Options` type, including `systemPrompt`, `settingSources`, and `settings` +* [Python SDK reference](/docs/en/agent-sdk/python): the full `ClaudeAgentOptions` type, including `system_prompt` and `setting_sources` +* [Settings](/docs/en/settings): the `settings.json` reference, including where output styles and other configuration are stored diff --git a/content/en/docs/claude-code/agent-sdk/observability.md b/content/en/docs/claude-code/agent-sdk/observability.md index bbc43b1ef..34c372944 100644 --- a/content/en/docs/claude-code/agent-sdk/observability.md +++ b/content/en/docs/claude-code/agent-sdk/observability.md @@ -15,7 +15,7 @@ When you run agents in production, you need visibility into what they did: The Agent SDK can export this data as OpenTelemetry traces, metrics, and log events to any backend that accepts the OpenTelemetry Protocol (OTLP), such as Honeycomb, Datadog, Grafana, Langfuse, or a self-hosted collector. -This guide explains how the SDK emits telemetry, how to configure the export, and how to tag and filter the data once it reaches your backend. To read token usage and cost directly from the SDK response stream instead of exporting to a backend, see [Track cost and usage](/en/agent-sdk/cost-tracking). +This guide explains how the SDK emits telemetry, how to configure the export, and how to tag and filter the data once it reaches your backend. To read token usage and cost directly from the SDK response stream instead of exporting to a backend, see [Track cost and usage](/docs/en/agent-sdk/cost-tracking). ## How telemetry flows from the SDK @@ -34,7 +34,7 @@ The CLI exports three independent OpenTelemetry signals. Each has its own enable | Log events | Structured records for each prompt, API request, API error, and tool result | `OTEL_LOGS_EXPORTER` | | Traces | Spans for each interaction, model request, tool call, and hook (beta) | `OTEL_TRACES_EXPORTER` plus `CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1` | -For the complete list of metric names, event names, and attributes, see the Claude Code [Monitoring](/en/monitoring-usage) reference. The Agent SDK emits the same data because it runs the same CLI. Span names are listed in [Read agent traces](#read-agent-traces) below. +For the complete list of metric names, event names, and attributes, see the Claude Code [Monitoring](/docs/en/monitoring-usage) reference. The Agent SDK emits the same data because it runs the same CLI. Span names are listed in [Read agent traces](#read-agent-traces) below. ## Enable telemetry export @@ -144,15 +144,15 @@ Traces give you the most detailed view of an agent run. With `CLAUDE_CODE_ENHANC * **`claude_code.interaction`:** wraps a single turn of the agent loop, from receiving a prompt to producing a response. * **`claude_code.llm_request`:** wraps each call to the Claude API, with model name, latency, and token counts as attributes. * **`claude_code.tool`:** wraps each tool invocation, with child spans for the permission wait (`claude_code.tool.blocked_on_user`) and the execution itself (`claude_code.tool.execution`). -* **`claude_code.hook`:** wraps each [hook](/en/agent-sdk/hooks) execution. Requires detailed beta tracing (`ENABLE_BETA_TRACING_DETAILED=1` and `BETA_TRACING_ENDPOINT`) in addition to the variables above. +* **`claude_code.hook`:** wraps each [hook](/docs/en/agent-sdk/hooks) execution. Requires detailed beta tracing (`ENABLE_BETA_TRACING_DETAILED=1` and `BETA_TRACING_ENDPOINT`) in addition to the variables above. The `llm_request`, `tool`, and `hook` spans are children of the enclosing `claude_code.interaction` span. When the agent spawns a subagent through the Task tool, the subagent's `llm_request` and `tool` spans nest under the parent agent's `claude_code.tool` span, so the full delegation chain appears as one trace. -Spans carry a `session.id` attribute by default. When you make several `query()` calls against the same [session](/en/agent-sdk/sessions), filter on `session.id` in your backend to see them as one timeline. The attribute is omitted if `OTEL_METRICS_INCLUDE_SESSION_ID` is set to a falsy value. +Spans carry a `session.id` attribute by default. When you make several `query()` calls against the same [session](/docs/en/agent-sdk/sessions), filter on `session.id` in your backend to see them as one timeline. The attribute is omitted if `OTEL_METRICS_INCLUDE_SESSION_ID` is set to a falsy value. Tracing is in beta. Span names and attributes may change between releases. See - [Traces (beta)](/en/monitoring-usage#traces-beta) in the Monitoring reference + [Traces (beta)](/docs/en/monitoring-usage#traces-beta) in the Monitoring reference for the trace exporter configuration variables. @@ -160,11 +160,11 @@ Spans carry a `session.id` attribute by default. When you make several `query()` The SDK automatically propagates W3C trace context into the CLI subprocess. When you call `query()` while an OpenTelemetry span is active in your application, the SDK injects `TRACEPARENT` and `TRACESTATE` into the child process environment, and the CLI reads them so its `claude_code.interaction` span becomes a child of your span. The agent run then appears inside your application's trace instead of as a disconnected root. -OTLP event log records emitted during the run carry the same trace context: with `TRACEPARENT` set, each record's `trace_id` and `span_id` match your application's trace, so you can join [events](/en/monitoring-usage#events) to spans in your backend. {/* min-version: 2.1.212 */}Before v2.1.212, event records emitted outside an active span didn't carry `trace_id` or `span_id`. +OTLP event log records emitted during the run carry the same trace context: with `TRACEPARENT` set, each record's `trace_id` and `span_id` match your application's trace, so you can join [events](/docs/en/monitoring-usage#events) to spans in your backend. {/* min-version: 2.1.212 */}Before v2.1.212, event records emitted outside an active span didn't carry `trace_id` or `span_id`. When trace-context propagation is enabled, the CLI also forwards `TRACEPARENT` to every Bash and PowerShell command it runs. If a command launched through the Bash tool emits its own OpenTelemetry spans, those spans nest under the `claude_code.tool.execution` span that wraps the command. -Auto-injection is skipped when you set `TRACEPARENT` explicitly in `options.env`, so you can pin a specific parent context if needed. Interactive CLI sessions ignore inbound `TRACEPARENT` entirely; only Agent SDK and `claude -p` runs honor it. See [Traces (beta)](/en/monitoring-usage#traces-beta) in the Monitoring reference for the full span and attribute reference. +Auto-injection is skipped when you set `TRACEPARENT` explicitly in `options.env`, so you can pin a specific parent context if needed. Interactive CLI sessions ignore inbound `TRACEPARENT` entirely; only Agent SDK and `claude -p` runs honor it. See [Traces (beta)](/docs/en/monitoring-usage#traces-beta) in the Monitoring reference for the full span and attribute reference. ## Tag telemetry from your agent @@ -198,9 +198,9 @@ The following example renames the service and attaches deployment metadata. Thes ## Attribute actions to your end users -The CLI attaches [identity attributes](/en/monitoring-usage#standard-attributes) to every event based on the credential it uses to call Anthropic. When you build an application that serves many end users from one deployment, these attributes identify your service's credential, not the end user on whose behalf the agent acted. +The CLI attaches [identity attributes](/docs/en/monitoring-usage#standard-attributes) to every event based on the credential it uses to call Anthropic. When you build an application that serves many end users from one deployment, these attributes identify your service's credential, not the end user on whose behalf the agent acted. -To make tool calls and MCP activity attributable to your application's end users, inject end-user identity as resource attributes on each `query()` call. Percent-encode values before interpolating them, since `OTEL_RESOURCE_ATTRIBUTES` [reserves commas, spaces, and equals signs](/en/monitoring-usage#multi-team-organization-support). The following example attaches the requesting user and tenant to every span and event from one request. It assumes a `request` object from your web framework carrying the user and tenant IDs: +To make tool calls and MCP activity attributable to your application's end users, inject end-user identity as resource attributes on each `query()` call. Percent-encode values before interpolating them, since `OTEL_RESOURCE_ATTRIBUTES` [reserves commas, spaces, and equals signs](/docs/en/monitoring-usage#multi-team-organization-support). The following example attaches the requesting user and tenant to every span and event from one request. It assumes a `request` object from your web framework carrying the user and tenant IDs: ```python Python theme={null} @@ -225,7 +225,7 @@ To make tool calls and MCP activity attributable to your application's end users ``` -With end-user identity attached, the `tool_decision`, `tool_result`, `mcp_server_connection`, and `permission_mode_changed` events, which export as log records named with a `claude_code.` prefix, become a per-user audit trail you can forward to a Security Information and Event Management (SIEM) platform. See [Audit security events](/en/monitoring-usage#audit-security-events) in the Monitoring reference for the full list of security-relevant events and the attributes each one carries. +With end-user identity attached, the `tool_decision`, `tool_result`, `mcp_server_connection`, and `permission_mode_changed` events, which export as log records named with a `claude_code.` prefix, become a per-user audit trail you can forward to a Security Information and Event Management (SIEM) platform. See [Audit security events](/docs/en/monitoring-usage#audit-security-events) in the Monitoring reference for the full list of security-relevant events and the attributes each one carries. ## Control sensitive data in exports @@ -238,12 +238,12 @@ Telemetry is structural by default. Durations, model names, and tool names are r | `OTEL_LOG_TOOL_CONTENT=1` | Full tool input and output bodies as span events on `claude_code.tool`, truncated at 60 KB. Requires [tracing](#read-agent-traces) to be enabled | | `OTEL_LOG_RAW_API_BODIES` | Full Anthropic Messages API request and response JSON as `claude_code.api_request_body` and `claude_code.api_response_body` log events. Set to `1` for inline bodies truncated at 60 KB, or `file:` for untruncated bodies on disk with a `body_ref` path in the event. Bodies include the entire conversation history and have extended-thinking content redacted. Enabling this implies consent to everything the three variables above would reveal | -Leave these unset unless your observability pipeline is approved to store the data your agent handles. See [Security and privacy](/en/monitoring-usage#security-and-privacy) in the Monitoring reference for the full list of attributes and redaction behavior. +Leave these unset unless your observability pipeline is approved to store the data your agent handles. See [Security and privacy](/docs/en/monitoring-usage#security-and-privacy) in the Monitoring reference for the full list of attributes and redaction behavior. ## Related documentation These guides cover adjacent topics for monitoring and deploying agents: -* [Track cost and usage](/en/agent-sdk/cost-tracking): read token and cost data from the message stream without an external backend. -* [Hosting the Agent SDK](/en/agent-sdk/hosting): deploy agents in containers where you can set OpenTelemetry variables at the environment level. -* [Monitoring](/en/monitoring-usage): the complete reference for every environment variable, metric, and event the CLI emits. +* [Track cost and usage](/docs/en/agent-sdk/cost-tracking): read token and cost data from the message stream without an external backend. +* [Hosting the Agent SDK](/docs/en/agent-sdk/hosting): deploy agents in containers where you can set OpenTelemetry variables at the environment level. +* [Monitoring](/docs/en/monitoring-usage): the complete reference for every environment variable, metric, and event the CLI emits. diff --git a/content/en/docs/claude-code/agent-sdk/overview.md b/content/en/docs/claude-code/agent-sdk/overview.md index b0da1eb0d..3522e3d68 100644 --- a/content/en/docs/claude-code/agent-sdk/overview.md +++ b/content/en/docs/claude-code/agent-sdk/overview.md @@ -6,7 +6,7 @@ > Build production AI agents with Claude Code as a library -Build AI agents that autonomously read files, run commands, search the web, edit code, and more. The Agent SDK gives you the same tools, agent loop, and context management that power Claude Code, programmable in Python and TypeScript. For other languages, [run the CLI programmatically](/en/headless) with the `-p` flag and `--output-format json`. For the thinking behind agent harness design, see [A harness for every task: dynamic workflows in Claude Code](https://claude.com/blog/a-harness-for-every-task-dynamic-workflows-in-claude-code) on the blog. +Build AI agents that autonomously read files, run commands, search the web, edit code, and more. The Agent SDK gives you the same tools, agent loop, and context management that power Claude Code, programmable in Python and TypeScript. For other languages, [run the CLI programmatically](/docs/en/headless) with the `-p` flag and `--output-format json`. For the thinking behind agent harness design, see [A harness for every task: dynamic workflows in Claude Code](https://claude.com/blog/a-harness-for-every-task-dynamic-workflows-in-claude-code) on the blog. To run the example below, install the SDK first by following the steps in [Get started](#get-started). ```python Python theme={null} @@ -40,7 +40,7 @@ Build AI agents that autonomously read files, run commands, search the web, edit The Agent SDK includes built-in tools for reading files, running commands, and editing code, so your agent can start working immediately without you implementing tool execution. Dive into the quickstart or explore real agents built with the SDK: - + Build a bug-fixing agent in minutes @@ -56,8 +56,13 @@ The Agent SDK includes built-in tools for reading files, running commands, and e ```bash theme={null} + npm init -y + npm pkg set type=module npm install @anthropic-ai/claude-agent-sdk + npm install --save-dev tsx ``` + + Setting `"type": "module"` in `package.json` lets your agent script use top-level `await`, and [tsx](https://tsx.is) runs TypeScript files directly. In an existing CommonJS project, skip the first two commands and name your script `agent.mts` instead of `agent.ts`. @@ -95,7 +100,7 @@ The Agent SDK includes built-in tools for reading files, running commands, and e - The TypeScript SDK bundles a native Claude Code binary for your platform as an optional dependency, so you don't need to install Claude Code separately. + Both the TypeScript and Python SDKs bundle a native Claude Code binary for your platform, so you don't need to install Claude Code separately. @@ -121,7 +126,7 @@ The Agent SDK includes built-in tools for reading files, running commands, and e * **Google Cloud's Agent Platform**: set `CLAUDE_CODE_USE_VERTEX=1` environment variable and configure Google Cloud credentials * **Microsoft Foundry**: set `CLAUDE_CODE_USE_FOUNDRY=1` environment variable and configure Azure credentials - See the setup guides for [Amazon Bedrock](/en/amazon-bedrock), [Claude Platform on AWS](/en/claude-platform-on-aws), [Google Cloud's Agent Platform](/en/google-vertex-ai), or [Microsoft Foundry](/en/microsoft-foundry) for details. + See the setup guides for [Amazon Bedrock](/docs/en/amazon-bedrock), [Claude Platform on AWS](/docs/en/claude-platform-on-aws), [Google Cloud's Agent Platform](/docs/en/google-vertex-ai), or [Microsoft Foundry](/docs/en/microsoft-foundry) for details. Unless previously approved, Anthropic does not allow third party developers to offer claude.ai login or rate limits for their products, including agents built on the Claude Agent SDK. Please use the API key authentication methods described in this document instead. @@ -160,10 +165,38 @@ The Agent SDK includes built-in tools for reading files, running commands, and e } ``` + + Save the example as `agent.py` or `agent.ts`, then run it. The agent prints a short summary of the files in the directory. + + + + ```bash theme={null} + npx tsx agent.ts + ``` + + If you named your script `agent.mts` for a CommonJS project, run `npx tsx agent.mts` instead. + + + + ```bash theme={null} + uv run agent.py + ``` + + + + With the virtual environment activated, on macOS or Linux: + + ```bash theme={null} + python3 agent.py + ``` + + On Windows, run `python agent.py`. + + -**Ready to build?** Follow the [Quickstart](/en/agent-sdk/quickstart) to create an agent that finds and fixes bugs in minutes. +**Ready to build?** Follow the [Quickstart](/docs/en/agent-sdk/quickstart) to create an agent that finds and fixes bugs in minutes. ## Capabilities @@ -184,9 +217,9 @@ Everything that makes Claude Code powerful is available in the SDK: | **Grep** | Search file contents with regex | | **WebSearch** | Search the web for current information | | **WebFetch** | Fetch and parse web page content | - | **[AskUserQuestion](/en/agent-sdk/user-input#handle-clarifying-questions)** | Ask the user clarifying questions with multiple choice options | + | **[AskUserQuestion](/docs/en/agent-sdk/user-input#handle-clarifying-questions)** | Ask the user clarifying questions with multiple choice options | - For the full list, including scheduling and worktree tools, see the [tools reference](/en/tools-reference). + For the full list, including scheduling and worktree tools, see the [tools reference](/docs/en/tools-reference). This example creates an agent that searches your codebase for TODO comments: @@ -244,7 +277,7 @@ Everything that makes Claude Code powerful is available in the SDK: async def main(): async for message in query( - prompt="Refactor utils.py to improve readability", + prompt="Create a file named hello.py that prints a greeting", options=ClaudeAgentOptions( allowed_tools=["Read", "Edit"], permission_mode="acceptEdits", @@ -273,7 +306,7 @@ Everything that makes Claude Code powerful is available in the SDK: }; for await (const message of query({ - prompt: "Refactor utils.py to improve readability", + prompt: "Create a file named hello.ts that prints a greeting", options: { allowedTools: ["Read", "Edit"], permissionMode: "acceptEdits", @@ -287,7 +320,9 @@ Everything that makes Claude Code powerful is available in the SDK: ``` - [Learn more about hooks →](/en/agent-sdk/hooks) + After the agent finishes, run `cat audit.log` to see the recorded file changes. + + [Learn more about hooks →](/docs/en/agent-sdk/hooks) @@ -345,7 +380,7 @@ Everything that makes Claude Code powerful is available in the SDK: Messages from within a subagent's context include a `parent_tool_use_id` field, letting you track which messages belong to which subagent execution. - [Learn more about subagents →](/en/agent-sdk/subagents) + [Learn more about subagents →](/docs/en/agent-sdk/subagents) @@ -365,7 +400,8 @@ Everything that makes Claude Code powerful is available in the SDK: options=ClaudeAgentOptions( mcp_servers={ "playwright": {"command": "npx", "args": ["@playwright/mcp@latest"]} - } + }, + allowed_tools=["mcp__playwright__*"], ), ): if hasattr(message, "result"): @@ -383,7 +419,8 @@ Everything that makes Claude Code powerful is available in the SDK: options: { mcpServers: { playwright: { command: "npx", args: ["@playwright/mcp@latest"] } - } + }, + allowedTools: ["mcp__playwright__*"] } })) { if ("result" in message) console.log(message.result); @@ -391,14 +428,14 @@ Everything that makes Claude Code powerful is available in the SDK: ``` - [Learn more about MCP →](/en/agent-sdk/mcp) + [Learn more about MCP →](/docs/en/agent-sdk/mcp) Control exactly which tools your agent can use. Allow safe operations, block dangerous ones, or require approval for sensitive actions. - For interactive approval prompts and the `AskUserQuestion` tool, see [Handle approvals and user input](/en/agent-sdk/user-input). + For interactive approval prompts and the `AskUserQuestion` tool, see [Handle approvals and user input](/docs/en/agent-sdk/user-input). This example creates a read-only agent that can analyze but not modify code. `allowed_tools` pre-approves `Read`, `Glob`, and `Grep` so they run without prompting. Tools not listed are still available but fall through to the permission mode; to block tools entirely, use `disallowed_tools`. @@ -437,7 +474,7 @@ Everything that makes Claude Code powerful is available in the SDK: ``` - [Learn more about permissions →](/en/agent-sdk/permissions) + [Learn more about permissions →](/docs/en/agent-sdk/permissions) @@ -513,7 +550,7 @@ Everything that makes Claude Code powerful is available in the SDK: ``` - [Learn more about sessions →](/en/agent-sdk/sessions) + [Learn more about sessions →](/docs/en/agent-sdk/sessions) @@ -523,10 +560,10 @@ The SDK also supports Claude Code's filesystem-based configuration. With default | Feature | Description | Location | | ------------------------------------------------ | ----------------------------------------------------------------------------- | ---------------------------------- | -| [Skills](/en/agent-sdk/skills) | Specialized capabilities Claude uses automatically or you invoke with `/name` | `.claude/skills/*/SKILL.md` | -| [Commands](/en/agent-sdk/slash-commands) | Custom commands in the legacy format. Use skills for new custom commands | `.claude/commands/*.md` | -| [Memory](/en/agent-sdk/modifying-system-prompts) | Project context and instructions | `CLAUDE.md` or `.claude/CLAUDE.md` | -| [Plugins](/en/agent-sdk/plugins) | Extend with skills, agents, hooks, and MCP servers | Programmatic via `plugins` option | +| [Skills](/docs/en/agent-sdk/skills) | Specialized capabilities Claude uses automatically or you invoke with `/name` | `.claude/skills/*/SKILL.md` | +| [Commands](/docs/en/agent-sdk/slash-commands) | Custom commands in the legacy format. Use skills for new custom commands | `.claude/commands/*.md` | +| [Memory](/docs/en/agent-sdk/modifying-system-prompts) | Project context and instructions | `CLAUDE.md` or `.claude/CLAUDE.md` | +| [Plugins](/docs/en/agent-sdk/plugins) | Extend with skills, agents, hooks, and MCP servers | Programmatic via `plugins` option | ## Compare the Agent SDK to other Claude tools @@ -536,7 +573,7 @@ The Claude Platform offers multiple ways to build with Claude. Here's how the Ag The [Anthropic Client SDK](https://platform.claude.com/docs/en/api/client-sdks) gives you direct API access: you send prompts and implement tool execution yourself. The **Agent SDK** gives you Claude with built-in tool execution. - With the Client SDK, you implement a tool loop. With the Agent SDK, Claude handles it: + With the Client SDK, you implement a tool loop. With the Agent SDK, Claude handles it. This simplified pseudocode shows the difference: ```python Python theme={null} @@ -635,7 +672,7 @@ Use of the Claude Agent SDK is governed by [Anthropic's Commercial Terms of Serv ## Next steps - + Build an agent that finds and fixes bugs in minutes @@ -643,11 +680,11 @@ Use of the Claude Agent SDK is governed by [Anthropic's Commercial Terms of Serv Email assistant, research agent, and more - + Full TypeScript API reference and examples - + Full Python API reference and examples diff --git a/content/en/docs/claude-code/agent-sdk/permissions.md b/content/en/docs/claude-code/agent-sdk/permissions.md index 66a241d7a..f36be5455 100644 --- a/content/en/docs/claude-code/agent-sdk/permissions.md +++ b/content/en/docs/claude-code/agent-sdk/permissions.md @@ -6,10 +6,10 @@ > Control how your agent uses tools with permission modes, hooks, and declarative allow/deny rules. -The Claude Agent SDK provides permission controls to manage how Claude uses tools. Use permission modes and rules to define what's allowed automatically, and the [`canUseTool` callback](/en/agent-sdk/user-input) to handle everything else at runtime. +The Claude Agent SDK provides permission controls to manage how Claude uses tools. Use permission modes and rules to define what's allowed automatically, and the [`canUseTool` callback](/docs/en/agent-sdk/user-input) to handle everything else at runtime. - This page covers permission modes and rules. To build interactive approval flows where users approve or deny tool requests at runtime, see [Handle approvals and user input](/en/agent-sdk/user-input). + This page covers permission modes and rules. To build interactive approval flows where users approve or deny tool requests at runtime, see [Handle approvals and user input](/docs/en/agent-sdk/user-input). ## How permissions are evaluated @@ -18,19 +18,19 @@ When Claude requests a tool, the SDK checks permissions in this order: - Run [hooks](/en/agent-sdk/hooks) first. A hook can deny the call outright or pass it on. A hook that returns `allow` does not skip the deny and ask rules below; those are evaluated regardless of the hook result. + Run [hooks](/docs/en/agent-sdk/hooks) first. A hook can deny the call outright or pass it on. A hook that returns `allow` does not skip the deny and ask rules below; those are evaluated regardless of the hook result. - Check `deny` rules (from `disallowed_tools` and [settings.json](/en/settings#permission-settings)). If a deny rule matches, the tool is blocked, even in `bypassPermissions` mode. Bare-name deny rules like `Bash` remove the tool from Claude's context before this evaluation begins, so only scoped rules like `Bash(rm *)` are checked at this step. + Check `deny` rules (from `disallowed_tools` and [settings.json](/docs/en/settings#permission-settings)). If a deny rule matches, the tool is blocked, even in `bypassPermissions` mode. Bare-name deny rules like `Bash` remove the tool from Claude's context before this evaluation begins, so only scoped rules like `Bash(rm *)` are checked at this step. - Check `ask` rules from [settings.json](/en/settings#permission-settings). If an ask rule matches, the call falls through to your [`canUseTool` callback](/en/agent-sdk/user-input) for confirmation, even in `bypassPermissions` mode. + Check `ask` rules from [settings.json](/docs/en/settings#permission-settings). If an ask rule matches, the call falls through to your [`canUseTool` callback](/docs/en/agent-sdk/user-input) for confirmation, even in `bypassPermissions` mode. - Tools that require user interaction behave the same way: `AskUserQuestion` and MCP tools whose server sets [`_meta["anthropic/requiresUserInteraction"]`](/en/mcp#require-approval-for-a-specific-tool) always fall through to the callback, even when an allow rule matches. In `dontAsk` mode both cases are denied instead, because that mode never prompts. {/* min-version: 2.1.199 */}The MCP annotation requires Claude Code v2.1.199 or later. + Tools that require user interaction behave the same way: `AskUserQuestion` and MCP tools whose server sets [`_meta["anthropic/requiresUserInteraction"]`](/docs/en/mcp#require-approval-for-a-specific-tool) always fall through to the callback, even when an allow rule matches. In `dontAsk` mode both cases are denied instead, because that mode never prompts. {/* min-version: 2.1.199 */}The MCP annotation requires Claude Code v2.1.199 or later. - [claude.ai connector](/en/mcp#organization-controls-on-connector-tools) tools your organization has set to `ask` also leave the flow at this step. Every call falls through to the callback, even in `bypassPermissions` mode and even when an allow rule matches. The callback receives the reason `Your organization requires approval for this tool`. In `dontAsk` mode the call is denied instead, because that mode never prompts. + [claude.ai connector](/docs/en/mcp#organization-controls-on-connector-tools) tools your organization has set to `ask` also leave the flow at this step. Every call falls through to the callback, even in `bypassPermissions` mode and even when an allow rule matches. The callback receives the reason `Your organization requires approval for this tool`. In `dontAsk` mode the call is denied instead, because that mode never prompts. @@ -42,7 +42,7 @@ When Claude requests a tool, the SDK checks permissions in this order: - If not resolved by any of the above, call your [`canUseTool` callback](/en/agent-sdk/user-input) for a decision. In `dontAsk` mode, this step is skipped and the tool is denied. + If not resolved by any of the above, call your [`canUseTool` callback](/docs/en/agent-sdk/user-input) for a decision. In `dontAsk` mode, this step is skipped and the tool is denied. @@ -55,12 +55,12 @@ As of v2.1.198, if you pass a `canUseTool` callback that this evaluation order c Entries with a specifier such as `Bash(ls *)` and the `acceptEdits` mode don't trigger it, and allow rules coming from settings files aren't visible to the check. -Listen with `process.on('warning', ...)` and match the code to log or suppress it. To gate every tool call regardless of mode and rules, use a [`PreToolUse` hook](/en/agent-sdk/hooks) instead. +Listen with `process.on('warning', ...)` and match the code to log or suppress it. To gate every tool call regardless of mode and rules, use a [`PreToolUse` hook](/docs/en/agent-sdk/hooks) instead. This page focuses on **allow and deny rules** and **permission modes**. For the other steps: -* **Hooks:** run custom code to allow, deny, or modify tool requests. See [Control execution with hooks](/en/agent-sdk/hooks). -* **canUseTool callback:** prompt users for approval at runtime, when no earlier step resolves the call. See [Handle approvals and user input](/en/agent-sdk/user-input). +* **Hooks:** run custom code to allow, deny, or modify tool requests. See [Control execution with hooks](/docs/en/agent-sdk/hooks). +* **canUseTool callback:** prompt users for approval at runtime, when no earlier step resolves the call. See [Handle approvals and user input](/docs/en/agent-sdk/user-input). ## Allow and deny rules @@ -77,12 +77,12 @@ Allow rules accept tool-name globs only after a literal `mcp____` prefix Scoped rules for `Read` and `Edit` take a path pattern. `Edit(path)` rules govern all built-in tools that write files, including `Write` and `NotebookEdit`; a `Write(path)` rule is never matched by the file permission checks. -Use `//path` for an absolute filesystem path: a deny rule of `Edit(//secrets/**)` blocks writes anywhere under `/secrets` on disk. With a single leading slash, `Edit(/secrets/**)` anchors at the rule's source instead. For rules passed through `allowed_tools` or `disallowed_tools`, that means the session's working directory, so the rule doesn't block `/secrets` on disk. See [Read and Edit rules](/en/permissions#read-and-edit) for the four anchor forms and how rules from settings files resolve. +Use `//path` for an absolute filesystem path: a deny rule of `Edit(//secrets/**)` blocks writes anywhere under `/secrets` on disk. With a single leading slash, `Edit(/secrets/**)` anchors at the rule's source instead. For rules passed through `allowed_tools` or `disallowed_tools`, that means the session's working directory, so the rule doesn't block `/secrets` on disk. See [Read and Edit rules](/docs/en/permissions#read-and-edit) for the four anchor forms and how rules from settings files resolve. - **Auto-approved tools never reach `canUseTool`.** A tool call approved at any earlier step, by `acceptEdits` or `bypassPermissions`, or by an allow rule, skips your `canUseTool` callback, so permission checks you put there are silently bypassed for that tool. `AskUserQuestion`, MCP tools marked [`_meta["anthropic/requiresUserInteraction"]`](/en/mcp#require-approval-for-a-specific-tool), and connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools) still reach the callback, even when an allow rule matches. + **Auto-approved tools never reach `canUseTool`.** A tool call approved at any earlier step, by `acceptEdits` or `bypassPermissions`, or by an allow rule, skips your `canUseTool` callback, so permission checks you put there are silently bypassed for that tool. `AskUserQuestion`, MCP tools marked [`_meta["anthropic/requiresUserInteraction"]`](/docs/en/mcp#require-approval-for-a-specific-tool), and connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools) still reach the callback, even when an allow rule matches. - Coverage depends on the entry's form: a bare name like `Read` or `mcp__github__get_issue` auto-approves every call to that tool, while a scoped rule like `Bash(ls *)` auto-approves only matching calls and other `Bash` calls still fall through to the callback. For checks that must run on every tool call, use a [`PreToolUse` hook](/en/agent-sdk/hooks): hooks run before every other step, and a hook deny applies even in `bypassPermissions` mode. + Coverage depends on the entry's form: a bare name like `Read` or `mcp__github__get_issue` auto-approves every call to that tool, while a scoped rule like `Bash(ls *)` auto-approves only matching calls and other `Bash` calls still fall through to the callback. For checks that must run on every tool call, use a [`PreToolUse` hook](/docs/en/agent-sdk/hooks): hooks run before every other step, and a hook deny applies even in `bypassPermissions` mode. For a locked-down agent, pair `allowedTools` with `permissionMode: "dontAsk"`. Listed tools are approved, apart from the always-prompt tools in the Warning above; anything else is denied outright instead of prompting: @@ -98,7 +98,7 @@ const options = { **`allowed_tools` does not constrain `bypassPermissions`.** `allowed_tools` only pre-approves the tools you list. Unlisted tools are not matched by any allow rule and fall through to the permission mode, where `bypassPermissions` approves them. Setting `allowed_tools=["Read"]` alongside `permission_mode="bypassPermissions"` still approves every tool, including `Bash`, `Write`, and `Edit`. If you need `bypassPermissions` but want specific tools blocked, use `disallowed_tools`. -You can also configure allow, deny, and ask rules declaratively in `.claude/settings.json`. These rules are read when the `project` setting source is enabled, which it is for default `query()` options. If you set `setting_sources` (TypeScript: `settingSources`) explicitly, include `"project"` for them to apply. See [Permission settings](/en/settings#permission-settings) for the rule syntax. +You can also configure allow, deny, and ask rules declaratively in `.claude/settings.json`. These rules are read when the `project` setting source is enabled, which it is for default `query()` options. If you set `setting_sources` (TypeScript: `settingSources`) explicitly, include `"project"` for them to apply. See [Permission settings](/docs/en/settings#permission-settings) for the rule syntax. ## Permission modes @@ -111,16 +111,16 @@ The SDK supports these permission modes: | Mode | Description | Tool behavior | | :------------------ | :--------------------------- | :------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `default` | Standard permission behavior | No auto-approvals; unmatched tools trigger your `canUseTool` callback | -| `dontAsk` | Deny instead of prompting | Anything not pre-approved by `allowed_tools` or rules is denied; connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools) and tools that require user interaction are denied even if you've pre-approved them. `canUseTool` is never called | +| `dontAsk` | Deny instead of prompting | Anything not pre-approved by `allowed_tools` or rules is denied; connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools) and tools that require user interaction are denied even if you've pre-approved them. `canUseTool` is never called | | `acceptEdits` | Auto-accept file edits | File edits and [filesystem operations](#accept-edits-mode-acceptedits) (`mkdir`, `rm`, `mv`, etc.) are automatically approved | -| `bypassPermissions` | Bypass permission checks | Tools run without permission prompts, except tools matched by an explicit [`ask` rule](#how-permissions-are-evaluated), connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools), and tools that require user interaction (use with caution) | +| `bypassPermissions` | Bypass permission checks | Tools run without permission prompts, except tools matched by an explicit [`ask` rule](#how-permissions-are-evaluated), connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools), and tools that require user interaction (use with caution) | | `plan` | Planning mode | Claude explores and plans without editing your source files; file edits are never auto-approved and prompt through your `canUseTool` callback | -| `auto` | Model-classified approvals | A model classifier approves or denies each tool call. See [Auto mode](/en/permission-modes#eliminate-prompts-with-auto-mode) for availability | +| `auto` | Model-classified approvals | A model classifier approves or denies permission prompts. See [Auto mode](/docs/en/permission-modes#eliminate-prompts-with-auto-mode) for availability | - **Subagent inheritance:** Subagents inherit the parent session's permission mode. An [`AgentDefinition`'s `permissionMode`](/en/agent-sdk/typescript#agentdefinition) can override it, except when the parent uses `bypassPermissions`, `acceptEdits`, or `auto`: those modes apply to every subagent and can't be overridden per subagent. + **Subagent inheritance:** Subagents inherit the parent session's permission mode. An [`AgentDefinition`'s `permissionMode`](/docs/en/agent-sdk/typescript#agentdefinition) can override it, except when the parent uses `bypassPermissions`, `acceptEdits`, or `auto`: those modes apply to every subagent and can't be overridden per subagent. - Subagents may have different system prompts and less constrained behavior than your main agent, so inheriting `bypassPermissions` grants them full, autonomous system access. Explicit [`ask` rules](#how-permissions-are-evaluated), connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools), and tools that require user interaction still force a prompt. + Subagents may have different system prompts and less constrained behavior than your main agent, so inheriting `bypassPermissions` grants them full, autonomous system access. Explicit [`ask` rules](#how-permissions-are-evaluated), connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools), and tools that require user interaction still force a prompt. ### Set permission mode @@ -246,7 +246,7 @@ Both apply only to paths inside the working directory or `additionalDirectories` #### Don't ask mode (`dontAsk`) -Converts any permission prompt into a denial. Tools pre-approved by `allowed_tools`, `settings.json` allow rules, or a hook run as normal. Connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools) and tools that require user interaction are denied even when an allow rule matches. Everything else is denied without calling `canUseTool`. +Converts any permission prompt into a denial. Tools pre-approved by `allowed_tools`, `settings.json` allow rules, or a hook run as normal. Connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools) and tools that require user interaction are denied even when an allow rule matches. Everything else is denied without calling `canUseTool`. **Use when:** you want a fixed, explicit tool surface for a headless agent and prefer a hard deny over silent reliance on `canUseTool` being absent. @@ -257,7 +257,7 @@ Auto-approves all tool uses without prompts. Hooks still execute and can block o Use with extreme caution. Claude has full system access in this mode. Only use in controlled environments where you trust all possible operations. - `allowed_tools` does not constrain this mode. Every tool is approved, not just the ones you listed. Deny rules (`disallowed_tools`), explicit `ask` rules, and hooks are evaluated before the mode check and can still block a tool. Connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools) and tools that require user interaction still fall through to your `canUseTool` callback. + `allowed_tools` does not constrain this mode. Every tool is approved, not just the ones you listed. Deny rules (`disallowed_tools`), explicit `ask` rules, and hooks are evaluated before the mode check and can still block a tool. Connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools) and tools that require user interaction still fall through to your `canUseTool` callback. #### Plan mode (`plan`) @@ -266,7 +266,7 @@ Claude explores the codebase and produces a plan without editing your source fil File edits are never auto-approved in plan mode, even when an allow rule matches. They prompt through your `canUseTool` callback instead. {/* min-version: 2.1.212 */}On Claude Code v2.1.212 or later, shell commands that modify files, such as `touch` and `rm`, reach your `canUseTool` callback the same way. -Claude may use `AskUserQuestion` to clarify requirements before finalizing the plan. See [Handle approvals and user input](/en/agent-sdk/user-input#handle-clarifying-questions) for handling these prompts. +Claude may use `AskUserQuestion` to clarify requirements before finalizing the plan. See [Handle approvals and user input](/docs/en/agent-sdk/user-input#handle-clarifying-questions) for handling these prompts. **Use when:** you want Claude to propose changes without executing them, such as during code review or when you need to approve changes before they're made. @@ -274,6 +274,6 @@ Claude may use `AskUserQuestion` to clarify requirements before finalizing the p For the other steps in the permission evaluation flow: -* [Handle approvals and user input](/en/agent-sdk/user-input): interactive approval prompts and clarifying questions -* [Hooks guide](/en/agent-sdk/hooks): run custom code at key points in the agent lifecycle -* [Permission rules](/en/settings#permission-settings): declarative allow/deny rules in `settings.json` +* [Handle approvals and user input](/docs/en/agent-sdk/user-input): interactive approval prompts and clarifying questions +* [Hooks guide](/docs/en/agent-sdk/hooks): run custom code at key points in the agent lifecycle +* [Permission rules](/docs/en/settings#permission-settings): declarative allow/deny rules in `settings.json` diff --git a/content/en/docs/claude-code/agent-sdk/plugins.md b/content/en/docs/claude-code/agent-sdk/plugins.md index 7eeb0b6b2..e0e9e1f13 100644 --- a/content/en/docs/claude-code/agent-sdk/plugins.md +++ b/content/en/docs/claude-code/agent-sdk/plugins.md @@ -21,11 +21,11 @@ Plugins are packages of Claude Code extensions that can include: The `commands/` directory is a legacy format. Use `skills/` for new plugins. Claude Code continues to support both formats for backward compatibility. -For complete information on plugin structure and how to create plugins, see [Plugins](/en/plugins). +For complete information on plugin structure and how to create plugins, see [Plugins](/docs/en/plugins). ## Loading plugins -Load plugins by providing their local file system paths in your options configuration. The `type` field must be `"local"`, the only value the SDK accepts. To use a plugin distributed through a [marketplace](/en/plugin-marketplaces) or remote repository, download it first and provide the local directory path. The SDK supports loading multiple plugins from different locations. +Load plugins by providing their local file system paths in your options configuration. The `type` field must be `"local"`, the only value the SDK accepts. To use a plugin distributed through a [marketplace](/docs/en/plugin-marketplaces) or remote repository, download it first and provide the local directory path. The SDK supports loading multiple plugins from different locations. ```typescript TypeScript theme={null} @@ -291,8 +291,8 @@ my-plugin/ For detailed information on creating plugins, see: -* [Plugins](/en/plugins) - Complete plugin development guide -* [Plugins reference](/en/plugins-reference) - Technical specifications and schemas +* [Plugins](/docs/en/plugins) - Complete plugin development guide +* [Plugins reference](/docs/en/plugins-reference) - Technical specifications and schemas ## Common use cases @@ -351,8 +351,8 @@ If relative paths don't work: ## See also -* [Plugins](/en/plugins) - Complete plugin development guide -* [Plugins reference](/en/plugins-reference) - Technical specifications -* [Commands](/en/agent-sdk/slash-commands) - Using commands in the SDK -* [Subagents](/en/agent-sdk/subagents) - Working with specialized agents -* [Skills](/en/agent-sdk/skills) - Using Agent Skills +* [Plugins](/docs/en/plugins) - Complete plugin development guide +* [Plugins reference](/docs/en/plugins-reference) - Technical specifications +* [Commands](/docs/en/agent-sdk/slash-commands) - Using commands in the SDK +* [Subagents](/docs/en/agent-sdk/subagents) - Working with specialized agents +* [Skills](/docs/en/agent-sdk/skills) - Using Agent Skills diff --git a/content/en/docs/claude-code/agent-sdk/python.md b/content/en/docs/claude-code/agent-sdk/python.md index d53b36d08..d9098fe73 100644 --- a/content/en/docs/claude-code/agent-sdk/python.md +++ b/content/en/docs/claude-code/agent-sdk/python.md @@ -16,7 +16,7 @@ source .venv/bin/activate pip install claude-agent-sdk ``` -For uv, Windows PowerShell, and API key setup, see [Get started in the Agent SDK overview](/en/agent-sdk/overview#get-started). +For uv, Windows PowerShell, and API key setup, see [Get started in the Agent SDK overview](/docs/en/agent-sdk/overview#get-started). ## Choosing between `query()` and `ClaudeSDKClient` @@ -61,7 +61,7 @@ The Python SDK provides two ways to interact with Claude Code: ### `query()` -Creates a new session for each interaction with Claude Code by default. Returns an async iterator that yields messages as they arrive. Each call to `query()` starts fresh with no memory of previous interactions unless you pass `continue_conversation=True` or `resume` in [`ClaudeAgentOptions`](#claudeagentoptions). See [Sessions](/en/agent-sdk/sessions). +Creates a new session for each interaction with Claude Code by default. Returns an async iterator that yields messages as they arrive. Each call to `query()` starts fresh with no memory of previous interactions unless you pass `continue_conversation=True` or `resume` in [`ClaudeAgentOptions`](#claudeagentoptions). See [Sessions](/docs/en/agent-sdk/sessions). ```python theme={null} async def query( @@ -486,7 +486,7 @@ class ClaudeSDKClient: | `interrupt()` | Send interrupt signal (only works in streaming mode) | | `set_permission_mode(mode)` | Change the permission mode for the current session | | `set_model(model)` | Change the model for the current session. Pass `None` to reset to default | -| `rewind_files(user_message_id)` | Restore files to their state at the specified user message. Requires `enable_file_checkpointing=True`. See [File checkpointing](/en/agent-sdk/file-checkpointing) | +| `rewind_files(user_message_id)` | Restore files to their state at the specified user message. Requires `enable_file_checkpointing=True`. See [File checkpointing](/docs/en/agent-sdk/file-checkpointing) | | `get_mcp_status()` | Get the status of all configured MCP servers. Returns [`McpStatusResponse`](#mcpstatusresponse) | | `reconnect_mcp_server(server_name)` | Retry connecting to an MCP server that failed or was disconnected | | `toggle_mcp_server(server_name, enabled)` | Enable or disable an MCP server mid-session. Disabling removes its tools | @@ -822,47 +822,47 @@ class ClaudeAgentOptions: | Property | Type | Default | Description | | :---------------------------- | :------------------------------------------------------------------------------------ | :--------------------------------- | :--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `tools` | `list[str] \| ToolsPreset \| None` | `None` | Tools configuration. Use `{"type": "preset", "preset": "claude_code"}` for Claude Code's default tools | -| `allowed_tools` | `list[str]` | `[]` | Tools to auto-approve without prompting. This does not restrict Claude to only these tools; unlisted tools fall through to `permission_mode` and `can_use_tool`. Use `disallowed_tools` to block tools. See [Permissions](/en/agent-sdk/permissions#allow-and-deny-rules) | +| `allowed_tools` | `list[str]` | `[]` | Tools to auto-approve without prompting. This does not restrict Claude to only these tools; unlisted tools fall through to `permission_mode` and `can_use_tool`. Use `disallowed_tools` to block tools. See [Permissions](/docs/en/agent-sdk/permissions#allow-and-deny-rules) | | `system_prompt` | `str \| SystemPromptPreset \| SystemPromptFile \| None` | `None` | System prompt configuration. Pass a string for a custom prompt, `{"type": "preset", "preset": "claude_code"}` for Claude Code's system prompt with optional `"append"`, or `{"type": "file", "path": "..."}` to load a large prompt from disk. See [`SystemPromptPreset`](#systempromptpreset) and [`SystemPromptFile`](#systempromptfile) | | `mcp_servers` | `dict[str, McpServerConfig] \| str \| Path` | `{}` | MCP server configurations or path to config file | -| `strict_mcp_config` | `bool` | `False` | When `True`, use only the servers passed in `mcp_servers` and ignore project `.mcp.json`, user settings, plugin-provided MCP servers, and [claude.ai connectors](/en/mcp#use-mcp-servers-from-claude-ai). Maps to the CLI `--strict-mcp-config` flag | +| `strict_mcp_config` | `bool` | `False` | When `True`, use only the servers passed in `mcp_servers` and ignore project `.mcp.json`, user settings, plugin-provided MCP servers, and [claude.ai connectors](/docs/en/mcp#use-mcp-servers-from-claude-ai). Maps to the CLI `--strict-mcp-config` flag | | `permission_mode` | `PermissionMode \| None` | `None` | Permission mode for tool usage | | `continue_conversation` | `bool` | `False` | Continue the most recent conversation | | `resume` | `str \| None` | `None` | Session ID to resume | | `session_id` | `str \| None` | `None` | Use a specific session ID instead of an auto-generated one. Must be a valid UUID. Can't be combined with `continue_conversation` or `resume` unless `fork_session` is also set | | `max_turns` | `int \| None` | `None` | Maximum agentic turns (tool-use round trips) | -| `max_budget_usd` | `float \| None` | `None` | Stop the query when the client-side cost estimate reaches this USD value. Compared against the same estimate as `total_cost_usd`; see [Track cost and usage](/en/agent-sdk/cost-tracking) for accuracy caveats | -| `disallowed_tools` | `list[str]` | `[]` | Tools to deny. A bare name such as `"Bash"` removes the tool from Claude's context. A scoped rule such as `"Bash(rm *)"` leaves the tool available and denies matching calls in every permission mode, including `bypassPermissions`. See [Permissions](/en/agent-sdk/permissions#allow-and-deny-rules) | -| `enable_file_checkpointing` | `bool` | `False` | Enable file change tracking for rewinding. See [File checkpointing](/en/agent-sdk/file-checkpointing) | -| `model` | `str \| None` | `None` | Claude model alias or full model name. See [accepted values and provider-specific IDs](/en/model-config#available-models) | +| `max_budget_usd` | `float \| None` | `None` | Stop the query when the client-side cost estimate reaches this USD value. Compared against the same estimate as `total_cost_usd`; see [Track cost and usage](/docs/en/agent-sdk/cost-tracking) for accuracy caveats | +| `disallowed_tools` | `list[str]` | `[]` | Tools to deny. A bare name such as `"Bash"` removes the tool from Claude's context. A scoped rule such as `"Bash(rm *)"` leaves the tool available and denies matching calls in every permission mode, including `bypassPermissions`. See [Permissions](/docs/en/agent-sdk/permissions#allow-and-deny-rules) | +| `enable_file_checkpointing` | `bool` | `False` | Enable file change tracking for rewinding. See [File checkpointing](/docs/en/agent-sdk/file-checkpointing) | +| `model` | `str \| None` | `None` | Claude model alias or full model name. See [accepted values and provider-specific IDs](/docs/en/model-config#available-models) | | `fallback_model` | `str \| None` | `None` | Fallback model to use if the primary model fails | | `betas` | `list[SdkBeta]` | `[]` | Beta features to enable. See [`SdkBeta`](#sdkbeta) for available options | -| `output_format` | `dict[str, Any] \| None` | `None` | Output format for structured responses (e.g., `{"type": "json_schema", "schema": {...}}`). See [Structured outputs](/en/agent-sdk/structured-outputs) for details | +| `output_format` | `dict[str, Any] \| None` | `None` | Output format for structured responses (e.g., `{"type": "json_schema", "schema": {...}}`). See [Structured outputs](/docs/en/agent-sdk/structured-outputs) for details | | `permission_prompt_tool_name` | `str \| None` | `None` | MCP tool name for permission prompts | | `cwd` | `str \| Path \| None` | `None` | Current working directory | | `cli_path` | `str \| Path \| None` | `None` | Custom path to the Claude Code CLI executable | | `settings` | `str \| None` | `None` | Path to settings file | | `add_dirs` | `list[str \| Path]` | `[]` | Additional directories Claude can access | -| `env` | `dict[str, str]` | `{}` | Environment variables merged on top of the inherited process environment. See [Environment variables](/en/env-vars) for variables the underlying CLI reads, and [Handle slow or stalled API responses](#handle-slow-or-stalled-api-responses) for timeout-related variables | +| `env` | `dict[str, str]` | `{}` | Environment variables merged on top of the inherited process environment. See [Environment variables](/docs/en/env-vars) for variables the underlying CLI reads, and [Handle slow or stalled API responses](#handle-slow-or-stalled-api-responses) for timeout-related variables | | `extra_args` | `dict[str, str \| None]` | `{}` | Additional CLI arguments to pass directly to the CLI | | `max_buffer_size` | `int \| None` | `None` | Maximum bytes when buffering CLI stdout | | `debug_stderr` | `Any` | `sys.stderr` | *Deprecated* - File-like object for debug output. Use `stderr` callback instead | | `stderr` | `Callable[[str], None] \| None` | `None` | Callback function for stderr output from CLI | -| `can_use_tool` | [`CanUseTool`](#canusetool) ` \| None` | `None` | Tool permission callback, invoked only when the [permission flow](/en/agent-sdk/permissions#how-permissions-are-evaluated) falls through to a prompt. Not invoked for calls auto-approved by `allowed_tools`, allow rules, or `permission_mode`. `AskUserQuestion`, connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools), and MCP tools marked [`requiresUserInteraction`](/en/mcp#require-approval-for-a-specific-tool) reach it even if you've allowed them; in `dontAsk` mode these are denied instead. See [`CanUseTool`](#canusetool) for details | +| `can_use_tool` | [`CanUseTool`](#canusetool) ` \| None` | `None` | Tool permission callback, invoked only when the [permission flow](/docs/en/agent-sdk/permissions#how-permissions-are-evaluated) falls through to a prompt. Not invoked for calls auto-approved by `allowed_tools`, allow rules, or `permission_mode`. `AskUserQuestion`, connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools), and MCP tools marked [`requiresUserInteraction`](/docs/en/mcp#require-approval-for-a-specific-tool) reach it even if you've allowed them; in `dontAsk` mode these are denied instead. See [`CanUseTool`](#canusetool) for details | | `hooks` | `dict[HookEvent, list[HookMatcher]] \| None` | `None` | Hook configurations for intercepting events | | `user` | `str \| None` | `None` | User identifier | | `include_partial_messages` | `bool` | `False` | Include partial message streaming events. When enabled, [`StreamEvent`](#streamevent) messages are yielded | | `include_hook_events` | `bool` | `False` | Include hook lifecycle events in the message stream as `HookEventMessage` objects | | `fork_session` | `bool` | `False` | When resuming with `resume`, fork to a new session ID instead of continuing the original session | | `agents` | `dict[str, AgentDefinition] \| None` | `None` | Programmatically defined subagents | -| `plugins` | `list[SdkPluginConfig]` | `[]` | Load custom plugins from local paths. See [Plugins](/en/agent-sdk/plugins) for details | +| `plugins` | `list[SdkPluginConfig]` | `[]` | Load custom plugins from local paths. See [Plugins](/docs/en/agent-sdk/plugins) for details | | `sandbox` | [`SandboxSettings`](#sandboxsettings) ` \| None` | `None` | Configure sandbox behavior programmatically. See [Sandbox settings](#sandboxsettings) for details | -| `setting_sources` | `list[SettingSource] \| None` | `None` (CLI defaults: all sources) | Control which filesystem settings to load. Pass `[]` to disable user, project, and local settings. Endpoint-managed policy loads regardless; server-managed settings are fetched when the session authenticates with an organization credential on an [eligible configuration](/en/server-managed-settings#platform-availability). See [Use Claude Code features](/en/agent-sdk/claude-code-features#what-settingsources-does-not-control) | -| `skills` | `list[str] \| Literal["all"] \| None` | `None` | Skills available to the session. Pass `"all"` to enable every discovered skill, or a list of skill names. When set, the SDK adds the Skill tool to `allowed_tools` automatically. If you also pass `tools`, include `"Skill"` in that list. See [Skills](/en/agent-sdk/skills) | +| `setting_sources` | `list[SettingSource] \| None` | `None` (CLI defaults: all sources) | Control which filesystem settings to load. Pass `[]` to disable user, project, and local settings. Endpoint-managed policy loads regardless; server-managed settings are fetched when the session authenticates with an organization credential on an [eligible configuration](/docs/en/server-managed-settings#platform-availability). See [Use Claude Code features](/docs/en/agent-sdk/claude-code-features#what-settingsources-does-not-control) | +| `skills` | `list[str] \| Literal["all"] \| None` | `None` | Skills available to the session. Pass `"all"` to enable every discovered skill, or a list of skill names. When set, the SDK adds the Skill tool to `allowed_tools` automatically. If you also pass `tools`, include `"Skill"` in that list. See [Skills](/docs/en/agent-sdk/skills) | | `max_thinking_tokens` | `int \| None` | `None` | *Deprecated* - Maximum tokens for thinking blocks. Use `thinking` instead | | `thinking` | [`ThinkingConfig`](#thinkingconfig) ` \| None` | `None` | Controls extended thinking behavior. Takes precedence over `max_thinking_tokens` | -| `effort` | [`EffortLevel`](#effortlevel) ` \| None` | `None` | Effort level for thinking depth. See [adjust the effort level](/en/model-config#adjust-effort-level) | -| `session_store` | [`SessionStore`](/en/agent-sdk/session-storage#the-sessionstore-interface) ` \| None` | `None` | Mirror session transcripts to an external backend so any host can resume them. See [Persist sessions to external storage](/en/agent-sdk/session-storage) | +| `effort` | [`EffortLevel`](#effortlevel) ` \| None` | `None` | Effort level for thinking depth. See [adjust the effort level](/docs/en/model-config#adjust-effort-level) | +| `session_store` | [`SessionStore`](/docs/en/agent-sdk/session-storage#the-sessionstore-interface) ` \| None` | `None` | Mirror session transcripts to an external backend so any host can resume them. See [Persist sessions to external storage](/docs/en/agent-sdk/session-storage) | | `session_store_flush` | `Literal["batched", "eager"]` | `"batched"` | When to flush mirrored transcript entries to `session_store`. `"batched"` flushes once per turn or when the buffer fills; `"eager"` triggers a background flush after every frame. Ignored when `session_store` is `None` | | `load_timeout_ms` | `int` | `60000` | Per-call timeout for `session_store.load()` and `list_subkeys()` during resume materialization, in milliseconds | | `task_budget` | `TaskBudget \| None` | `None` | API-side token budget. Sent as `output_config.task_budget` with the `task-budgets-2026-03-13` beta header. Pass `{"total": }`. | @@ -922,11 +922,11 @@ class SystemPromptPreset(TypedDict): | `type` | Yes | Must be `"preset"` to use a preset system prompt | | `preset` | Yes | Must be `"claude_code"` to use Claude Code's system prompt | | `append` | No | Additional instructions to append to the preset system prompt | -| `exclude_dynamic_sections` | No | Move per-session context such as working directory, the git-repo flag, and auto-memory paths from the system prompt into the first user message. Improves prompt-cache reuse across users and machines. See [Modify system prompts](/en/agent-sdk/modifying-system-prompts#improve-prompt-caching-across-users-and-machines) | +| `exclude_dynamic_sections` | No | Move per-session context such as working directory, the git-repo flag, and auto-memory paths from the system prompt into the first user message. Improves prompt-cache reuse across users and machines. See [Modify system prompts](/docs/en/agent-sdk/modifying-system-prompts#improve-prompt-caching-across-users-and-machines) | ### `SystemPromptFile` -Configuration for loading a custom system prompt from a file instead of passing it as a string. The SDK maps this to the CLI [`--system-prompt-file`](/en/cli-reference#system-prompt-flags) flag. Use the file form when the prompt is large: the SDK passes a string `system_prompt` on the CLI subprocess argv, which is subject to OS command-line length limits before the SDK sends any API request. On Linux a single argument longer than roughly 128 KB fails at process spawn with `Argument list too long`. On Windows the whole command line is capped at roughly 32 KB, so the string form fails at a lower threshold. +Configuration for loading a custom system prompt from a file instead of passing it as a string. The SDK maps this to the CLI [`--system-prompt-file`](/docs/en/cli-reference#system-prompt-flags) flag. Use the file form when the prompt is large: the SDK passes a string `system_prompt` on the CLI subprocess argv, which is subject to OS command-line length limits before the SDK sends any API request. On Linux a single argument longer than roughly 128 KB fails at process spawn with `Argument list too long`. On Windows the whole command line is capped at roughly 32 KB, so the string form fails at a lower threshold. ```python theme={null} class SystemPromptFile(TypedDict): @@ -955,7 +955,7 @@ SettingSource = Literal["user", "project", "local"] #### Default behavior -When `setting_sources` is omitted or `None`, `query()` loads the same filesystem settings as the Claude Code CLI: user, project, and local. Endpoint-managed policy is loaded in all cases; server-managed settings are fetched when the session authenticates with an organization credential on an [eligible configuration](/en/server-managed-settings#platform-availability). See [What settingSources does not control](/en/agent-sdk/claude-code-features#what-settingsources-does-not-control) for inputs that are read regardless of this option, and how to disable them. +When `setting_sources` is omitted or `None`, `query()` loads the same filesystem settings as the Claude Code CLI: user, project, and local. Endpoint-managed policy is loaded in all cases; server-managed settings are fetched when the session authenticates with an organization credential on an [eligible configuration](/docs/en/server-managed-settings#platform-availability). See [What settingSources does not control](/docs/en/agent-sdk/claude-code-features#what-settingsources-does-not-control) for inputs that are read regardless of this option, and how to disable them. #### Why use setting\_sources @@ -1201,9 +1201,9 @@ The callback receives: Returns a `PermissionResult` (either `PermissionResultAllow` or `PermissionResultDeny`). -The callback is the SDK replacement for the interactive permission prompt: it's invoked only when the [permission evaluation flow](/en/agent-sdk/permissions#how-permissions-are-evaluated) resolves to a prompt. Tool calls already approved by an `allowed_tools` entry, a settings allow rule, or the permission mode, such as `acceptEdits` or `bypassPermissions`, never invoke it. To gate every tool call, use a [`PreToolUse` hook](/en/agent-sdk/hooks) instead. +The callback is the SDK replacement for the interactive permission prompt: it's invoked only when the [permission evaluation flow](/docs/en/agent-sdk/permissions#how-permissions-are-evaluated) resolves to a prompt. Tool calls already approved by an `allowed_tools` entry, a settings allow rule, or the permission mode, such as `acceptEdits` or `bypassPermissions`, never invoke it. To gate every tool call, use a [`PreToolUse` hook](/docs/en/agent-sdk/hooks) instead. -`AskUserQuestion`, MCP tools marked [`requiresUserInteraction`](/en/mcp#require-approval-for-a-specific-tool), and connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools) reach the callback even when an allow rule matches. In `dontAsk` mode these calls are denied instead, without invoking the callback. +`AskUserQuestion`, MCP tools marked [`requiresUserInteraction`](/docs/en/mcp#require-approval-for-a-specific-tool), and connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools) reach the callback even when an allow rule matches. In `dontAsk` mode these calls are denied instead, without invoking the callback. ### `ToolPermissionContext` @@ -1533,7 +1533,7 @@ plugins = [ ] ``` -For complete information on creating and using plugins, see [Plugins](/en/agent-sdk/plugins). +For complete information on creating and using plugins, see [Plugins](/docs/en/agent-sdk/plugins). ## Message Types @@ -1668,12 +1668,12 @@ The `usage` dict contains the following keys when present: | Key | Type | Description | | ----------------------------- | ----- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `input_tokens` | `int` | Input tokens consumed by the top-level agent loop. [Subagent tokens aren't included](/en/agent-sdk/cost-tracking#get-the-total-cost-of-a-query); use `model_usage` for whole-tree accounting. | +| `input_tokens` | `int` | Input tokens consumed by the top-level agent loop. [Subagent tokens aren't included](/docs/en/agent-sdk/cost-tracking#get-the-total-cost-of-a-query); use `model_usage` for whole-tree accounting. | | `output_tokens` | `int` | Output tokens generated by the top-level agent loop. Subagent tokens aren't included. | | `cache_creation_input_tokens` | `int` | Tokens used to create new cache entries. | | `cache_read_input_tokens` | `int` | Tokens read from existing cache entries. | -The `model_usage` dict maps model names to per-model usage. The inner dict keys use camelCase because the value is passed through unmodified from the underlying CLI process, matching the TypeScript [`ModelUsage`](/en/agent-sdk/typescript#modelusage) type: +The `model_usage` dict maps model names to per-model usage. The inner dict keys use camelCase because the value is passed through unmodified from the underlying CLI process, matching the TypeScript [`ModelUsage`](/docs/en/agent-sdk/typescript#modelusage) type: | Key | Type | Description | | -------------------------- | ------- | ---------------------------------------------------------------------------------------------------------------------------------------- | @@ -1682,7 +1682,7 @@ The `model_usage` dict maps model names to per-model usage. The inner dict keys | `cacheReadInputTokens` | `int` | Cache read tokens for this model. | | `cacheCreationInputTokens` | `int` | Cache creation tokens for this model. | | `webSearchRequests` | `int` | Web search requests made by this model. | -| `costUSD` | `float` | Estimated cost in USD for this model, computed client-side. See [Track cost and usage](/en/agent-sdk/cost-tracking) for billing caveats. | +| `costUSD` | `float` | Estimated cost in USD for this model, computed client-side. See [Track cost and usage](/docs/en/agent-sdk/cost-tracking) for billing caveats. | | `contextWindow` | `int` | Context window size for this model. | | `maxOutputTokens` | `int` | Maximum output token limit for this model. | @@ -1969,7 +1969,7 @@ class CLIJSONDecodeError(ClaudeSDKError): ## Hook Types -For a comprehensive guide on using hooks with examples and common patterns, see the [Hooks guide](/en/agent-sdk/hooks). +For a comprehensive guide on using hooks with examples and common patterns, see the [Hooks guide](/docs/en/agent-sdk/hooks). ### `HookEvent` @@ -1991,7 +1991,7 @@ HookEvent = Literal[ ``` - The TypeScript SDK supports additional hook events not yet available in Python. See the [hook availability table](/en/agent-sdk/hooks#available-hooks) for per-SDK support. + The TypeScript SDK supports additional hook events not yet available in Python. See the [hook availability table](/docs/en/agent-sdk/hooks#available-hooks) for per-SDK support. ### `HookCallback` @@ -2315,7 +2315,7 @@ class SyncHookJSONOutput(TypedDict): #### `HookSpecificOutput` -A `TypedDict` containing the hook event name and event-specific fields. The shape depends on the `hookEventName` value. For full details on available fields per hook event, see [Control execution with hooks](/en/agent-sdk/hooks#outputs). +A `TypedDict` containing the hook event name and event-specific fields. The shape depends on the `hookEventName` value. For full details on available fields per hook event, see [Control execution with hooks](/docs/en/agent-sdk/hooks#outputs). A discriminated union of event-specific output types. The `hookEventName` field determines which fields are valid. @@ -2458,9 +2458,10 @@ Documentation of input/output schemas for all built-in Claude Code tools. While "prompt": str, # The task for the agent to perform "subagent_type": str | None, # The type of specialized agent to use "model": "sonnet" | "opus" | "haiku" | "fable" | None, # Model override for this agent - "run_in_background": bool | None, # Launch the agent in the background + "run_in_background": bool | None, # Agents run in the background by default; set to False to run synchronously "name": str | None, # Name for the spawned agent - "mode": "acceptEdits" | "auto" | "bypassPermissions" | "default" | "dontAsk" | "plan" | None, # Permission mode for the agent + "team_name": str | None, # Deprecated; ignored + "mode": "acceptEdits" | "auto" | "bypassPermissions" | "default" | "dontAsk" | "plan" | None, # Deprecated; ignored. Subagents inherit the parent session's permission mode; agent-definition frontmatter may override it "isolation": "worktree" | "remote" | None, # Isolation mode for the agent's changes } ``` @@ -2481,7 +2482,8 @@ Launches a new agent to handle complex, multi-step tasks autonomously. "citations": list | None, } ], - "resolvedModel": str | None, # Model the subagent actually ran on + "resolvedModel": str | None, # Model the subagent started on + "modelsUsed": list[str] | None, # Models used in order, with consecutive repeats collapsed "totalToolUseCount": int, # Number of tool calls the agent made "totalDurationMs": int, # Execution duration in milliseconds "totalTokens": int, # Total tokens used @@ -2521,7 +2523,8 @@ Launches a new agent to handle complex, multi-step tasks autonomously. "isAsync": bool | None, # True on background launches "agentId": str, # ID of the launched agent "description": str, # The task description - "resolvedModel": str | None, # Model the subagent runs on + "resolvedModel": str | None, # Model in use at the backgrounding transition + "modelsUsed": list[str] | None, # Models used before backgrounding, in order, with consecutive repeats collapsed "prompt": str, # The prompt the agent runs "outputFile": str, # File path where the agent's output is written "canReadOutputFile": bool | None, # Whether the output file can be read directly @@ -2543,13 +2546,13 @@ Launches a new agent to handle complex, multi-step tasks autonomously. Returns the result from the subagent. The output is discriminated on the `status` field: `"completed"` for finished tasks, `"async_launched"` for background tasks, and `"remote_launched"` for tasks Claude Code dispatched to a remote cloud session, where `sessionUrl` links to that session and `taskId` identifies it. Worktree-isolated runs include `worktreePath` and `worktreeBranch` on the `completed` variant. -The `resolvedModel` field on the `completed` and `async_launched` variants names the model the subagent actually ran on, which can differ from the requested `model` input when [`availableModels`](/en/model-config#restrict-model-selection) or another override applies. {/* min-version: 2.1.174 */}This field requires Claude Code v2.1.174 or later. +On the `completed` variant, `resolvedModel` names the model the subagent started on, which can differ from the requested `model` input when [`availableModels`](/docs/en/model-config#restrict-model-selection) or another override applies. {/* min-version: 2.1.174 */}This field requires Claude Code v2.1.174 or later. On the `async_launched` variant, `resolvedModel` names the model in use when the agent moved to the background, so a swap that happened before backgrounding is reflected there. The `modelsUsed` field on both variants lists the models used in order, with consecutive repeats collapsed; it's set only when the model was swapped mid-run. {/* min-version: 2.1.212 */}`modelsUsed` and the backgrounding-time `resolvedModel` behavior require Claude Code v2.1.212 or later. ### AskUserQuestion **Tool name:** `AskUserQuestion` -Asks the user clarifying questions during execution. See [Handle approvals and user input](/en/agent-sdk/user-input#handle-clarifying-questions) for usage details. +Asks the user clarifying questions during execution. See [Handle approvals and user input](/docs/en/agent-sdk/user-input#handle-clarifying-questions) for usage details. **Input:** @@ -2563,6 +2566,7 @@ Asks the user clarifying questions during execution. See [Handle approvals and u { "label": str, # Display text for this option (1-5 words) "description": str, # Explanation of what this option means + "preview": str | None, # Preview content rendered when the option is focused } ], "multiSelect": bool, # Set to true to allow multiple selections @@ -2572,6 +2576,11 @@ Asks the user clarifying questions during execution. See [Handle approvals and u # User answers populated by the permission system. Multi-select # answers are a comma-joined string of the selected labels; a # list of labels is accepted on input and coerced to that form + "annotations": dict[str, dict] | None, + # Per-question annotations from the user, keyed by question text. + # Each value can carry "preview" (the selected option's preview + # content) and "notes" (free-text notes on the selection) + "metadata": dict | None, # Analytics metadata, such as {"source": "remember"}; not displayed to the user } ``` @@ -2583,12 +2592,17 @@ Asks the user clarifying questions during execution. See [Handle approvals and u { "question": str, "header": str, - "options": [{"label": str, "description": str}], + "options": [{"label": str, "description": str, "preview": str | None}], "multiSelect": bool, } ], "answers": dict[str, str], # Maps question text to answer string # Multi-select answers are comma-separated + "response": str | None, + # Freeform reply typed instead of answering the questions; when set, + # Claude receives "The user responded: ..." in place of the answer list + "annotations": dict[str, dict] | None, # Per-question "preview" and "notes" from the user's selections + "afkTimeoutMs": int | None, # Set when the dialog auto-resolved after this many milliseconds of user inactivity; absent when the user answered } ``` @@ -2624,7 +2638,7 @@ Asks the user clarifying questions during execution. See [Handle approvals and u Runs a background source and delivers each event to Claude so it can react without polling: `command` runs a script and emits one event per stdout line, and `ws` opens a WebSocket and emits one event per text frame. Provide exactly one of `command` or `ws`. -When Monitor runs a command, it follows the same permission rules as Bash; a WebSocket watch prompts for approval separately. {/* min-version: 2.1.195 */}The `ws` source requires Claude Code v2.1.195 or later. See the [Monitor tool reference](/en/tools-reference#monitor-tool) for behavior and provider availability. +When Monitor runs a command, it follows the same permission rules as Bash; a WebSocket watch prompts for approval separately. {/* min-version: 2.1.195 */}The `ws` source requires Claude Code v2.1.195 or later. See the [Monitor tool reference](/docs/en/tools-reference#monitor-tool) for behavior and provider availability. **Input:** @@ -2884,7 +2898,7 @@ When Monitor runs a command, it follows the same permission rules as Bash; a Web **Tool name:** `TodoWrite` - As of Claude Code v2.1.142, `TodoWrite` is disabled by default. Use `TaskCreate`, `TaskGet`, `TaskUpdate`, and `TaskList` instead. See [Migrate to Task tools](/en/agent-sdk/todo-tracking#migrate-to-task-tools) to update your monitoring code, or set `CLAUDE_CODE_ENABLE_TASKS=0` to revert to `TodoWrite`. + As of Claude Code v2.1.142, `TodoWrite` is disabled by default. Use `TaskCreate`, `TaskGet`, `TaskUpdate`, and `TaskList` instead. See [Migrate to Task tools](/docs/en/agent-sdk/todo-tracking#migrate-to-task-tools) to update your monitoring code, or set `CLAUDE_CODE_ENABLE_TASKS=0` to revert to `TodoWrite`. **Input:** @@ -3528,7 +3542,7 @@ class SandboxSettings(TypedDict, total=False): The sandbox depends on platform support and, on Linux, tools like `bubblewrap` and `socat`. By default, when `enabled` is `True` but the sandbox can't start, commands run unsandboxed with a warning on stderr. This default differs from the TypeScript SDK, where `failIfUnavailable` defaults to `true`. - Set `"failIfUnavailable": True` in your sandbox settings to stop instead. The key isn't declared on `SandboxSettings` yet, but the SDK forwards it to Claude Code, which honors it. `query()` then reports a `ResultMessage` with `subtype="error_during_execution"` and the reason in `errors`. Because this is a single-shot `query()` call, the SDK raises after yielding that error result, so wrap the loop in a try block to continue past it. See [Handle the result](/en/agent-sdk/agent-loop#handle-the-result) for the error contract. + Set `"failIfUnavailable": True` in your sandbox settings to stop instead. The key isn't declared on `SandboxSettings` yet, but the SDK forwards it to Claude Code, which honors it. `query()` then reports a `ResultMessage` with `subtype="error_during_execution"` and the reason in `errors`. Because this is a single-shot `query()` call, the SDK raises after yielding that error result, so wrap the loop in a try block to continue past it. See [Handle the result](/docs/en/agent-sdk/agent-loop#handle-the-result) for the error contract. #### Example usage @@ -3567,7 +3581,7 @@ asyncio.run(main()) ### `SandboxNetworkConfig` -Network-specific configuration for sandbox mode. These settings apply to sandboxed Bash commands when `enabled` is `True` in the parent [`SandboxSettings`](#sandboxsettings). They do not restrict the WebFetch tool, which uses [permission rules](/en/permissions#webfetch) instead. +Network-specific configuration for sandbox mode. These settings apply to sandboxed Bash commands when `enabled` is `True` in the parent [`SandboxSettings`](#sandboxsettings). They do not restrict the WebFetch tool, which uses [permission rules](/docs/en/permissions#webfetch) instead. ```python theme={null} class SandboxNetworkConfig(TypedDict, total=False): @@ -3595,7 +3609,7 @@ class SandboxNetworkConfig(TypedDict, total=False): | `socksProxyPort` | `int` | `None` | SOCKS proxy port for network requests | - The built-in sandbox proxy enforces the network allowlist based on the requested hostname and does not terminate or inspect TLS traffic, so techniques such as [domain fronting](https://en.wikipedia.org/wiki/Domain_fronting) can potentially bypass it. See [Sandboxing security limitations](/en/sandboxing#security-limitations) for details and [Secure deployment](/en/agent-sdk/secure-deployment#traffic-forwarding) for configuring a TLS-terminating proxy. + The built-in sandbox proxy enforces the network allowlist based on the requested hostname and does not terminate or inspect TLS traffic, so techniques such as [domain fronting](https://en.wikipedia.org/wiki/Domain_fronting) can potentially bypass it. See [Sandboxing security limitations](/docs/en/sandboxing#security-limitations) for details and [Secure deployment](/docs/en/agent-sdk/secure-deployment#traffic-forwarding) for configuring a TLS-terminating proxy. ### `SandboxIgnoreViolations` @@ -3698,12 +3712,12 @@ This pattern enables you to: Commands running with `dangerouslyDisableSandbox: True` have full system access. Ensure your `can_use_tool` handler validates these requests carefully. - If `permission_mode` is set to `bypassPermissions` and `allow_unsandboxed_commands` is enabled, the model can autonomously execute commands outside the sandbox without approval prompts (an explicit [`ask` rule](/en/agent-sdk/permissions#how-permissions-are-evaluated) still forces one). This combination effectively allows the model to escape sandbox isolation silently. + If `permission_mode` is set to `bypassPermissions` and `allow_unsandboxed_commands` is enabled, the model can autonomously execute commands outside the sandbox without approval prompts (an explicit [`ask` rule](/docs/en/agent-sdk/permissions#how-permissions-are-evaluated) still forces one). This combination effectively allows the model to escape sandbox isolation silently. ## See also -* [SDK overview](/en/agent-sdk/overview) - General SDK concepts -* [TypeScript SDK reference](/en/agent-sdk/typescript) - TypeScript SDK documentation -* [CLI reference](/en/cli-reference) - Command-line interface -* [Common workflows](/en/common-workflows) - Step-by-step guides +* [SDK overview](/docs/en/agent-sdk/overview) - General SDK concepts +* [TypeScript SDK reference](/docs/en/agent-sdk/typescript) - TypeScript SDK documentation +* [CLI reference](/docs/en/cli-reference) - Command-line interface +* [Common workflows](/docs/en/common-workflows) - Step-by-step guides diff --git a/content/en/docs/claude-code/agent-sdk/quickstart.md b/content/en/docs/claude-code/agent-sdk/quickstart.md index 65d74aaf8..4a040401f 100644 --- a/content/en/docs/claude-code/agent-sdk/quickstart.md +++ b/content/en/docs/claude-code/agent-sdk/quickstart.md @@ -90,7 +90,7 @@ Use the Agent SDK to build an AI agent that reads your code, finds bugs, and fix - The TypeScript SDK bundles a native Claude Code binary for your platform as an optional dependency, so you don't need to install Claude Code separately. + Both the TypeScript and Python SDKs bundle a native Claude Code binary for your platform, so you don't need to install Claude Code separately. @@ -120,7 +120,7 @@ Use the Agent SDK to build an AI agent that reads your code, finds bugs, and fix * **Google Cloud's Agent Platform**: set `CLAUDE_CODE_USE_VERTEX=1` environment variable and configure Google Cloud credentials * **Microsoft Foundry**: set `CLAUDE_CODE_USE_FOUNDRY=1` environment variable and configure Azure credentials - See the setup guides for [Amazon Bedrock](/en/amazon-bedrock), [Claude Platform on AWS](/en/claude-platform-on-aws), [Google Cloud's Agent Platform](/en/google-vertex-ai), or [Microsoft Foundry](/en/microsoft-foundry) for details. + See the setup guides for [Amazon Bedrock](/docs/en/amazon-bedrock), [Claude Platform on AWS](/docs/en/claude-platform-on-aws), [Google Cloud's Agent Platform](/docs/en/google-vertex-ai), or [Microsoft Foundry](/docs/en/microsoft-foundry) for details. Unless previously approved, Anthropic does not allow third party developers to offer claude.ai login or rate limits for their products, including agents built on the Claude Agent SDK. Please use the API key authentication methods described in this document instead. @@ -211,18 +211,18 @@ Create `agent.py` if you're using the Python SDK, or `agent.ts` for TypeScript. This code has three main parts: -1. **`query`**: the main entry point that creates the agentic loop. It returns an async iterator, so you use `async for` to stream messages as Claude works. See the full API in the [Python](/en/agent-sdk/python#query) or [TypeScript](/en/agent-sdk/typescript#query) SDK reference. +1. **`query`**: the main entry point that creates the agentic loop. It returns an async iterator, so you use `async for` to stream messages as Claude works. See the full API in the [Python](/docs/en/agent-sdk/python#query) or [TypeScript](/docs/en/agent-sdk/typescript#query) SDK reference. 2. **`prompt`**: what you want Claude to do. Claude figures out which tools to use based on the task. -3. **`options`**: configuration for the agent. This example uses `allowedTools` to pre-approve `Read`, `Edit`, and `Glob`, and `permissionMode: "acceptEdits"` to auto-approve file changes. Other options include `systemPrompt`, `mcpServers`, and more. See all options for [Python](/en/agent-sdk/python#claudeagentoptions) or [TypeScript](/en/agent-sdk/typescript#options). +3. **`options`**: configuration for the agent. This example uses `allowedTools` to pre-approve `Read`, `Edit`, and `Glob`, and `permissionMode: "acceptEdits"` to auto-approve file changes. Other options include `systemPrompt`, `mcpServers`, and more. See all options for [Python](/docs/en/agent-sdk/python#claudeagentoptions) or [TypeScript](/docs/en/agent-sdk/typescript#options). The `async for` loop keeps running as Claude thinks, calls tools, observes results, and decides what to do next. Each iteration yields a message: Claude's reasoning, a tool call, a tool result, or the final outcome. The SDK handles the orchestration (tool execution, context management, retries) so you just consume the stream. The loop ends when Claude finishes the task or hits an error. The message handling inside the loop filters for human-readable output. Without filtering, you'd see raw message objects including system initialization and internal state, which is useful for debugging but noisy otherwise. - This example uses streaming to show progress in real-time. If you don't need live output (e.g., for background jobs or CI pipelines), you can collect all messages at once. See [Streaming vs. single-turn mode](/en/agent-sdk/streaming-vs-single-mode) for details. + This example uses streaming to show progress in real-time. If you don't need live output (e.g., for background jobs or CI pipelines), you can collect all messages at once. See [Streaming vs. single-turn mode](/docs/en/agent-sdk/streaming-vs-single-mode) for details. ### Run your agent @@ -262,7 +262,7 @@ As it works, the agent prints its reasoning and each tool it calls, ending with This is what makes the Agent SDK different: Claude executes tools directly instead of asking you to implement them. - If you see "API key not found", make sure you've set the `ANTHROPIC_API_KEY` environment variable in the shell where you run your agent. The SDK doesn't load `.env` files automatically. See the [full troubleshooting guide](/en/troubleshooting) for more help. + If you see "API key not found", make sure you've set the `ANTHROPIC_API_KEY` environment variable in the shell where you run your agent. The SDK doesn't load `.env` files automatically. See the [full troubleshooting guide](/docs/en/troubleshooting) for more help. ### Try other prompts @@ -355,20 +355,20 @@ With `Bash` enabled, try: `"Write unit tests for utils.py, run them, and fix any | ------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ----------------------------------------- | | `acceptEdits` | Auto-approves file edits and common filesystem commands, asks for other actions | Trusted development workflows | | `plan` | Runs read-only tools; file edits are never auto-approved and reach your `canUseTool` callback | Scoping a task before approving execution | -| `dontAsk` | Denies anything not in `allowedTools`; connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools) and tools that require user interaction are denied even if you've listed them | Locked-down headless agents | -| `auto` | A model classifier approves or denies each tool call | Autonomous agents with safety guardrails | -| `bypassPermissions` | Runs every tool without prompting, except tools matched by an explicit [`ask` rule](/en/agent-sdk/permissions#how-permissions-are-evaluated), connector tools [your organization set to `ask`](/en/mcp#organization-controls-on-connector-tools), and tools that require user interaction. In the TypeScript SDK, also requires `allowDangerouslySkipPermissions: true` in `options` | Sandboxed CI, fully trusted environments | +| `dontAsk` | Denies anything not in `allowedTools`; connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools) and tools that require user interaction are denied even if you've listed them | Locked-down headless agents | +| `auto` | A model classifier approves or denies permission prompts | Autonomous agents with safety guardrails | +| `bypassPermissions` | Runs every tool without prompting, except tools matched by an explicit [`ask` rule](/docs/en/agent-sdk/permissions#how-permissions-are-evaluated), connector tools [your organization set to `ask`](/docs/en/mcp#organization-controls-on-connector-tools), and tools that require user interaction. In the TypeScript SDK, also requires `allowDangerouslySkipPermissions: true` in `options` | Sandboxed CI, fully trusted environments | | `default` | Requires a `canUseTool` callback to handle approval | Custom approval flows | -The example above uses `acceptEdits` mode, which auto-approves file operations so the agent can run without interactive prompts. If you want to prompt users for approval, use `default` mode and provide a [`canUseTool` callback](/en/agent-sdk/user-input) that collects user input. For more control, see [Permissions](/en/agent-sdk/permissions). +The example above uses `acceptEdits` mode, which auto-approves file operations so the agent can run without interactive prompts. If you want to prompt users for approval, use `default` mode and provide a [`canUseTool` callback](/docs/en/agent-sdk/user-input) that collects user input. For more control, see [Permissions](/docs/en/agent-sdk/permissions). ## Next steps Now that you've created your first agent, learn how to extend its capabilities and tailor it to your use case: -* **[Permissions](/en/agent-sdk/permissions)**: control what your agent can do and when it needs approval -* **[Hooks](/en/agent-sdk/hooks)**: run custom code before or after tool calls -* **[Sessions](/en/agent-sdk/sessions)**: build multi-turn agents that maintain context -* **[MCP servers](/en/agent-sdk/mcp)**: connect to databases, browsers, APIs, and other external systems -* **[Hosting](/en/agent-sdk/hosting)**: deploy agents to Docker, cloud, and CI/CD +* **[Permissions](/docs/en/agent-sdk/permissions)**: control what your agent can do and when it needs approval +* **[Hooks](/docs/en/agent-sdk/hooks)**: run custom code before or after tool calls +* **[Sessions](/docs/en/agent-sdk/sessions)**: build multi-turn agents that maintain context +* **[MCP servers](/docs/en/agent-sdk/mcp)**: connect to databases, browsers, APIs, and other external systems +* **[Hosting](/docs/en/agent-sdk/hosting)**: deploy agents to Docker, cloud, and CI/CD * **[Example agents](https://github.com/anthropics/claude-agent-sdk-demos)**: see complete examples: email assistant, research agent, and more diff --git a/content/en/docs/claude-code/agent-sdk/secure-deployment.md b/content/en/docs/claude-code/agent-sdk/secure-deployment.md index 212d1e522..3d089f5d2 100644 --- a/content/en/docs/claude-code/agent-sdk/secure-deployment.md +++ b/content/en/docs/claude-code/agent-sdk/secure-deployment.md @@ -22,12 +22,12 @@ Defense in depth is still good practice though. For example, if an agent process ## Built-in security features -Claude Code includes several security features that address common concerns. See the [security documentation](/en/security) for full details. +Claude Code includes several security features that address common concerns. See the [security documentation](/docs/en/security) for full details. -* **Permissions system**: Every tool and bash command can be configured to allow, block, or prompt the user for approval. Use glob patterns to create rules like "allow all npm commands" or "block any command with sudo". Organizations can set policies that apply across all users. See [permissions](/en/permissions). +* **Permissions system**: Every tool and bash command can be configured to allow, block, or prompt the user for approval. Use glob patterns to create rules like "allow all npm commands" or "block any command with sudo". Organizations can set policies that apply across all users. See [permissions](/docs/en/permissions). * **Command parsing for permissions**: Before executing bash commands, Claude Code parses them into an AST and matches the result against your permission rules. Commands that cannot be parsed cleanly, or that do not match an allow rule, require explicit approval. A small set of constructs such as `eval` always require approval regardless of allow rules. This is a permission gate, not a sandbox; it does not infer whether a command is dangerous from its target path or effects. * **Web search summarization**: Search results are summarized rather than passing raw content directly into the context, reducing the risk of prompt injection from malicious web content. -* **Sandbox mode**: Bash commands can run in a sandboxed environment that restricts filesystem and network access. See the [sandboxing documentation](/en/sandboxing) for details. +* **Sandbox mode**: Bash commands can run in a sandboxed environment that restricts filesystem and network access. See the [sandboxing documentation](/docs/en/sandboxing) for details. ## Security principles @@ -100,7 +100,7 @@ Then create a configuration file specifying allowed paths and domains. 1. **Same-host kernel**: Unlike VMs, sandboxed processes share the host kernel. A kernel vulnerability could theoretically enable escape. For some threat models this is acceptable, but if you need kernel-level isolation, use gVisor or a separate VM. -2. **No TLS inspection**: The proxy allowlists domains based on the client-supplied hostname and does not terminate or inspect encrypted traffic. Code running inside the sandbox can potentially use [domain fronting](https://en.wikipedia.org/wiki/Domain_fronting) or similar techniques to reach hosts outside the allowlist. If your threat model requires stronger guarantees, configure a [TLS-terminating proxy](#traffic-forwarding). See the [sandboxing security limitations](/en/sandboxing#security-limitations) for more detail. Separately, if the agent has permissive credentials for an allowed domain, ensure it cannot use that domain to trigger other network requests or to exfiltrate data. +2. **No TLS inspection**: The proxy allowlists domains based on the client-supplied hostname and does not terminate or inspect encrypted traffic. Code running inside the sandbox can potentially use [domain fronting](https://en.wikipedia.org/wiki/Domain_fronting) or similar techniques to reach hosts outside the allowlist. If your threat model requires stronger guarantees, configure a [TLS-terminating proxy](#traffic-forwarding). See the [sandboxing security limitations](/docs/en/sandboxing#security-limitations) for more detail. Separately, if the agent has permissive credentials for an allowed domain, ensure it cannot use that domain to trigger other network requests or to exfiltrate data. For many single-developer and CI/CD use cases, sandbox-runtime raises the bar significantly with minimal setup. The sections below cover containers and VMs for deployments requiring stronger isolation. @@ -339,9 +339,9 @@ If you want to review changes before persisting them, an overlay filesystem lets ## Further reading -* [Claude Code security documentation](/en/security) -* [Hosting the Agent SDK](/en/agent-sdk/hosting) -* [Handling permissions](/en/agent-sdk/permissions) +* [Claude Code security documentation](/docs/en/security) +* [Hosting the Agent SDK](/docs/en/agent-sdk/hosting) +* [Handling permissions](/docs/en/agent-sdk/permissions) * [Sandbox runtime](https://github.com/anthropic-experimental/sandbox-runtime) * [The Lethal Trifecta for AI Agents](https://simonwillison.net/2025/Jun/16/the-lethal-trifecta/) * [OWASP Top 10 for LLM Applications](https://owasp.org/www-project-top-10-for-large-language-model-applications/) diff --git a/content/en/docs/claude-code/agent-sdk/session-storage.md b/content/en/docs/claude-code/agent-sdk/session-storage.md index e12340daa..a5713932a 100644 --- a/content/en/docs/claude-code/agent-sdk/session-storage.md +++ b/content/en/docs/claude-code/agent-sdk/session-storage.md @@ -276,23 +276,23 @@ The SDK never deletes from your store on its own. Retention is the adapter's res The following TypeScript SDK functions accept a `sessionStore` option and operate against the store instead of the local filesystem when it is provided: -* [`query()`](/en/agent-sdk/typescript#query) -* [`startup()`](/en/agent-sdk/typescript#startup) -* [`listSessions()`](/en/agent-sdk/typescript#listsessions) -* [`getSessionInfo()`](/en/agent-sdk/typescript#getsessioninfo) -* [`getSessionMessages()`](/en/agent-sdk/typescript#getsessionmessages) -* [`renameSession()`](/en/agent-sdk/typescript#renamesession) -* [`tagSession()`](/en/agent-sdk/typescript#tagsession) -* [`deleteSession()`](/en/agent-sdk/typescript) -* [`forkSession()`](/en/agent-sdk/typescript) -* [`listSubagents()`](/en/agent-sdk/typescript) -* [`getSubagentMessages()`](/en/agent-sdk/typescript) - -In the Python SDK, set `session_store` in [`ClaudeAgentOptions`](/en/agent-sdk/python#claudeagentoptions) to run `query()` against a store. The remaining operations each have a store-backed Python function that takes the store as an argument: `list_sessions_from_store()`, `get_session_info_from_store()`, `get_session_messages_from_store()`, `list_subagents_from_store()`, `get_subagent_messages_from_store()`, `rename_session_via_store()`, `tag_session_via_store()`, `delete_session_via_store()`, and `fork_session_via_store()`. `startup()` has no Python equivalent. The standalone functions documented in the [Python SDK reference](/en/agent-sdk/python#functions), such as `list_sessions()`, read local session files. +* [`query()`](/docs/en/agent-sdk/typescript#query) +* [`startup()`](/docs/en/agent-sdk/typescript#startup) +* [`listSessions()`](/docs/en/agent-sdk/typescript#listsessions) +* [`getSessionInfo()`](/docs/en/agent-sdk/typescript#getsessioninfo) +* [`getSessionMessages()`](/docs/en/agent-sdk/typescript#getsessionmessages) +* [`renameSession()`](/docs/en/agent-sdk/typescript#renamesession) +* [`tagSession()`](/docs/en/agent-sdk/typescript#tagsession) +* [`deleteSession()`](/docs/en/agent-sdk/typescript) +* [`forkSession()`](/docs/en/agent-sdk/typescript) +* [`listSubagents()`](/docs/en/agent-sdk/typescript) +* [`getSubagentMessages()`](/docs/en/agent-sdk/typescript) + +In the Python SDK, set `session_store` in [`ClaudeAgentOptions`](/docs/en/agent-sdk/python#claudeagentoptions) to run `query()` against a store. The remaining operations each have a store-backed Python function that takes the store as an argument: `list_sessions_from_store()`, `get_session_info_from_store()`, `get_session_messages_from_store()`, `list_subagents_from_store()`, `get_subagent_messages_from_store()`, `rename_session_via_store()`, `tag_session_via_store()`, `delete_session_via_store()`, and `fork_session_via_store()`. `startup()` has no Python equivalent. The standalone functions documented in the [Python SDK reference](/docs/en/agent-sdk/python#functions), such as `list_sessions()`, read local session files. ## Related resources -* [Work with sessions](/en/agent-sdk/sessions): Continue, resume, and fork without a custom store -* [Host the SDK](/en/agent-sdk/hosting): Deployment patterns for multi-host environments -* [TypeScript `Options`](/en/agent-sdk/typescript#options): Full option reference +* [Work with sessions](/docs/en/agent-sdk/sessions): Continue, resume, and fork without a custom store +* [Host the SDK](/docs/en/agent-sdk/hosting): Deployment patterns for multi-host environments +* [TypeScript `Options`](/docs/en/agent-sdk/typescript#options): Full option reference * [`examples/session-stores/`](https://github.com/anthropics/claude-agent-sdk-typescript/tree/main/examples/session-stores): Runnable S3, Redis, and Postgres reference adapters diff --git a/content/en/docs/claude-code/agent-sdk/sessions.md b/content/en/docs/claude-code/agent-sdk/sessions.md index 48f5e6756..8944f2453 100644 --- a/content/en/docs/claude-code/agent-sdk/sessions.md +++ b/content/en/docs/claude-code/agent-sdk/sessions.md @@ -11,14 +11,14 @@ A session is the conversation history the SDK accumulates while your agent works Returning to a session means the agent has full context from before: files it already read, analysis it already performed, decisions it already made. You can ask a follow-up question, recover from an interruption, or branch off to try a different approach. - Sessions persist the **conversation**, not the filesystem. To snapshot and revert file changes the agent made, use [file checkpointing](/en/agent-sdk/file-checkpointing). + Sessions persist the **conversation**, not the filesystem. To snapshot and revert file changes the agent made, use [file checkpointing](/docs/en/agent-sdk/file-checkpointing). This guide covers how to pick the right approach for your app, the SDK interfaces that track sessions automatically, how to capture session IDs and use `resume` and `fork` manually, and what to know about resuming sessions across hosts. ## Choose an approach -How much session handling you need depends on your application's shape. Session management comes into play when you send multiple prompts that should share context. Within a single `query()` call, the agent already takes as many turns as it needs, and permission prompts and `AskUserQuestion` are [handled in-loop](/en/agent-sdk/user-input) (they don't end the call). +How much session handling you need depends on your application's shape. Session management comes into play when you send multiple prompts that should share context. Within a single `query()` call, the agent already takes as many turns as it needs, and permission prompts and `AskUserQuestion` are [handled in-loop](/docs/en/agent-sdk/user-input) (they don't end the call). | What you're building | What to use | | :-------------------------------------------------------------------- | :--------------------------------------------------------------------------------------------------------------------------------------------------------------- | @@ -27,11 +27,11 @@ How much session handling you need depends on your application's shape. Session | Pick up where you left off after a process restart | `continue_conversation=True` (Python) / `continue: true` (TypeScript). Resumes the most recent session in the directory, no ID needed. | | Resume a specific past session (not the most recent) | Capture the session ID and pass it to `resume`. | | Try an alternative approach without losing the original | Fork the session. | -| Stateless task, don't want anything written to disk (TypeScript only) | Set [`persistSession: false`](/en/agent-sdk/typescript#options). The session exists only in memory for the duration of the call. Python always persists to disk. | +| Stateless task, don't want anything written to disk (TypeScript only) | Set [`persistSession: false`](/docs/en/agent-sdk/typescript#options). The session exists only in memory for the duration of the call. Python always persists to disk. | ### Continue, resume, and fork -Continue, resume, and fork are option fields you set on `query()` ([`ClaudeAgentOptions`](/en/agent-sdk/python#claudeagentoptions) in Python, [`Options`](/en/agent-sdk/typescript#options) in TypeScript). +Continue, resume, and fork are option fields you set on `query()` ([`ClaudeAgentOptions`](/docs/en/agent-sdk/python#claudeagentoptions) in Python, [`Options`](/docs/en/agent-sdk/typescript#options) in TypeScript). **Continue** and **resume** both pick up an existing session and add to it. The difference is how they find that session: @@ -46,7 +46,7 @@ Both SDKs offer an interface that tracks session state for you across calls, so ### Python: `ClaudeSDKClient` -[`ClaudeSDKClient`](/en/agent-sdk/python#claudesdkclient) handles session IDs internally. Each call to `client.query()` automatically continues the same session. Call [`client.receive_response()`](/en/agent-sdk/python#claudesdkclient) to iterate over the messages for the current query. Use the client as an async context manager so connection setup and teardown are handled for you, or call `connect()` and `disconnect()` manually. +[`ClaudeSDKClient`](/docs/en/agent-sdk/python#claudesdkclient) handles session IDs internally. Each call to `client.query()` automatically continues the same session. Call [`client.receive_response()`](/docs/en/agent-sdk/python#claudesdkclient) to iterate over the messages for the current query. Use the client as an async context manager so connection setup and teardown are handled for you, or call `connect()` and `disconnect()` manually. This example runs two queries against the same `client`. The first asks the agent to analyze a module; the second asks it to refactor that module. Because both calls go through the same client instance, the second query has full context from the first without any explicit `resume` or session ID: @@ -98,7 +98,7 @@ asyncio.run(main()) Each query prints the agent's text response followed by a status line from the result message, such as `[done: success, cost: $0.0042]`. -See the [Python SDK reference](/en/agent-sdk/python#choosing-between-query-and-claudesdkclient) for details on when to use `ClaudeSDKClient` vs the standalone `query()` function. +See the [Python SDK reference](/docs/en/agent-sdk/python#choosing-between-query-and-claudesdkclient) for details on when to use `ClaudeSDKClient` vs the standalone `query()` function. ### TypeScript: `continue: true` @@ -140,14 +140,14 @@ for await (const message of query({ ``` - The experimental [V2 session API](/en/agent-sdk/typescript-v2-preview), which provided `createSession()` with a `send` / `stream` pattern, was removed in TypeScript Agent SDK 0.3.142. Use the `query()` function and the session options described on this page instead. + The experimental [V2 session API](/docs/en/agent-sdk/typescript-v2-preview), which provided `createSession()` with a `send` / `stream` pattern, was removed in TypeScript Agent SDK 0.3.142. Use the `query()` function and the session options described on this page instead. ## Use session options with `query()` ### Capture the session ID -Resume and fork require a session ID. Read it from the `session_id` field on the result message ([`ResultMessage`](/en/agent-sdk/python#resultmessage) in Python, [`SDKResultMessage`](/en/agent-sdk/typescript#sdkresultmessage) in TypeScript), which is present on every result regardless of success or error. In TypeScript the ID is also available earlier as a direct field on the init `SystemMessage`; in Python it's nested inside `SystemMessage.data`. +Resume and fork require a session ID. Read it from the `session_id` field on the result message ([`ResultMessage`](/docs/en/agent-sdk/python#resultmessage) in Python, [`SDKResultMessage`](/docs/en/agent-sdk/typescript#sdkresultmessage) in TypeScript), which is present on every result regardless of success or error. In TypeScript the ID is also available earlier as a direct field on the init `SystemMessage`; in Python it's nested inside `SystemMessage.data`. ```python Python theme={null} @@ -219,7 +219,7 @@ When the query completes, the script prints the agent's response followed by a l Pass a session ID to `resume` to return to that specific session. The agent picks up with full context from wherever the session left off. Common reasons to resume: * **Follow up on a completed task.** The agent already analyzed something; now you want it to act on that analysis without re-reading files. -* **Recover from a limit.** The first run ended with `error_max_turns` or `error_max_budget_usd` (see [Handle the result](/en/agent-sdk/agent-loop#handle-the-result)); resume with a higher limit. In a single-shot `query()` call the SDK raises after yielding that error result, so catch the error before resuming. +* **Recover from a limit.** The first run ended with `error_max_turns` or `error_max_budget_usd` (see [Handle the result](/docs/en/agent-sdk/agent-loop#handle-the-result)); resume with a higher limit. In a single-shot `query()` call the SDK raises after yielding that error result, so catch the error before resuming. * **Restart your process.** You captured the ID before shutdown and want to restore the conversation. This example resumes the session from [Capture the session ID](#capture-the-session-id) with a follow-up prompt. Because you're resuming, the agent already has the prior analysis in context: @@ -274,14 +274,14 @@ You should see a response that builds on the earlier analysis instead of startin If a `resume` call returns a fresh session instead of the expected history, the most common cause is a mismatched `cwd`. Sessions are stored under `~/.claude/projects//*.jsonl`, or under `$CLAUDE_CONFIG_DIR/projects//*.jsonl` if you set the `CLAUDE_CONFIG_DIR` environment variable, where `` is the absolute working directory with every non-alphanumeric character replaced by `-` (so `/Users/me/proj` becomes `-Users-me-proj`). If your resume call runs from a different directory, the SDK looks in the wrong place. The session file also needs to exist on the current machine. -To resume sessions across machines or in serverless environments, mirror transcripts to shared storage with a [`SessionStore` adapter](/en/agent-sdk/session-storage). +To resume sessions across machines or in serverless environments, mirror transcripts to shared storage with a [`SessionStore` adapter](/docs/en/agent-sdk/session-storage). ### Fork to explore alternatives Forking creates a new session that starts with a copy of the original's history but diverges from that point. The fork gets its own session ID; the original's ID and history stay unchanged. You end up with two independent sessions you can resume separately. - Forking branches the conversation history, not the filesystem. If a forked agent edits files, those changes are real and visible to any session working in the same directory. To branch and revert file changes, use [file checkpointing](/en/agent-sdk/file-checkpointing). + Forking branches the conversation history, not the filesystem. If a forked agent edits files, those changes are real and visible to any session working in the same directory. To branch and revert file changes, use [file checkpointing](/docs/en/agent-sdk/file-checkpointing). This example builds on [Capture the session ID](#capture-the-session-id): you've already analyzed an auth module in `session_id` and want to explore OAuth2 without losing the JWT-focused thread. The first block forks the session and captures the fork's ID (`forked_id`); the second block resumes the original `session_id` to continue down the JWT path. You now have two session IDs pointing at two separate histories: @@ -393,13 +393,13 @@ Session files are local to the machine that created them. To resume a session on * **Move the session file.** Persist `~/.claude/projects//.jsonl` from the first run and restore it to the same path on the new host before calling `resume`. The `cwd` must match. * **Don't rely on session resume.** Capture the results you need (analysis output, decisions, file diffs) as application state and pass them into a fresh session's prompt. This is often more robust than shipping transcript files around. -Both SDKs expose functions for enumerating sessions on disk and reading their messages: [`listSessions()`](/en/agent-sdk/typescript#listsessions) and [`getSessionMessages()`](/en/agent-sdk/typescript#getsessionmessages) in TypeScript, [`list_sessions()`](/en/agent-sdk/python#list_sessions) and [`get_session_messages()`](/en/agent-sdk/python#get_session_messages) in Python. Use them to build custom session pickers, cleanup logic, or transcript viewers. +Both SDKs expose functions for enumerating sessions on disk and reading their messages: [`listSessions()`](/docs/en/agent-sdk/typescript#listsessions) and [`getSessionMessages()`](/docs/en/agent-sdk/typescript#getsessionmessages) in TypeScript, [`list_sessions()`](/docs/en/agent-sdk/python#list_sessions) and [`get_session_messages()`](/docs/en/agent-sdk/python#get_session_messages) in Python. Use them to build custom session pickers, cleanup logic, or transcript viewers. -Both SDKs also expose functions for looking up and mutating individual sessions: [`get_session_info()`](/en/agent-sdk/python#get_session_info), [`rename_session()`](/en/agent-sdk/python#rename_session), and [`tag_session()`](/en/agent-sdk/python#tag_session) in Python, and [`getSessionInfo()`](/en/agent-sdk/typescript#getsessioninfo), [`renameSession()`](/en/agent-sdk/typescript#renamesession), and [`tagSession()`](/en/agent-sdk/typescript#tagsession) in TypeScript. Use them to organize sessions by tag or give them human-readable titles. +Both SDKs also expose functions for looking up and mutating individual sessions: [`get_session_info()`](/docs/en/agent-sdk/python#get_session_info), [`rename_session()`](/docs/en/agent-sdk/python#rename_session), and [`tag_session()`](/docs/en/agent-sdk/python#tag_session) in Python, and [`getSessionInfo()`](/docs/en/agent-sdk/typescript#getsessioninfo), [`renameSession()`](/docs/en/agent-sdk/typescript#renamesession), and [`tagSession()`](/docs/en/agent-sdk/typescript#tagsession) in TypeScript. Use them to organize sessions by tag or give them human-readable titles. ## Related resources -* [How the agent loop works](/en/agent-sdk/agent-loop): Understand turns, messages, and context accumulation within a session -* [File checkpointing](/en/agent-sdk/file-checkpointing): Snapshot and revert file changes the agent made within a session -* [Python `ClaudeAgentOptions`](/en/agent-sdk/python#claudeagentoptions): Full session option reference for Python -* [TypeScript `Options`](/en/agent-sdk/typescript#options): Full session option reference for TypeScript +* [How the agent loop works](/docs/en/agent-sdk/agent-loop): Understand turns, messages, and context accumulation within a session +* [File checkpointing](/docs/en/agent-sdk/file-checkpointing): Snapshot and revert file changes the agent made within a session +* [Python `ClaudeAgentOptions`](/docs/en/agent-sdk/python#claudeagentoptions): Full session option reference for Python +* [TypeScript `Options`](/docs/en/agent-sdk/typescript#options): Full session option reference for TypeScript diff --git a/content/en/docs/claude-code/agent-sdk/skills.md b/content/en/docs/claude-code/agent-sdk/skills.md index 59f9a1ed3..dd50c1e4d 100644 --- a/content/en/docs/claude-code/agent-sdk/skills.md +++ b/content/en/docs/claude-code/agent-sdk/skills.md @@ -25,7 +25,7 @@ When using the Claude Agent SDK, Skills are: Unlike subagents (which can be defined programmatically), Skills must be created as filesystem artifacts. The SDK does not provide a programmatic API for registering Skills. - Skills are discovered through the filesystem setting sources. With default `query()` options, the SDK loads user and project sources, so skills in `~/.claude/skills/`, `/.claude/skills/`, and `.claude/skills/` in any parent directory of `` up to the repository root are available. If you set `settingSources` explicitly, include `'user'` or `'project'` to keep skill discovery, or use the [`plugins` option](/en/agent-sdk/plugins) to load skills from a specific path. + Skills are discovered through the filesystem setting sources. With default `query()` options, the SDK loads user and project sources, so skills in `~/.claude/skills/`, `/.claude/skills/`, and `.claude/skills/` in any parent directory of `` up to the repository root are available. If you set `settingSources` explicitly, include `'user'` or `'project'` to keep skill discovery, or use the [`plugins` option](/docs/en/agent-sdk/plugins) to load skills from a specific path. ## Using Skills with the SDK @@ -109,7 +109,7 @@ Skills are defined as directories containing a `SKILL.md` file with YAML frontma For complete guidance on creating Skills, including SKILL.md structure, multi-file Skills, and examples, see: -* [Agent Skills in Claude Code](/en/skills): Complete guide with examples +* [Agent Skills in Claude Code](/docs/en/skills): Complete guide with examples * [Agent Skills Best Practices](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices): Authoring guidelines and naming conventions ## Tool Restrictions @@ -250,7 +250,7 @@ Claude automatically invokes the relevant Skill if the description matches your ``` -For more details on `settingSources`/`setting_sources`, see the [TypeScript SDK reference](/en/agent-sdk/typescript#settingsource) or [Python SDK reference](/en/agent-sdk/python#settingsource). +For more details on `settingSources`/`setting_sources`, see the [TypeScript SDK reference](/docs/en/agent-sdk/typescript#settingsource) or [Python SDK reference](/docs/en/agent-sdk/python#settingsource). **Check working directory**: The SDK loads Skills from `.claude/skills/` in the `cwd` option and in every parent directory up to the repository root. Ensure `cwd` points at or below the directory containing `.claude/skills/`, within the same repository: @@ -294,21 +294,21 @@ ls ~/.claude/skills/*/SKILL.md ### Additional Troubleshooting -For general Skills troubleshooting (YAML syntax, debugging, etc.), see the [Claude Code Skills troubleshooting section](/en/skills#troubleshooting). +For general Skills troubleshooting (YAML syntax, debugging, etc.), see the [Claude Code Skills troubleshooting section](/docs/en/skills#troubleshooting). ## Related Documentation ### Skills Guides -* [Agent Skills in Claude Code](/en/skills): Complete Skills guide with creation, examples, and troubleshooting +* [Agent Skills in Claude Code](/docs/en/skills): Complete Skills guide with creation, examples, and troubleshooting * [Agent Skills Overview](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/overview): Conceptual overview, benefits, and architecture * [Agent Skills Best Practices](https://platform.claude.com/docs/en/agents-and-tools/agent-skills/best-practices): Authoring guidelines for effective Skills * [Agent Skills Cookbook](https://platform.claude.com/cookbook/skills-notebooks-01-skills-introduction): Example Skills and templates ### SDK Resources -* [Subagents in the SDK](/en/agent-sdk/subagents): Similar filesystem-based agents with programmatic options -* [Slash Commands in the SDK](/en/agent-sdk/slash-commands): User-invoked commands -* [SDK Overview](/en/agent-sdk/overview): General SDK concepts -* [TypeScript SDK Reference](/en/agent-sdk/typescript): Complete API documentation -* [Python SDK Reference](/en/agent-sdk/python): Complete API documentation +* [Subagents in the SDK](/docs/en/agent-sdk/subagents): Similar filesystem-based agents with programmatic options +* [Slash Commands in the SDK](/docs/en/agent-sdk/slash-commands): User-invoked commands +* [SDK Overview](/docs/en/agent-sdk/overview): General SDK concepts +* [TypeScript SDK Reference](/docs/en/agent-sdk/typescript): Complete API documentation +* [Python SDK Reference](/docs/en/agent-sdk/python): Complete API documentation diff --git a/content/en/docs/claude-code/agent-sdk/slash-commands.md b/content/en/docs/claude-code/agent-sdk/slash-commands.md index 9b6e7918f..66c49ddeb 100644 --- a/content/en/docs/claude-code/agent-sdk/slash-commands.md +++ b/content/en/docs/claude-code/agent-sdk/slash-commands.md @@ -119,7 +119,7 @@ Send slash commands by including them in your prompt string, just like regular t After yielding that final result message, the SDK raises an error, because the CLI process exits with a non-zero code. - Wrap the loop in a `try`/`catch` in TypeScript or `try`/`except` in Python if your command might hit the limit, as shown in [Single Message Input](/en/agent-sdk/streaming-vs-single-mode#single-message-input), or set `maxTurns` high enough for the work to complete. In Python, catch `Exception`: the SDK surfaces error results as a plain `Exception`. + Wrap the loop in a `try`/`catch` in TypeScript or `try`/`except` in Python if your command might hit the limit, as shown in [Single Message Input](/docs/en/agent-sdk/streaming-vs-single-mode#single-message-input), or set `maxTurns` high enough for the work to complete. In Python, catch `Exception`: the SDK surfaces error results as a plain `Exception`. ## Common Slash Commands @@ -204,14 +204,14 @@ The `/compact` command reduces the size of your conversation history by summariz - A `compact_boundary` message only arrives when compaction ran. With nothing to summarize, `/compact` reports the reason instead of raising: the run still ends with a `success` result, no `compact_boundary` message is emitted, and the result text carries the message, for example `Not enough messages to compact.` after a single short exchange. A fresh one-shot `query()` call starts with empty context, so use this pattern in a session with prior turns, for example in [streaming input mode](/en/agent-sdk/streaming-vs-single-mode) or when resuming a session. + A `compact_boundary` message only arrives when compaction ran. With nothing to summarize, `/compact` reports the reason instead of raising: the run still ends with a `success` result, no `compact_boundary` message is emitted, and the result text carries the message, for example `Not enough messages to compact.` after a single short exchange. A fresh one-shot `query()` call starts with empty context, so use this pattern in a session with prior turns, for example in [streaming input mode](/docs/en/agent-sdk/streaming-vs-single-mode) or when resuming a session. ### `/clear` - Reset conversation context -The `/clear` command resets the conversation to an empty context, so subsequent prompts start with no prior conversation history. The previous conversation remains on disk and can be returned to by passing its session ID to the [`resume` option](/en/agent-sdk/sessions#resume-by-id). +The `/clear` command resets the conversation to an empty context, so subsequent prompts start with no prior conversation history. The previous conversation remains on disk and can be returned to by passing its session ID to the [`resume` option](/docs/en/agent-sdk/sessions#resume-by-id). -This is useful in [streaming input mode](/en/agent-sdk/streaming-vs-single-mode), where you send multiple prompts over a single connection. For one-shot `query()` calls, each call already starts with empty context, so sending `/clear` has no practical effect; start a new `query()` instead. +This is useful in [streaming input mode](/docs/en/agent-sdk/streaming-vs-single-mode), where you send multiple prompts over a single connection. For one-shot `query()` calls, each call already starts with empty context, so sending `/clear` has no practical effect; start a new `query()` instead. `/clear` in the SDK requires Claude Code v2.1.117 or later. In earlier versions it is omitted from `slash_commands`. @@ -222,7 +222,7 @@ This is useful in [streaming input mode](/en/agent-sdk/streaming-vs-single-mode) In addition to using built-in slash commands, you can create your own custom commands that are available through the SDK. Custom commands are defined as markdown files in specific directories, similar to how subagents are configured. - The `.claude/commands/` directory is the legacy format. The recommended format is `.claude/skills//SKILL.md`, which supports the same slash-command invocation (`/name`) plus autonomous invocation by Claude. See [Skills](/en/agent-sdk/skills) for the current format. The CLI continues to support both formats, and the examples below remain accurate for `.claude/commands/`. + The `.claude/commands/` directory is the legacy format. The recommended format is `.claude/skills//SKILL.md`, which supports the same slash-command invocation (`/name`) plus autonomous invocation by Claude. See [Skills](/docs/en/agent-sdk/skills) for the current format. The CLI continues to support both formats, and the examples below remain accurate for `.claude/commands/`. ### File Locations @@ -568,8 +568,8 @@ Use these commands through the SDK: ## See Also -* [Slash Commands](/en/skills) - Complete slash command documentation -* [Subagents in the SDK](/en/agent-sdk/subagents) - Similar filesystem-based configuration for subagents -* [TypeScript SDK reference](/en/agent-sdk/typescript) - Complete API documentation -* [SDK overview](/en/agent-sdk/overview) - General SDK concepts -* [CLI reference](/en/cli-reference) - Command-line interface +* [Slash Commands](/docs/en/skills) - Complete slash command documentation +* [Subagents in the SDK](/docs/en/agent-sdk/subagents) - Similar filesystem-based configuration for subagents +* [TypeScript SDK reference](/docs/en/agent-sdk/typescript) - Complete API documentation +* [SDK overview](/docs/en/agent-sdk/overview) - General SDK concepts +* [CLI reference](/docs/en/cli-reference) - Command-line interface diff --git a/content/en/docs/claude-code/agent-sdk/streaming-output.md b/content/en/docs/claude-code/agent-sdk/streaming-output.md index 13f3b1967..4f792a655 100644 --- a/content/en/docs/claude-code/agent-sdk/streaming-output.md +++ b/content/en/docs/claude-code/agent-sdk/streaming-output.md @@ -9,7 +9,7 @@ By default, the Agent SDK yields complete `AssistantMessage` objects after Claude finishes generating each response. To receive incremental updates as text and tool calls are generated, enable partial message streaming by setting `include_partial_messages` (Python) or `includePartialMessages` (TypeScript) to `true` in your options. - This page covers output streaming (receiving tokens in real-time). For input modes (how you send messages), see [Send messages to agents](/en/agent-sdk/streaming-vs-single-mode). You can also [stream responses using the Agent SDK via the CLI](/en/headless). + This page covers output streaming (receiving tokens in real-time). For input modes (how you send messages), see [Send messages to agents](/docs/en/agent-sdk/streaming-vs-single-mode). You can also [stream responses using the Agent SDK via the CLI](/docs/en/headless). ## Enable streaming output @@ -102,7 +102,7 @@ Both contain raw Claude API events, not accumulated text. You need to extract an ``` -The `parent_tool_use_id` field is always `None` in Python and `null` in TypeScript. Stream events are emitted for the main session only; token-level deltas from subagents aren't forwarded. To attribute output to a subagent, use complete messages, which carry `parent_tool_use_id`. See [Detect subagent invocation](/en/agent-sdk/subagents#detect-subagent-invocation). +The `parent_tool_use_id` field is always `None` in Python and `null` in TypeScript. Stream events are emitted for the main session only; token-level deltas from subagents aren't forwarded. To attribute output to a subagent, use complete messages, which carry `parent_tool_use_id`. See [Detect subagent invocation](/docs/en/agent-sdk/subagents#detect-subagent-invocation). The `event` field contains the raw streaming event from the [Claude API](https://platform.claude.com/docs/en/build-with-claude/streaming#event-types). Common event types include: @@ -385,12 +385,12 @@ This example combines text and tool streaming into a cohesive UI. It tracks whet ## Known limitations -* **Structured output**: the JSON result appears only in the final `ResultMessage.structured_output`, not as streaming deltas. See [structured outputs](/en/agent-sdk/structured-outputs) for details. +* **Structured output**: the JSON result appears only in the final `ResultMessage.structured_output`, not as streaming deltas. See [structured outputs](/docs/en/agent-sdk/structured-outputs) for details. ## Next steps Now that you can stream text and tool calls in real-time, explore these related topics: -* [Interactive vs one-shot queries](/en/agent-sdk/streaming-vs-single-mode): choose between input modes for your use case -* [Structured outputs](/en/agent-sdk/structured-outputs): get typed JSON responses from the agent -* [Permissions](/en/agent-sdk/permissions): control which tools the agent can use +* [Interactive vs one-shot queries](/docs/en/agent-sdk/streaming-vs-single-mode): choose between input modes for your use case +* [Structured outputs](/docs/en/agent-sdk/structured-outputs): get typed JSON responses from the agent +* [Permissions](/docs/en/agent-sdk/permissions): control which tools the agent can use diff --git a/content/en/docs/claude-code/agent-sdk/streaming-vs-single-mode.md b/content/en/docs/claude-code/agent-sdk/streaming-vs-single-mode.md index bb066178e..63152cddb 100644 --- a/content/en/docs/claude-code/agent-sdk/streaming-vs-single-mode.md +++ b/content/en/docs/claude-code/agent-sdk/streaming-vs-single-mode.md @@ -210,7 +210,7 @@ These examples read an image named `diagram.png` from the working directory. Cre ``` -When you run the example, the TypeScript version prints each response as it completes. The Python version's `receive_response()` loop ends at the first result message, so it prints the security analysis; to read both responses, use one `query()` and `receive_response()` pair per message as shown in the [Python reference's example of continuing a conversation](/en/agent-sdk/python#example-continuing-a-conversation). +When you run the example, the TypeScript version prints each response as it completes. The Python version's `receive_response()` loop ends at the first result message, so it prints the security analysis; to read both responses, use one `query()` and `receive_response()` pair per message as shown in the [Python reference's example of continuing a conversation](/docs/en/agent-sdk/python#example-continuing-a-conversation). In the TypeScript SDK, if your message generator throws, for example when a file it reads is missing, the stream ends with an error that reads `Claude Code process aborted by user` instead of the original error, so check the code inside your generator first when you see that message. The error may also be preceded by a long minified line of bundled SDK source, so read to the end of the output for the error text. @@ -241,7 +241,7 @@ Use single message input when: * Natural multi-turn conversations -If a query ends with an error result, such as `error_max_turns`, a single message `query()` call raises an error that includes the failure text after yielding the final result message, so wrap the loop in a try block if your code needs to continue. See [Handle the result](/en/agent-sdk/agent-loop#handle-the-result) for the result subtypes. +If a query ends with an error result, such as `error_max_turns`, a single message `query()` call raises an error that includes the failure text after yielding the final result message, so wrap the loop in a try block if your code needs to continue. See [Handle the result](/docs/en/agent-sdk/agent-loop#handle-the-result) for the result subtypes. ### Implementation Example diff --git a/content/en/docs/claude-code/agent-sdk/structured-outputs.md b/content/en/docs/claude-code/agent-sdk/structured-outputs.md index 6d96ba699..7b07513f0 100644 --- a/content/en/docs/claude-code/agent-sdk/structured-outputs.md +++ b/content/en/docs/claude-code/agent-sdk/structured-outputs.md @@ -385,7 +385,7 @@ The schema includes optional fields (`author` and `date`) since git blame inform ## Error handling -Structured output generation can fail when the agent cannot produce valid JSON matching your schema. This typically happens when the schema is too complex for the task, the task itself is ambiguous, or the agent hits its retry limit trying to fix validation errors. It can also happen without any validation failure: a [model fallback](/en/model-config#automatic-model-fallback) can retract an already-completed output mid-stream, and if no retry replaces it the run ends with the same error. Check the `errors` list on the result message to tell the two causes apart before debugging your schema. +Structured output generation can fail when the agent cannot produce valid JSON matching your schema. This typically happens when the schema is too complex for the task, the task itself is ambiguous, or the agent hits its retry limit trying to fix validation errors. It can also happen without any validation failure: a [model fallback](/docs/en/model-config#automatic-model-fallback) can retract an already-completed output mid-stream, and if no retry replaces it the run ends with the same error. Check the `errors` list on the result message to tell the two causes apart before debugging your schema. When an error occurs, the result message has a `subtype` indicating what went wrong: @@ -493,4 +493,4 @@ A result can also end with subtype `success` but no `structured_output` value, f * [JSON Schema documentation](https://json-schema.org/): learn JSON Schema syntax for defining complex schemas with nested objects, arrays, enums, and validation constraints * [API Structured Outputs](https://platform.claude.com/docs/en/build-with-claude/structured-outputs): use structured outputs with the Claude API directly for single-turn requests without tool use -* [Custom tools](/en/agent-sdk/custom-tools): give your agent custom tools to call during execution before returning structured output +* [Custom tools](/docs/en/agent-sdk/custom-tools): give your agent custom tools to call during execution before returning structured output diff --git a/content/en/docs/claude-code/agent-sdk/subagents.md b/content/en/docs/claude-code/agent-sdk/subagents.md index b02da4ae1..986cc012b 100644 --- a/content/en/docs/claude-code/agent-sdk/subagents.md +++ b/content/en/docs/claude-code/agent-sdk/subagents.md @@ -15,8 +15,8 @@ This guide explains how to define and use subagents in the SDK using the `agents You can create subagents in three ways: -* **Programmatically**: use the `agents` parameter in your `query()` options. See the [TypeScript](/en/agent-sdk/typescript#agentdefinition) and [Python](/en/agent-sdk/python#agentdefinition) references -* **Filesystem-based**: define agents as markdown files in `.claude/agents/` directories. See [defining subagents as files](/en/sub-agents) +* **Programmatically**: use the `agents` parameter in your `query()` options. See the [TypeScript](/docs/en/agent-sdk/typescript#agentdefinition) and [Python](/docs/en/agent-sdk/python#agentdefinition) references +* **Filesystem-based**: define agents as markdown files in `.claude/agents/` directories. See [defining subagents as files](/docs/en/sub-agents) * **Built-in general-purpose**: Claude can invoke the built-in `general-purpose` subagent at any time via the Agent tool without you defining anything This guide focuses on the programmatic approach, which is recommended for SDK applications. @@ -179,20 +179,20 @@ This example creates two subagents: a code reviewer with read-only access and a | `effort` | `'low' \| 'medium' \| 'high' \| 'xhigh' \| 'max' \| number` | No | Reasoning effort level for this agent | | `permissionMode` | `PermissionMode` | No | Permission mode for tool execution within this agent | -In the Python SDK, multi-word field names such as `disallowedTools` and `mcpServers` keep their camelCase spelling to match the wire format rather than following Python's snake\_case convention. See the [`AgentDefinition` reference](/en/agent-sdk/python#agentdefinition) for details. +In the Python SDK, multi-word field names such as `disallowedTools` and `mcpServers` keep their camelCase spelling to match the wire format rather than following Python's snake\_case convention. See the [`AgentDefinition` reference](/docs/en/agent-sdk/python#agentdefinition) for details. Two subagent behaviors changed in Claude Code v2.1.198: -* Subagents run in the background by default. An Agent tool call that omits the [`run_in_background`](/en/agent-sdk/typescript) input launches a background subagent, and Claude sets `run_in_background: false` when it needs the result before continuing. Before v2.1.198, omitting `run_in_background` ran the subagent synchronously. Set the `background` field to `true` to force background execution for a specific agent regardless of what Claude requests. +* Subagents run in the background by default. An Agent tool call that omits the [`run_in_background`](/docs/en/agent-sdk/typescript) input launches a background subagent, and Claude sets `run_in_background: false` when it needs the result before continuing. Before v2.1.198, omitting `run_in_background` ran the subagent synchronously. Set the `background` field to `true` to force background execution for a specific agent regardless of what Claude requests. * A subagent inherits the main session's extended thinking configuration. On earlier versions, extended thinking is disabled inside subagents regardless of the main session's setting. - {/* min-version: 2.1.172 */}As of Claude Code v2.1.172, subagents can spawn their own subagents. A subagent five levels below the main agent can't spawn further subagents, regardless of whether it runs in the foreground or background. To prevent a subagent from spawning others, omit `Agent` from its `tools` array or add it to `disallowedTools`. See [nested subagents](/en/sub-agents#spawn-nested-subagents) for the full depth rules. + {/* min-version: 2.1.172 */}As of Claude Code v2.1.172, subagents can spawn their own subagents. A subagent five levels below the main agent can't spawn further subagents, regardless of whether it runs in the foreground or background. To prevent a subagent from spawning others, omit `Agent` from its `tools` array or add it to `disallowedTools`. See [nested subagents](/docs/en/sub-agents#spawn-nested-subagents) for the full depth rules. ### Filesystem-based definition (alternative) -You can also define subagents as markdown files in `.claude/agents/` directories. See the [Claude Code subagents documentation](/en/sub-agents) for details on this approach. Programmatically defined agents take precedence over filesystem-based agents with the same name. +You can also define subagents as markdown files in `.claude/agents/` directories. See the [Claude Code subagents documentation](/docs/en/sub-agents) for details on this approach. Programmatically defined agents take precedence over filesystem-based agents with the same name. Even without defining custom subagents, Claude can spawn the built-in `general-purpose` subagent. This is useful for delegating research or exploration tasks without creating specialized agents. Include `Agent` in `allowedTools` so these invocations auto-approve without a permission prompt. @@ -202,18 +202,18 @@ You can also define subagents as markdown files in `.claude/agents/` directories A subagent's context window starts fresh, with no parent conversation, but isn't empty. The only content you pass from parent to subagent is the Agent tool's prompt string, so include any file paths, error messages, or decisions the subagent needs directly in that prompt. -{/* min-version: 2.1.206 */}A subagent that has the [`SendMessage`](/en/tools-reference) tool starts with a list of the other named agents running in the session, so it knows which names it can send messages to. Claude Code adds the list to the subagent's first turn automatically. A [fork](/en/sub-agents#fork-the-current-conversation) doesn't get the list because it inherits the parent conversation instead. The list requires Claude Code v2.1.206 or later. +{/* min-version: 2.1.206 */}A subagent that has the [`SendMessage`](/docs/en/tools-reference) tool starts with a list of the other named agents running in the session, so it knows which names it can send messages to. Claude Code adds the list to the subagent's first turn automatically. A [fork](/docs/en/sub-agents#fork-the-current-conversation) doesn't get the list because it inherits the parent conversation instead. The list requires Claude Code v2.1.206 or later. | The subagent receives | The subagent doesn't receive | | :------------------------------------------------------------------------------------------------------------------------------------ | :----------------------------------------------------------------- | | Its own system prompt (`AgentDefinition.prompt`) and the Agent tool's prompt | The parent's conversation history or tool results | -| Project CLAUDE.md (loaded via [`settingSources`](/en/agent-sdk/claude-code-features#control-filesystem-settings-with-settingsources)) | Preloaded skill content, unless listed in `AgentDefinition.skills` | +| Project CLAUDE.md (loaded via [`settingSources`](/docs/en/agent-sdk/claude-code-features#control-filesystem-settings-with-settingsources)) | Preloaded skill content, unless listed in `AgentDefinition.skills` | | Tool definitions (inherited from parent, or the subset in `tools`) | The parent's system prompt | The parent receives the subagent's final message as the Agent tool result, but may summarize it in its own response. To preserve subagent output verbatim in the user-facing response, include an instruction to do so in the prompt or `systemPrompt` option you pass to the main `query()` call. - {/* min-version: 2.1.210 */}In v2.1.210 and later, Claude Code [scans the final message for instruction-shaped patterns](/en/sub-agents#subagent-output-scanning) before the parent reads it. The scan treats three kinds of pattern differently: + {/* min-version: 2.1.210 */}In v2.1.210 and later, Claude Code [scans the final message for instruction-shaped patterns](/docs/en/sub-agents#subagent-output-scanning) before the parent reads it. The scan treats three kinds of pattern differently: * **Control-tag imitation**: Claude Code neutralizes a tag that only the harness emits, such as a `` block, in place. It inserts a backslash after the opening angle bracket and deletes nothing. * **Permission-configuration mentions**: Claude Code keeps references to the permission configuration, such as `.claude/settings.json`, `bypassPermissions`, or `--dangerously-skip-permissions`, as written. @@ -222,7 +222,7 @@ A subagent's context window starts fresh, with no parent conversation, but isn't For a control-tag or permission-configuration match, Claude Code prepends a `[harness: ...]` marker line naming the matched patterns; a turn-marker match doesn't add the marker line. Those are the only modifications the scan makes: it never removes or rewords the subagent's text. -{/* min-version: 2.1.199 */}An API error that ends the subagent early, such as a rate limit, is never delivered as its result. If a rate limit, overload, or server error cuts off a foreground subagent that already produced text output, the Agent tool returns that partial output with a note that the subagent didn't finish. {/* min-version: 2.1.200 */}A subagent that produced nothing, or whose only output was tool calls with no text, fails with an error message, `Agent terminated early due to an API error`, followed by the error detail. See [API errors in subagents](/en/sub-agents#api-errors-in-subagents) for the foreground and background behavior. +{/* min-version: 2.1.199 */}An API error that ends the subagent early, such as a rate limit, is never delivered as its result. If a rate limit, overload, or server error cuts off a foreground subagent that already produced text output, the Agent tool returns that partial output with a note that the subagent didn't finish. {/* min-version: 2.1.200 */}A subagent that produced nothing, or whose only output was tool calls with no text, fails with an error message, `Agent terminated early due to an API error`, followed by the error detail. See [API errors in subagents](/docs/en/sub-agents#api-errors-in-subagents) for the foreground and background behavior. This partial-output handling requires Claude Code v2.1.199 or later. In v2.1.199, a rate limit, overload, or server error left the tool-calls-only shape with an empty partial result containing only the cutoff note. @@ -415,7 +415,7 @@ This example iterates through streamed messages, logging when a subagent is invo You can resume a subagent to continue where it left off rather than starting fresh. A resumed subagent retains its full conversation history, including all previous tool calls, results, and reasoning. -When a subagent completes, the Agent tool result includes a text block containing `agentId: `. The built-in [`Explore` and `Plan` agents](/en/sub-agents#built-in-subagents) are one-shot and don't return an `agentId`, so use a custom agent or `general-purpose` when you need to resume. To resume a subagent programmatically: +When a subagent completes, the Agent tool result includes a text block containing `agentId: `. The built-in [`Explore` and `Plan` agents](/docs/en/sub-agents#built-in-subagents) are one-shot and don't return an `agentId`, so use a custom agent or `general-purpose` when you need to resume. To resume a subagent programmatically: 1. **Capture the session ID**: extract `session_id` from messages during the first query 2. **Extract the agent ID**: parse `agentId` from the Agent tool result text @@ -629,9 +629,9 @@ This example creates a read-only analysis agent that can examine code but can't ## Scale up with dynamic workflows -Subagents work well for a few delegated tasks per turn. For runs that coordinate dozens to hundreds of agents, use the `Workflow` tool, which moves the orchestration into a script the runtime executes outside the conversation context. See [dynamic workflows](/en/workflows) for how workflows differ from turn-by-turn subagent delegation. +Subagents work well for a few delegated tasks per turn. For runs that coordinate dozens to hundreds of agents, use the `Workflow` tool, which moves the orchestration into a script the runtime executes outside the conversation context. See [dynamic workflows](/docs/en/workflows) for how workflows differ from turn-by-turn subagent delegation. -The `Workflow` tool is available in the TypeScript Agent SDK v0.3.149 and later. Include `Workflow` in `allowedTools` to auto-approve workflow runs. The tool input and output schemas are listed in the [TypeScript reference](/en/agent-sdk/typescript#workflow). +The `Workflow` tool is available in the TypeScript Agent SDK v0.3.149 and later. Include `Workflow` in `allowedTools` to auto-approve workflow runs. The tool input and output schemas are listed in the [TypeScript reference](/docs/en/agent-sdk/typescript#workflow). ## Troubleshooting @@ -652,7 +652,7 @@ Claude Code watches `~/.claude/agents/` and `.claude/agents/` and picks up a new * **`--disable-slash-commands`**: sessions started with this flag don't watch these directories and always need a restart to load new files. * **A programmatic agent with the same name**: `agents` passed to `query()` override a filesystem agent with the same name. -For the file format, see [how to write subagent files](/en/sub-agents#write-subagent-files). +For the file format, see [how to write subagent files](/docs/en/sub-agents#write-subagent-files). ### Long prompt failures on Windows @@ -660,6 +660,6 @@ On Windows, subagents with very long prompts may fail due to the command line le ## Related documentation -* [Claude Code subagents](/en/sub-agents): comprehensive subagent documentation including filesystem-based definitions -* [Dynamic workflows](/en/workflows): orchestrate many subagents from a script for jobs too large for one conversation -* [SDK overview](/en/agent-sdk/overview): getting started with the Claude Agent SDK +* [Claude Code subagents](/docs/en/sub-agents): comprehensive subagent documentation including filesystem-based definitions +* [Dynamic workflows](/docs/en/workflows): orchestrate many subagents from a script for jobs too large for one conversation +* [SDK overview](/docs/en/agent-sdk/overview): getting started with the Claude Agent SDK diff --git a/content/en/docs/claude-code/agent-sdk/todo-tracking.md b/content/en/docs/claude-code/agent-sdk/todo-tracking.md index 73af6710e..33dd7869d 100644 --- a/content/en/docs/claude-code/agent-sdk/todo-tracking.md +++ b/content/en/docs/claude-code/agent-sdk/todo-tracking.md @@ -34,13 +34,13 @@ It may skip todos for very short or single-step requests. ## Examples -Before running these examples, install the Claude Agent SDK by following the [quickstart](/en/agent-sdk/quickstart). +Before running these examples, install the Claude Agent SDK by following the [quickstart](/docs/en/agent-sdk/quickstart). Each example runs until the agent finishes and yields its final result message. If a session reaches its turn limit first, that result message has the `error_max_turns` subtype. Check `subtype` to detect that ending. These examples use single-shot `query()` calls. After yielding an `error_max_turns` result, `query()` raises an error that includes `Reached maximum number of turns`. Each example wraps its loop in a try block to exit cleanly when that happens. -See [Handle the result](/en/agent-sdk/agent-loop#handle-the-result) for the result subtypes. +See [Handle the result](/docs/en/agent-sdk/agent-loop#handle-the-result) for the result subtypes. ### Monitoring Todo Changes @@ -324,7 +324,7 @@ The streamed `tool_use` input is the raw shape the model emitted. Claude Code re ## Related Documentation -* [TypeScript SDK Reference](/en/agent-sdk/typescript) -* [Python SDK Reference](/en/agent-sdk/python) -* [Streaming vs Single Mode](/en/agent-sdk/streaming-vs-single-mode) -* [Custom Tools](/en/agent-sdk/custom-tools) +* [TypeScript SDK Reference](/docs/en/agent-sdk/typescript) +* [Python SDK Reference](/docs/en/agent-sdk/python) +* [Streaming vs Single Mode](/docs/en/agent-sdk/streaming-vs-single-mode) +* [Custom Tools](/docs/en/agent-sdk/custom-tools) diff --git a/content/en/docs/claude-code/agent-sdk/tool-search.md b/content/en/docs/claude-code/agent-sdk/tool-search.md index 1d699f5f8..44ca720f1 100644 --- a/content/en/docs/claude-code/agent-sdk/tool-search.md +++ b/content/en/docs/claude-code/agent-sdk/tool-search.md @@ -39,9 +39,9 @@ Tool search is on by default. It is disabled by default on Google Cloud's Agent | `auto:N` | Same as `auto` with a custom percentage. `auto:5` activates when tool definitions exceed 5% of the context window. Lower values activate sooner. | | `false` | Tool search is off. All tool definitions are loaded into context on every turn. | -Setting [`CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS`](/en/env-vars) keeps tool search off, and `ENABLE_TOOL_SEARCH` can't override it. The variable strips the beta header that `defer_loading` tool definitions and `tool_reference` content blocks require. +Setting [`CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS`](/docs/en/env-vars) keeps tool search off, and `ENABLE_TOOL_SEARCH` can't override it. The variable strips the beta header that `defer_loading` tool definitions and `tool_reference` content blocks require. -Tool search applies to all registered tools, whether they come from remote MCP servers or [custom SDK MCP servers](/en/agent-sdk/custom-tools). When using `auto`, the threshold is based on the combined size of all tool definitions across all servers. +Tool search applies to all registered tools, whether they come from remote MCP servers or [custom SDK MCP servers](/docs/en/agent-sdk/custom-tools). When using `auto`, the threshold is based on the combined size of all tool definitions across all servers. Set the value in the `env` option on `query()`. In TypeScript, `env` replaces the subprocess environment, so spread `...process.env` to keep inherited variables. In Python, `env` is merged on top of the inherited environment. This example connects to a remote MCP server that exposes many tools, pre-approves all of them with a wildcard, and uses `auto:5` so tool search activates when their definitions exceed 5% of the context window: @@ -116,7 +116,7 @@ Set the value in the `env` option on `query()`. In TypeScript, `env` replaces th To run this example, replace `https://tools.example.com/mcp` with the URL of your own MCP server. On success the result text prints to the console. -Because this is a single-shot `query()` call, the SDK raises after yielding an error result, so the example wraps the loop in a try block. To see why a run failed, check the result message's `subtype`, such as `error_during_execution`, inside the loop. For more on result messages, see [Handle the result](/en/agent-sdk/agent-loop#handle-the-result). +Because this is a single-shot `query()` call, the SDK raises after yielding an error result, so the example wraps the loop in a try block. To see why a run failed, check the result message's `subtype`, such as `error_during_execution`, inside the loop. For more on result messages, see [Handle the result](/docs/en/agent-sdk/agent-loop#handle-the-result). Setting `ENABLE_TOOL_SEARCH` to `"false"` disables tool search and loads all tool definitions into context on every turn. This removes the search round-trip, which can be faster when the tool set is small (fewer than \~10 tools) and the definitions fit comfortably in the context window. @@ -148,7 +148,7 @@ You can also add a system prompt section listing available tool categories. This ``` -For the full set of system prompt options, see [Modifying system prompts](/en/agent-sdk/modifying-system-prompts). +For the full set of system prompt options, see [Modifying system prompts](/docs/en/agent-sdk/modifying-system-prompts). ## Limits @@ -159,7 +159,7 @@ For the full set of system prompt options, see [Modifying system prompts](/en/ag ## Related documentation * [Tool search in the API](https://platform.claude.com/docs/en/agents-and-tools/tool-use/tool-search-tool): Full API documentation for tool search, including custom implementations -* [Connect MCP servers](/en/agent-sdk/mcp): Connect to external tools via MCP servers -* [Custom tools](/en/agent-sdk/custom-tools): Build your own tools with SDK MCP servers -* [TypeScript SDK reference](/en/agent-sdk/typescript): Full API reference -* [Python SDK reference](/en/agent-sdk/python): Full API reference +* [Connect MCP servers](/docs/en/agent-sdk/mcp): Connect to external tools via MCP servers +* [Custom tools](/docs/en/agent-sdk/custom-tools): Build your own tools with SDK MCP servers +* [TypeScript SDK reference](/docs/en/agent-sdk/typescript): Full API reference +* [Python SDK reference](/docs/en/agent-sdk/python): Full API reference diff --git a/content/en/docs/claude-code/agent-sdk/typescript-v2-preview.md b/content/en/docs/claude-code/agent-sdk/typescript-v2-preview.md index f47e41c40..c7f4e2317 100644 --- a/content/en/docs/claude-code/agent-sdk/typescript-v2-preview.md +++ b/content/en/docs/claude-code/agent-sdk/typescript-v2-preview.md @@ -9,7 +9,7 @@ The V2 session API is no longer supported. TypeScript Agent SDK 0.3.142 removes `unstable_v2_createSession`, `unstable_v2_resumeSession`, `unstable_v2_prompt`, and the `SDKSession` and `SDKSessionOptions` types. - To migrate, use the [`query()` API](/en/agent-sdk/typescript) and the [session options](/en/agent-sdk/sessions) it accepts. Pass an `AsyncIterable` for multi-turn conversations, or `options.resume` to continue a saved session. This page is kept for reference if you maintain code on Agent SDK 0.2.x or earlier. + To migrate, use the [`query()` API](/docs/en/agent-sdk/typescript) and the [session options](/docs/en/agent-sdk/sessions) it accepts. Pass an `AsyncIterable` for multi-turn conversations, or `options.resume` to continue a saved session. This page is kept for reference if you maintain code on Agent SDK 0.2.x or earlier. V2 was an experimental session API that removed the need for async generators and yield coordination. Instead of managing generator state across turns, each turn was a separate `send()`/`stream()` cycle. The API surface reduced to three concepts: @@ -382,13 +382,13 @@ interface SDKSession { ## Feature availability -The V2 session API does not support every V1 feature. The following require the [V1 SDK](/en/agent-sdk/typescript): +The V2 session API does not support every V1 feature. The following require the [V1 SDK](/docs/en/agent-sdk/typescript): * Session forking (`forkSession` option) * Some advanced streaming input patterns ## See also -* [TypeScript SDK reference (V1)](/en/agent-sdk/typescript) - Full V1 SDK documentation -* [SDK overview](/en/agent-sdk/overview) - General SDK concepts +* [TypeScript SDK reference (V1)](/docs/en/agent-sdk/typescript) - Full V1 SDK documentation +* [SDK overview](/docs/en/agent-sdk/overview) - General SDK concepts * [V2 examples on GitHub](https://github.com/anthropics/claude-agent-sdk-demos/tree/main/hello-world-v2) - Working code examples diff --git a/content/en/docs/claude-code/agent-sdk/typescript.md b/content/en/docs/claude-code/agent-sdk/typescript.md index c0ae0050e..d4b0ea4f3 100644 --- a/content/en/docs/claude-code/agent-sdk/typescript.md +++ b/content/en/docs/claude-code/agent-sdk/typescript.md @@ -6,7 +6,7 @@ > Complete API reference for the TypeScript Agent SDK, including all functions, types, and interfaces. - @@ -36,7 +37,7 @@
Contact sales
Try Claude