Fetch the complete documentation index at: /docs/llms.txt
Use this file to discover all available pages before exploring further.
Search...⌘KContact SalesCommunityPlaygroundUser CenterSearch...NavigationGet StartedKimi K3GuidesAPI ReferencePricingHelp CenterResources:root{--topbar-tabs-height:3rem}Get StartedQuickstartModel ListKimi K3Kimi K2.7 Code ModelKimi K2.6 ModelModel CapabilitiesThinking ModelsReasoning EffortMulti-turn ChatStreamingJSON ModePartial ModeVision InputContext CachingDynamic Tool LoadingTooling workflowsTool CallsWeb Search ToolOfficial ToolsTool ChoiceModelScope MCP ServerTool Calling Best PracticesCore workflowsresponse_formatAutomatic ReconnectionFile-Based Q&ABatch API GuideIntegrationsKimi Code CLIOpenClawClaude CodeOpenCodeHermes AgentCodexDebugging and operationsUse the API Debugging ToolPlaygroundBest Practices for Organization Management(function () { try { if (window.__mintlifyInitialSidebarScrollDone) return; window.__mintlifyInitialSidebarScrollDone = true; var path = (window.location.pathname || '/').split('#')[0].split('?')[0]; if (path.endsWith('/index')) path = path.slice(0, -6); else if (path === 'index') path = ''; var candidates = []; if (path) candidates.push(path); if (path.startsWith('/')) candidates.push(path.slice(1)); else candidates.push('/' + path); var item = null; for (var i = 0; i = parentRect.top && itemRect.bottom {"@context":"https://schema.org","@graph":[{"@type":"Organization","@id":"https://platform.kimi.ai/#organization","name":"Kimi API Platform","url":"https://platform.kimi.ai","logo":{"@type":"ImageObject","url":"https://mintcdn.com/moonshotai/X_5b_eA1iuJP595e/assets/logo/light.svg?fit=max&auto=format&n=X_5b_eA1iuJP595e&q=85&s=f68cf2d80f8d7f0363e088d2b988a598"}},{"@type":"WebSite","@id":"https://platform.kimi.ai/docs#website","name":"Kimi API Platform","url":"https://platform.kimi.ai/docs","publisher":{"@id":"https://platform.kimi.ai/#organization"}},{"@type":"WebPage","@id":"https://platform.kimi.ai/docs/guide/kimi-k3-quickstart#webpage","url":"https://platform.kimi.ai/docs/guide/kimi-k3-quickstart","name":"Kimi K3","dateModified":"2026-07-23T13:01:36.752Z","isPartOf":{"@id":"https://platform.kimi.ai/docs#website"},"breadcrumb":{"@id":"https://platform.kimi.ai/docs/guide/kimi-k3-quickstart#breadcrumb"}},{"@type":"BreadcrumbList","@id":"https://platform.kimi.ai/docs/guide/kimi-k3-quickstart#breadcrumb","itemListElement":[{"@type":"ListItem","position":1,"name":"Get Started","item":"https://platform.kimi.ai/docs/overview"},{"@type":"ListItem","position":2,"name":"Kimi K3","item":"https://platform.kimi.ai/docs/guide/kimi-k3-quickstart"}]},{"@type":["Article","TechArticle"],"@id":"https://platform.kimi.ai/docs/guide/kimi-k3-quickstart#article","headline":"Kimi K3","name":"Kimi K3","url":"https://platform.kimi.ai/docs/guide/kimi-k3-quickstart","mainEntityOfPage":{"@id":"https://platform.kimi.ai/docs/guide/kimi-k3-quickstart#webpage"},"image":"https://platform.kimi.ai/og.png?v=1","dateModified":"2026-07-23T13:01:36.752Z","publisher":{"@id":"https://platform.kimi.ai/#organization"},"isPartOf":{"@id":"https://platform.kimi.ai/docs#website"}}]}document.documentElement.setAttribute('data-page-mode', "none");(self.__next_s=self.__next_s||[]).push([0,{"suppressHydrationWarning":true,"children":"(function m(a,b,c){if(!document.getElementById("footer")?.classList.contains("advanced-footer")||"maple"===b||"willow"===b||"almond"===b||"luma"===b||"sequoia"===b)return;let d=document.documentElement.getAttribute("data-banner-state"),e=null!=d?"visible"===d:c,f=document.documentElement.getAttribute("data-page-mode"),g=document.getElementById("navbar"),h=document.getElementById("navigation-items"),i=document.getElementById("sidebar"),j=document.getElementById("footer"),k=document.getElementById("table-of-contents-content"),l=document.getElementById("banner"),m=e?l?.offsetHeight??40:0,n=getComputedStyle(document.documentElement).getPropertyValue("--mintlify-slot-header-height").trim(),o=n.endsWith("px")?parseFloat(n):0,p=getComputedStyle(document.documentElement).getPropertyValue("--topbar-tabs-height").trim(),q=p.endsWith("rem")?parseFloat(p):0,r=e?a-2.5:a,s="mint"===b?"calc(4rem + var(--topbar-tabs-height, 0rem))":${r}rem,t=(o||16*("mint"===b?4+q:r))+m;if(!j||"center"===f)return;let u=j.getBoundingClientRect().top,v=window.innerHeight-u,w=(h?.clientHeight??0)+t+32*("mint"===b||"linden"===b);if(i\u0026\u0026h)if(v\u003e0){let a=Math.max(0,w-u);i.style.bottom=${v}px,i.style.top=${t-a}px}else i.style.bottom="",i.style.top=e?calc(var(--mintlify-slot-header-height, ${s}) + var(--banner-height, 2.5rem)):var(--mintlify-slot-header-height, ${s}),i.style.height="auto";k\u0026\u0026g\u0026\u0026(v\u003e0?k.style.top="custom"===f?${g.clientHeight-v}px:${40+g.clientHeight-v}px:k.style.top="")})(\n (function l(a,b,c){let d=document.documentElement.getAttribute("data-banner-state"),e=2.5*!!(null!=d?"visible"===d:b),f=3*!!a,g=4,h=e+g+f;switch(c){case"mint":case"palm":break;case"aspen":f=2.5*!!a,g=3.5,h=e+f+g;break;case"luma":g=3,h=e+g;break;case"linden":g=4,h=e+g;break;case"almond":g=3.5,h=e+g;break;case"sequoia":f=3*!!a,g=3,h=e+g+f}return h})(true, true, "mint"),\n "mint",\n true,\n)","id":"_mintlify-footer-and-sidebar-scroll-script"}])/* Brand color system / :root { --brand-primary: #111111; --brand-primary-strong: #2a2a2a; --brand-primary-soft: rgba(17, 17, 17, 0.08); --brand-text: #171717; --brand-text-muted: #5f5f68; --brand-border: #e5e7eb; --brand-border-strong: #d4d4d8; --brand-surface: #f5f5f5; --brand-card-bg: rgba(255, 255, 255, 0.9); --brand-overlay: linear-gradient(to top, rgba(10, 10, 10, 0.78), rgba(10, 10, 10, 0.28), transparent); } .dark, [data-theme="dark"] { --brand-primary: #f5f5f5; --brand-primary-strong: #ffffff; --brand-primary-soft: rgba(245, 245, 245, 0.14); --brand-text: #fafafa; --brand-text-muted: #a1a1aa; --brand-border: #303038; --brand-border-strong: #3f3f46; --brand-surface: #151515; --brand-card-bg: rgba(24, 24, 27, 0.88); --brand-overlay: linear-gradient(to top, rgba(0, 0, 0, 0.84), rgba(0, 0, 0, 0.3), transparent); } / Footer: hide divider, "Powered by Mintlify", and theme toggle buttons / #footer > * > div:last-child, #footer > * > div:nth-last-child(2) { display: none !important; } / Use a neutral community icon for the forum footer link. / #footer a[href^="https://forum.moonshot.ai/"] svg { display: none !important; } #footer a[href^="https://forum.moonshot.ai/"]::before { content: ""; display: block; width: 1.25rem; height: 1.25rem; background-color: currentColor; -webkit-mask: url("/assets/icons/community.svg") center / contain no-repeat; mask: url("/assets/icons/community.svg") center / contain no-repeat; } / Global Mintlify UI tuning / #search-bar-entry, #search-bar-entry-mobile { border: 1px solid var(--brand-border) !important; background: var(--brand-card-bg) !important; box-shadow: none !important; } #search-input { color: var(--brand-text) !important; } #search-input::placeholder { color: var(--brand-text-muted) !important; } #topbar, #topbar-content-container, #topbar-content-wrapper { color: var(--brand-text) !important; } / Keep the desktop sidebar below the navbar's 3rem documentation tabs. Mintlify's default offset only includes the 4rem topbar and banner. / @media (min-width: 1024px) { #sidebar { top: calc(7rem + var(--banner-height, 0px)) !important; } } #sidebar-content li[data-active] { background: var(--brand-primary-soft) !important; border-radius: 14px; } / Active sidebar item text: color the link itself and let the title inherit. Method pills (POST/GET) set their own colors and are unaffected. / #sidebar-content li[data-active] a, #sidebar-content li[data-active] button { color: var(--brand-primary) !important; } / Prevent synthetic bold on sidebar items — Mintlify uses text-shadow to simulate bold, which distorts CJK glyphs / #sidebar-content a, #sidebar-content button { font-weight: inherit !important; text-shadow: none !important; } .sidebar-group[data-active] > .sidebar-group-header, .sidebar-group[data-active] > .sidebar-group-header * { color: var(--brand-primary) !important; } .method-nav-pill, .method-pill { border-radius: 999px !important; box-shadow: none !important; } #request-example, #response-example, #api-playground-input, #panel, #page-context-menu-button { border-color: var(--brand-border) !important; } #request-example, #response-example, #api-playground-input { background: var(--brand-card-bg) !important; } .dark .method-nav-pill, .dark .method-pill, [data-theme="dark"] .method-nav-pill, [data-theme="dark"] .method-pill { border-color: transparent !important; } / Fix Tabs active state visibility in dark mode / .dark li[role="tab"][aria-selected="true"] > div, [data-theme="dark"] li[role="tab"][aria-selected="true"] > div { color: #ffffff !important; border-color: #ffffff !important; } .nav-tabs-item[data-active], .mobile-nav-tabs-item[data-active] { color: var(--brand-primary) !important; } .navbar-link:hover, .nav-tabs-item:hover, .mobile-nav-tabs-item:hover { color: var(--brand-primary) !important; } .toc-item[data-active], .toc-item[data-active-deepest] { color: var(--brand-primary) !important; } .toc-item[data-active] a, .toc-item[data-active-deepest] a { color: var(--brand-primary) !important; } .mdx-content a { color: var(--brand-primary); text-decoration-color: color-mix(in srgb, var(--brand-primary) 58%, transparent); } .mdx-content a:hover { color: var(--brand-primary-strong); text-decoration-color: color-mix(in srgb, var(--brand-primary-strong) 72%, transparent); } .mdx-content a strong, .mdx-content a b, .mdx-content a code { color: inherit; } .code-block, pre { border-color: var(--brand-border) !important; } code:not(pre code) { background: var(--brand-surface) !important; border: 1px solid var(--brand-border); color: var(--brand-text) !important; } #page-context-menu-button:hover, .pagination-next:hover, .pagination-prev:hover, .topbar-cta-button:hover { border-color: var(--brand-primary) !important; } / Base styles for the overview sections / .overview-page { color: var(--brand-text); } .overview-page a { border-bottom: none !important; box-shadow: none !important; text-decoration: none !important; } .overview-page .overview-overlay-link { position: absolute !important; inset: 0 !important; display: block !important; border: 0 !important; border-bottom: 0 !important; box-shadow: none !important; text-decoration: none !important; z-index: 4 !important; } / Grid Layouts / .overview-page .overview-grid-2, .overview-page .overview-grid-3, .overview-page .overview-grid-4 { display: grid; align-items: stretch; gap: 16px; } .overview-page .overview-grid-2 { grid-template-columns: repeat(2, minmax(0, 1fr)); } .overview-page .overview-grid-3 { grid-template-columns: repeat(3, minmax(0, 1fr)); } .overview-page .overview-grid-4 { grid-template-columns: repeat(2, minmax(0, 1fr)); } / Card Common Styles / .overview-page .overview-link-card, .overview-page .overview-resource-card { position: relative; overflow: hidden; display: flex; height: 100%; border: 1px solid var(--brand-border); border-radius: 12px; background: var(--brand-card-bg); backdrop-filter: blur(10px); transition: border-color 0.2s, background-color 0.2s, box-shadow 0.2s; } .overview-page .overview-link-card:hover, .overview-page .overview-resource-card:hover { border-color: var(--brand-primary); box-shadow: 0 10px 30px rgba(15, 23, 42, 0.08); } / Card Body (Media Object) / .overview-page .overview-link-card__body, .overview-page .overview-resource-card__body { display: flex; padding: 20px 16px; width: 100%; flex-direction: row; gap: 12px; align-items: flex-start; } .overview-page .overview-link-card__content { display: flex; flex: 1; flex-direction: column; } .overview-page .overview-link-card__icon { display: flex; width: 24px; min-width: 24px; align-items: center; justify-content: center; line-height: 1; margin-top: 3px; color: var(--brand-text-muted); } .overview-page .overview-link-card__title, .overview-page .overview-resource-card__title { font-size: 16px; font-weight: 500; line-height: 1.2; color: var(--brand-text); } .overview-page .overview-link-card__description, .overview-page .overview-resource-card__description { margin-top: 4px; font-size: 14px; color: var(--brand-text-muted); line-height: 1.4; } / Model Card Styles / .overview-page .overview-model-card { position: relative; overflow: hidden; border: 1px solid var(--brand-border); border-radius: 12px; background: var(--brand-surface); } .overview-page .overview-model-card__media { position: relative; aspect-ratio: 123 / 92; line-height: 0; } .overview-page .overview-model-card__image { display: block; width: 100%; height: 100%; object-fit: cover; transition: transform 0.5s; } .overview-page .overview-model-card:hover .overview-model-card__image { transform: scale(1.05); } .overview-page .overview-model-card__overlay { position: absolute; inset: 0; display: flex; align-items: flex-end; background: var(--brand-overlay); } .overview-page .overview-model-card__content { width: 100%; padding: 16px; } .overview-page .overview-model-card__name { color: white; font-weight: 600; line-height: 1.2; font-size: 16px; } .overview-page .overview-model-card__tag-row { margin-top: 8px; line-height: 1; } .overview-page .overview-hero-card { background: var(--brand-surface); border-color: var(--brand-border) !important; } .overview-page .overview-hero-card__cta { display: inline-flex; align-items: center; gap: 6px; padding: 10px 20px; border: 1px solid rgba(255, 255, 255, 0.14); border-radius: 8px; background: rgba(255, 255, 255, 0.96); color: #111111 !important; font-size: 14px; font-weight: 500; white-space: nowrap; position: relative; z-index: 10; transition: transform 0.2s ease, background-color 0.2s ease, border-color 0.2s ease; } .overview-page .overview-hero-card__cta:hover { background: #ffffff; border-color: rgba(255, 255, 255, 0.3); transform: translateY(-1px); } .overview-page .overview-section-link { color: var(--brand-primary) !important; text-decoration: underline !important; text-underline-offset: 0.18em; } .overview-page .overview-section-link:hover { color: var(--brand-primary-strong) !important; } .overview-page .overview-resource-card a, .overview-page .overview-link-card a, .overview-page .overview-model-card a { color: inherit !important; } / Responsive adjustments / @media (max-width: 1100px) { .overview-page .overview-grid-3 { grid-template-columns: repeat(2, minmax(0, 1fr)); } } / Hide API page right-side request/response code examples */ #content-side-layout:has(.code-group) { display: none !important; } @media (max-width: 700px) { .overview-page .overview-grid-2, .overview-page .overview-grid-3, .overview-page .overview-grid-4 { grid-template-columns: 1fr; } } .doc-table-wrap { width: 100%; max-width: 100%; } .doc-table-wrap table { width: 100% !important; max-width: 100% !important; min-width: 0 !important; table-layout: fixed !important; margin: 0 !important; } .doc-table-wrap th, .doc-table-wrap td { white-space: normal !important; overflow-wrap: anywhere; word-break: break-word; text-align: left !important; vertical-align: top; } @media (max-width: 900px) { .doc-table-wrap th, .doc-table-wrap td { padding: 10px 8px !important; font-size: 13px; line-height: 1.45; } } @media (max-width: 640px) { .doc-table-wrap { width: 100%; } .doc-table-wrap th, .doc-table-wrap td { padding: 8px 6px !important; font-size: 12px; line-height: 1.4; } .doc-table-wrap td:first-child, .doc-table-wrap th:first-child { word-break: break-all; overflow-wrap: anywhere; } .doc-table-wrap code { font-size: 11px; white-space: normal !important; word-break: break-all; } } On this pageIntroducing Kimi K3A 3-trillion-scale open-source modelCodingKnowledge workAccess requirementsGet startedBasic callReasoning effortStreamingVision inputStructured outputPartial ModeCustom tools and tool_choiceDynamic tool loading1M context and automatic cachingOfficial toolsImportant limitsFAQModel PricingRelated docsGet StartedKimi K3Copy pageCopy pageCopy pageCopy pageIntroducing Kimi K3 Kimi K3 is Kimi’s most capable flagship model to date, with 2.8 trillion parameters. It is built on Kimi Delta Attention (KDA), a hybrid linear attention mechanism, and Attention Residuals, with native visual understanding and a 1M-token context window. It is the world’s first open-source model in the 3-trillion-parameter class, designed for frontier intelligence scenarios including long-horizon coding, knowledge work, and reasoning. For complete benchmarks and case studies, see the technical blog. Kimi is currently working closely with inference partners and open-source maintainers to align technical details and ensure the model launches reliably across the ecosystem. The full model weights will be released by July 27, 2026. More details on architecture, training, and evaluation will be published with the Kimi K3 technical report. A 3-trillion-scale open-source model Kimi K3 is the first open-source model to reach 2.8 trillion parameters. This is the latest step in Kimi’s continued push of model-scale boundaries: in 9 of the past 12 months (2025/07–2026/07), Kimi models have maintained the frontier in open-source model scale. Kimi K3 is built on Kimi Delta Attention (KDA) and Attention Residuals (AttnRes). Both architectural updates are designed to help information flow more smoothly through longer sequences and deeper models. We also further increased the sparsity of the Mixture of Experts (MoE): with the Stable LatentMoE framework, the model efficiently activates 16 out of 896 experts. Together with improvements in training methodology and data recipes, these structural advances give Kimi K3 roughly 2.5x the overall scaling efficiency of K2, converting compute into capability more effectively. Coding Kimi K3 has strong long-horizon coding capabilities. With minimal human supervision, it can sustain long-running engineering tasks, understand and work with large codebases, and coordinate terminal tools. Kimi K3 also excels at tasks that combine software engineering and visual reasoning. It can use screenshots and visual feedback to improve workflows in game development, frontend engineering, CAD, and related scenarios. Knowledge work Kimi K3 advances end-to-end knowledge work. Beyond public benchmarks, Kimi K3 (max) also shows consistent gains in our internal evaluations. These evaluations reflect recurring task patterns and challenges from real user-agent collaboration workflows. Kimi K3 demonstrates consistent advantages across production-oriented workflows, indicating broad improvements in agentic knowledge-work capabilities. Access requirements Kimi K3 is a flagship model: it is unlocked after a successful top-up (minimum $1). Your cumulative top-up amount also determines your account tier and rate limits (concurrency, RPM, TPM, TPD) — see Recharge and Rate Limits. Get started Playground Get an API Key The examples require Python 3.9+ and the OpenAI SDK. Install the SDK and initialize the client once; later Python examples reuse client. python3 -m pip install --upgrade 'openai>=1.0' import os from openai import OpenAI client = OpenAI( api_key=os.environ["MOONSHOT_API_KEY"], base_url="https://api.moonshot.ai/v1", ) Basic call Python cURLcompletion = client.chat.completions.create( model="kimi-k3", messages=[{"role": "user", "content": "Introduce Kimi K3 in one sentence."}], ) print(completion.choices[0].message.content) curl https://api.moonshot.ai/v1/chat/completions \ --header "Authorization: Bearer $MOONSHOT_API_KEY" \ --header "Content-Type: application/json" \ --data '{ "model": "kimi-k3", "messages": [{"role": "user", "content": "Introduce Kimi K3 in one sentence."}] }' Reasoning effort K3 always has thinking mode enabled and supports configuring its reasoning effort with the top-level reasoning_effort request field. Reasoning effort supports low, high, and max (default max). See Reasoning Effort for usage. completion = client.chat.completions.create( model="kimi-k3", reasoning_effort="max", messages=[{"role": "user", "content": "Prove that the square root of 2 is irrational."}], ) print(completion.choices[0].message.content) For multi-turn conversations and tool calls, add the complete assistant message returned by the API to the next request. Do not keep only content. Streaming Streaming responses provide separate reasoning_content and final-answer content deltas. See Streaming Output for details. stream = client.chat.completions.create( model="kimi-k3", messages=[{"role": "user", "content": "Explain why the sky is blue."}], stream=True, ) for chunk in stream: delta = chunk.choices[0].delta reasoning = getattr(delta, "reasoning_content", None) if reasoning: print(reasoning, end="", flush=True) if delta.content: print(delta.content, end="", flush=True) Vision input For vision messages, content must be an array of objects, not a serialized string. See Vision Input for formats and limits. Local image Video fileimport base64 from pathlib import Path image_data: str = base64.b64encode(Path("image.png").read_bytes()).decode() completion = client.chat.completions.create( model="kimi-k3", messages=[ { "role": "user", "content": [ { "type": "image_url", "image_url": {"url": f"data:image/png;base64,{image_data}"}, }, {"type": "text", "text": "Describe this image."}, ], } ], ) print(completion.choices[0].message.content) from pathlib import Path video = client.files.create(file=Path("video.mp4"), purpose="video") try: completion = client.chat.completions.create( model="kimi-k3", messages=[ { "role": "user", "content": [ { "type": "video_url", "video_url": {"url": f"ms://{video.id}"}, }, {"type": "text", "text": "Summarize this video."}, ], } ], ) print(completion.choices[0].message.content) finally: client.files.delete(video.id) Structured output Use json_schema with strict: true to constrain the final message.content. Parse only that field, not reasoning_content. Name and age schema
import json completion = client.chat.completions.create( model="kimi-k3", messages=[ {"role": "user", "content": "Lin is 28 years old. Extract the name and age."} ], response_format={ "type": "json_schema", "json_schema": { "name": "person", "strict": True, "schema": { "type": "object", "properties": { "name": {"type": "string"}, "age": {"type": "integer"}, }, "required": ["name", "age"], "additionalProperties": False, }, }, }, ) person: dict[str, object] = json.loads( completion.choices[0].message.content or "{}" ) print(person) See Structured Output. Partial Mode Add an assistant message with partial=True at the end of messages to continue from a text prefix. Prepend that prefix when displaying the final result. prefix: str = "Conclusion: " completion = client.chat.completions.create( model="kimi-k3", messages=[ {"role": "user", "content": "In one sentence, explain why API compatibility matters."}, {"role": "assistant", "content": prefix, "partial": True}, ], ) print(prefix + (completion.choices[0].message.content or "")) See Partial Mode. Custom tools and tool_choice Use tool_choice="required" on the first turn to require at least one tool call. After executing every call, return the complete assistant message and append one tool result with the matching tool_call_id for each call. Minimal weather agent loop
import json from typing import Any tools: list[dict[str, Any]] = [ { "type": "function", "function": { "name": "get_weather", "description": "Get the weather for a city", "parameters": { "type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"], }, }, } ] messages: list[Any] = [ {"role": "user", "content": "What is the weather in San Francisco today?"} ] first = client.chat.completions.create( model="kimi-k3", messages=messages, tools=tools, tool_choice="required", ) assistant_message = first.choices[0].message messages.append(assistant_message) for tool_call in assistant_message.tool_calls or []: arguments: dict[str, str] = json.loads(tool_call.function.arguments) result: str = json.dumps( {"city": arguments["city"], "weather": "sunny", "temperature_c": 24} ) messages.append( {"role": "tool", "tool_call_id": tool_call.id, "content": result} ) final = client.chat.completions.create( model="kimi-k3", messages=messages, tools=tools, ) print(final.choices[0].message.content) See Tool Choice. Dynamic tool loading Place a complete tool definition in a system message without content. The tool becomes available from that message onward. Load a calculator dynamically
from typing import Any dynamic_messages: list[dict[str, Any]] = [ {"role": "user", "content": "Calculate 23 times 47."}, { "role": "system", "tools": [ { "type": "function", "function": { "name": "calculate", "description": "Evaluate an arithmetic expression", "parameters": { "type": "object", "properties": { "expression": { "type": "string", "description": "The arithmetic expression to evaluate", } }, "required": ["expression"], }, }, } ], }, ] completion = client.chat.completions.create( model="kimi-k3", messages=dynamic_messages, ) print(completion.choices[0].message.tool_calls) Include the complete name, description, and parameters definition. The declaration takes effect at its position in messages. Keep this message in later request history; the server does not retain it. See Dynamic Tool Loading. 1M context and automatic caching A new request can hit the prefix cache only when the previous request’s prompt tokens exceed 256. If the previous request’s prompt tokens are below 256, the request is not cached and is discarded. See Context Caching for details. Context caching is automatic for regular model requests; no cache ID, TTL, or extra parameter is required. Keep the long prefix unchanged so later requests can automatically attempt a cache hit. from pathlib import Path knowledge: str = Path("knowledge-base.md").read_text(encoding="utf-8") for question in ["Summarize the key conclusions.", "List three implementation risks."]: completion = client.chat.completions.create( model="kimi-k3", messages=[ {"role": "system", "content": knowledge}, {"role": "user", "content": question}, ], ) print(completion.choices[0].message.content) See Context Caching. Official tools Official tools are integrated through Formula: Fetch tool definitions from the Formula /tools endpoint. Add those definitions to the Chat Completions tools field. When the model returns tool_calls, submit each function name and arguments to the Formula /fibers endpoint. Add the complete assistant message and Fiber output as the corresponding tool message. Call Chat Completions again until the model returns a final answer. See Official Tools for the complete client and API contract. Web search is being updated and is not recommended for use in the near term. Important limits Reasoning effort is configured with the top-level reasoning_effort request field and supports low, high, and max (default max); K3 always has thinking mode enabled. max_completion_tokens defaults to 131072 and can be set up to 1048576. temperature=1.0, top_p=0.95, n=1, presence_penalty=0, and frequency_penalty=0 are fixed; omit them from requests. Return the complete assistant message unchanged in multi-turn conversations and tool calls. Vision input does not support public image URLs. Use base64 or ms://<file-id>, and make content an array of objects. Web search is being updated and is not recommended for production workflows in the near term. FAQ How is Kimi K3 billed?
Model Pricing For token pricing details, refer to Model Pricing. Related docs Reasoning EffortConfigure reasoning_effort.Vision InputSend images and videos.Structured OutputUse strict JSON Schema.Partial ModeContinue from a prefix.Tool ChoiceControl whether the model calls tools.Dynamic Tool LoadingInject tool definitions on demand.Tool Calling Best PracticesCombine tool-calling features.Official ToolsIntegrate Formula tools.Kimi K3 PricingReview input and output prices.Was this page helpful?
Official Source
Read the full original update directly at Modelverse Editorial.
Stay tuned to Modelverse for real-time model analysis, benchmark coverage, and AI news.
