ການນັບ token
ນັບ tokens ຂອງຂໍ້ຄວາມ ຫຼື ຂອງທັງຄຳຮ້ອງຂໍກ່ອນທ່ານສົ່ງ.
POST https://api.shannon-ai.com/v1/tokenize
POST https://api.shannon-ai.com/v1/messages/count_tokens
ທັງສອງ endpoint ນັບດ້ວຍ tokenizer ຂອງ model ທີ່ທ່ານລະບຸ, ແລະ ບໍ່ມີ model ຮັນ. ພວກມັນຄອບຄຸມ hosted open-weight models. /v1/tokenize ຮັບຂໍ້ຄວາມທຳມະດາ ຫຼື ການສົນທະນາຮູບແບບ Chat Completions. /v1/messages/count_tokens ຮັບຄຳຮ້ອງຂໍໃນຮູບແບບ Anthropic Messages, ເຊິ່ງເປັນການເອີ້ນທີ່ SDK ຂອງ Anthropic ແລະ Claude Code ເຮັດ.
ການນັບບໍ່ເສຍຄ່າ. ການເອີ້ນຕ້ອງໃຊ້ API key ຂອງທ່ານ, ບໍ່ຫັກຫຍັງຈາກຍອດເງິນຂອງທ່ານ ແລະ ບໍ່ປາກົດໃນ usage log ຂອງທ່ານ.
ນັບຂໍ້ຄວາມ
ສົ່ງ model ແລະ text. ຂໍ້ຄວາມຖືກນັບຕາມທີ່ມັນເປັນ, ໂດຍບໍ່ມີການຈັດຮູບແບບ chat ອ້ອມຮອບ.
import requests
response = requests.post(
"https://api.shannon-ai.com/v1/tokenize",
headers={"Authorization": "Bearer YOUR_API_KEY"},
json={
"model": "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
"text": "Hello, world",
},
)
print(response.json()["tokens"]) const response = await fetch("https://api.shannon-ai.com/v1/tokenize", {
method: "POST",
headers: {
Authorization: "Bearer YOUR_API_KEY",
"Content-Type": "application/json",
},
body: JSON.stringify({
model: "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
text: "Hello, world",
}),
});
const { tokens } = await response.json();
console.log(tokens); curl https://api.shannon-ai.com/v1/tokenize \
-H "Authorization: Bearer YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
"text": "Hello, world"
}' {
"model": "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
"tokens": 3
} ຕົວເລກໃນຄຳຕອບຂອງໜ້ານີ້ເປັນຕົວຢ່າງ. ຂໍ້ຄວາມດຽວກັນໃຫ້ຈຳນວນນັບຕ່າງກັນໃນ model ອື່ນ.
ນັບຄຳຮ້ອງຂໍ chat
ສົ່ງ model ແລະ messages, ພ້ອມ tools ເມື່ອຄຳຮ້ອງຂໍມີ, ຄືກັນກັບທີ່ທ່ານສົ່ງໄປ /v1/chat/completions. ຄຳຕອບແມ່ນຂະໜາດຂອງ input ທັງໝົດ.
import requests
request = {
"model": "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
"messages": [
{"role": "system", "content": "You are a concise assistant."},
{"role": "user", "content": "What is the weather in Paris?"},
],
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Current weather for a city",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"],
},
},
}
],
}
response = requests.post(
"https://api.shannon-ai.com/v1/tokenize",
headers={"Authorization": "Bearer YOUR_API_KEY"},
json=request,
)
print(response.json()["tokens"]) const request = {
model: "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
messages: [
{ role: "system", content: "You are a concise assistant." },
{ role: "user", content: "What is the weather in Paris?" },
],
tools: [
{
type: "function",
function: {
name: "get_weather",
description: "Current weather for a city",
parameters: {
type: "object",
properties: { city: { type: "string" } },
required: ["city"],
},
},
},
],
};
const response = await fetch("https://api.shannon-ai.com/v1/tokenize", {
method: "POST",
headers: {
Authorization: "Bearer YOUR_API_KEY",
"Content-Type": "application/json",
},
body: JSON.stringify(request),
});
const { tokens } = await response.json();
console.log(tokens); curl https://api.shannon-ai.com/v1/tokenize \
-H "Authorization: Bearer YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
"messages": [
{"role": "system", "content": "You are a concise assistant."},
{"role": "user", "content": "What is the weather in Paris?"}
],
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Current weather for a city",
"parameters": {
"type": "object",
"properties": {"city": {"type": "string"}},
"required": ["city"]
}
}
}
]
}' {
"model": "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
"tokens": 164
} ຟີລດ໌ຂອງ /v1/tokenize
| ຟີລດ໌ | ປະເພດ | ຄຳອະທິບາຍ |
|---|---|---|
model | string | ຈຳເປັນ. id ຂອງ hosted open-weight model. ຕົວພິມນ້ອຍ-ໃຫຍ່ຖືກຖືວ່າຄືກັນ. |
text | string | ຂໍ້ຄວາມທີ່ຈະນັບຕາມທີ່ມັນເປັນ, ໂດຍບໍ່ມີການຈັດຮູບແບບ chat. ສູງສຸດ 4,000,000 bytes. ສົ່ງ text ຫຼື messages; ເມື່ອມີທັງສອງ, text ຖືກນັບ. |
messages | array | ຂໍ້ຄວາມ chat ໃນຮູບແບບ Chat Completions. ພວກມັນຖືກນັບເປັນ input ເຕັມຂອງຄຳຮ້ອງຂໍ: ທຸກຂໍ້ຄວາມພ້ອມການຈັດຮູບແບບທີ່ chat template ຂອງ model ໃສ່ອ້ອມມັນ. |
tools | array | ຄຳນິຍາມ tool ທີ່ຈະລວມໃນການນັບ. ໃຊ້ຮ່ວມກັບ messages. |
ຄຳຕອບແມ່ນ object JSON ທີ່ມີຟີລດ໌ເຫຼົ່ານີ້:
| ຟີລດ໌ | ປະເພດ | ຄຳອະທິບາຍ |
|---|---|---|
model | string | id ຂອງ model ທີ່ການນັບເຮັດໃຫ້, ໃນການສະກົດທີ່ເຜີຍແຜ່. |
tokens | integer | ກັບ text: tokens ຂອງຂໍ້ຄວາມ. ກັບ messages: tokens ຂອງ input ທັງໝົດ, ລວມທັງຮູບພາບ. |
ນັບຄຳຮ້ອງຂໍ Messages
ສົ່ງ body ທີ່ທ່ານຈະສົ່ງໄປ /v1/messages: model, messages, ແລະ system ກັບ tools ເມື່ອທ່ານໃຊ້. SDK ທາງການຂອງ Anthropic ເອີ້ນ endpoint ນີ້ຜ່ານ messages.count_tokens.
import anthropic
client = anthropic.Anthropic(
api_key="YOUR_API_KEY",
base_url="https://api.shannon-ai.com",
)
count = client.messages.count_tokens(
model="DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
system="You are a concise assistant.",
messages=[
{"role": "user", "content": "Summarise the attached report."}
],
)
print(count.input_tokens) import Anthropic from "@anthropic-ai/sdk";
const client = new Anthropic({
apiKey: "YOUR_API_KEY",
baseURL: "https://api.shannon-ai.com",
});
const count = await client.messages.countTokens({
model: "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
system: "You are a concise assistant.",
messages: [
{ role: "user", content: "Summarise the attached report." },
],
});
console.log(count.input_tokens); curl https://api.shannon-ai.com/v1/messages/count_tokens \
-H "x-api-key: YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "DeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAP",
"system": "You are a concise assistant.",
"messages": [
{"role": "user", "content": "Summarise the attached report."}
]
}' {
"input_tokens": 21
} ຟີລດ໌ທີ່ຮອງຮັບຂອງ endpoint /v1/messages/count_tokens
| ຟີລດ໌ | ປະເພດ | ຄຳອະທິບາຍ |
|---|---|---|
model | string | ຈຳເປັນ. id ຂອງ model ທີ່ເປັນ open-weight ແລະ ຖືກ host ໄວ້. |
messages | array | ຈຳເປັນ. ຂໍ້ຄວາມໃນຮູບແບບ Anthropic Messages. block text, image, tool_use ແລະ tool_result ຖືກນັບ. |
system | string | array | system prompt: ເປັນ string ຫຼື array ຂອງ block ຂໍ້ຄວາມ. |
tools | array | ຄຳນິຍາມ tool ທີ່ມີ name, description ແລະ input_schema. |
ຮັບໄວ້ເພື່ອຄວາມເຂົ້າກັນໄດ້, ໂດຍບໍ່ມີຜົນຕໍ່ການນັບ: tool_choice, max_tokens, temperature, top_p, stop_sequences, stream, thinking. ທ່ານສົ່ງ body ຂອງຄຳຮ້ອງຂໍຈິງໂດຍບໍ່ປ່ຽນໄດ້.
ຄຳຕອບແມ່ນ object JSON ທີ່ມີຟີລດ໌ເຫຼົ່ານີ້:
| ຟີລດ໌ | ປະເພດ | ຄຳອະທິບາຍ |
|---|---|---|
input_tokens | integer | tokens ຂອງ input ທັງໝົດ: system prompt, ຂໍ້ຄວາມ, tools ແລະ ຮູບພາບ. |
model ທີ່ຮອງຮັບ
ທັງສອງ endpoint ນັບໃຫ້ hosted open-weight models. GET /v1/models ລາຍຊື່ /v1/tokenize ແລະ /v1/messages/count_tokens ໃນ endpoints ຂອງແຕ່ລະ model ທີ່ຮອງຮັບ. ຄ່າ model ອື່ນໃດ, ລວມທັງ id ຂອງ Shannon, ຖືກຕອບດ້ວຍ 400.
DeepSeek-V4-Pro-0813-3BIT-REAPGLM-5.2-3BIT-REAPKimi-K3-3BIT-REAPNemotron3Ultra-3BIT-REAPMiniMax-M3-3BIT-REAPDeepSeek-V4-Flash-0731-W4A16-AUTOROUND-REAPKimi-K2.6-W4A16-AUTOROUND-REAPLaguna-S-2.1-W4A16-AUTOROUND-REAPinkling-W4A16-AUTOROUND-REAPMiMo-V2.5-Pro-W8A16MiMo-V2.5-W8A16Hy3-W8A16
ສຳລັບ model Shannon, ໃຫ້ອ່ານຈຳນວນ tokens ຈາກ object usage ຂອງຄຳຕອບ.
ການນັບເຮັດແນວໃດ
ແຕ່ລະ model ຖືກນັບດ້ວຍ tokenizer ແລະ chat template ຂອງມັນເອງ. ບໍ່ມີການຄາດຄະເນຈາກຕົວອັກສອນ ຫຼື ຄຳ.
| ສິ່ງທີ່ຖືກນັບ | ກົດ |
|---|---|
| ຂໍ້ຄວາມ | tokens ຂອງ string ຕາມທີ່ສົ່ງ. string ວ່າງນັບເປັນ 0. |
| ຂໍ້ຄວາມ | ຂໍ້ຄວາມ ແລະ tools ຖືກຈັດວາງດ້ວຍ chat template ຂອງ model ເອງ, ຈົນຮອດຈຸດທີ່ຄຳຕອບເລີ່ມ, ແລະ prompt ທັງໝົດນັ້ນຖືກນັບ. |
| Role | ຂໍ້ຄວາມ system, user, assistant ແລະ tool ຖືກນັບ. developer ຖືກນັບເປັນ system. ຂໍ້ຄວາມທີ່ບໍ່ມີເນື້ອຫາ ແລະ ບໍ່ມີການເອີ້ນ tool ບໍ່ເພີ່ມຫຍັງ. |
| ການເອີ້ນ tool ແລະ ຜົນລັບ | ການເອີ້ນ tool ຂອງຮອບ assistant ກ່ອນໜ້າ ແລະ ຜົນລັບຂອງມັນເປັນສ່ວນໜຶ່ງຂອງການນັບ, ໃນທັງສອງ endpoint. |
| ຮູບພາບ | ຮູບພາບທີ່ສົ່ງພາຍໃນ body (base64 ຫຼື data: URL) ເພີ່ມໜຶ່ງ token ຕໍ່ patch ຂະໜາດ 28 × 28 ພິກເຊວ: ceil(width / 28) × ceil(height / 28). ຮູບພາບທີ່ໃຫ້ເປັນ URL http(s) ບໍ່ຖືກດາວໂຫຼດໂດຍ endpoint ເຫຼົ່ານີ້ ແລະ ນັບເປັນ 1,024. |
ຕົວຢ່າງ: ຮູບພາບຂະໜາດ 1,024 × 768 ພິກເຊວນັບເປັນ ceil(1024 / 28) × ceil(768 / 28) = 37 × 28 = 1,036 tokens.
ການນັບ ແລະ ສິ່ງທີ່ຄຳຮ້ອງຂໍຖືກຄິດເງິນ
ການນັບຂອງທັງຄຳຮ້ອງຂໍເຮັດແບບດຽວກັນກັບການນັບ input ຂອງຄຳຮ້ອງຂໍຈິງທີ່ມີ model, ຂໍ້ຄວາມ ແລະ tools ດຽວກັນ. ຄຳຕອບລາຍງານຕົວເລກນັ້ນເປັນ usage.prompt_tokens ໃນ Chat Completions, ເປັນ usage.input_tokens ໃນ Responses, ແລະ ເປັນ usage.input_tokens ບວກ usage.cache_read_input_tokens ໃນ Messages.
- ການນັບແມ່ນ input ກ່ອນສ່ວນຫຼຸດ cached-input. ຄຳຮ້ອງຂໍຈິງອາດອ່ານສ່ວນໜຶ່ງຂອງ input ນັ້ນຈາກ cache ແລະ ຄິດເງິນສ່ວນນັ້ນໃນອັດຕາ cached. ການ Cache Prompt
- ຮູບພາບທີ່ໃຫ້ເປັນ URL
http(s)ນັບເປັນ 1,024 ໃນທີ່ນີ້. ຄຳຮ້ອງຂໍຈິງດາວໂຫຼດຮູບພາບ ແລະ ນັບຈາກຂະໜາດເປັນພິກເຊວ, ດັ່ງນັ້ນສອງຕົວເລກອາດຕ່າງກັນ. ສົ່ງຮູບພາບເປັນ base64 ເພື່ອໃຫ້ໄດ້ຕົວເລກດຽວກັນ. - Output ບໍ່ແມ່ນສ່ວນໜຶ່ງຂອງການນັບ. ຄຳຕອບຂອງຄຳຮ້ອງຂໍຈິງຖືກຄິດເງິນເປັນ output tokens ເພີ່ມ, ລວມທັງ reasoning.
- ການນັບ
textບໍ່ມີການຈັດຮູບແບບ chat. ໃຊ້ມັນເພື່ອວັດເອກະສານ ຫຼື ສ່ວນໜຶ່ງຂອງ prompt, ແລະ ຮູບແບບmessagesເພື່ອວັດຄຳຮ້ອງຂໍ.
ເພື່ອປ່ຽນຈຳນວນນັບເປັນຄ່າໃຊ້ຈ່າຍ, ຄູນມັນກັບລາຄາ input ຕໍ່ 1M tokens ຂອງ model. Model ແລະ ລາຄາ
ຂີດຈຳກັດ
| ຂີດຈຳກັດ | ຄ່າ | ເມື່ອເກີນ |
|---|---|---|
ຄວາມຍາວຂອງ text | 4,000,000 bytes (UTF-8) | 413 ພ້ອມຂໍ້ຄວາມ text too long |
| body ຂອງຄຳຮ້ອງຂໍ | 32 MiB | 413 |
| ຕໍ່ຄຳຮ້ອງຂໍ | ໜຶ່ງຂໍ້ຄວາມ ຫຼື ໜຶ່ງການສົນທະນາ | ສົ່ງໜຶ່ງຄຳຮ້ອງຂໍຕໍ່ໜຶ່ງຂໍ້ຄວາມ ເພື່ອນັບຫຼາຍຂໍ້ຄວາມ. |
ການເອີ້ນນັບບໍ່ຖືກນັບເຂົ້າຂີດຈຳກັດ 120 ຄຳຮ້ອງຂໍຕໍ່ນາທີ. ຂີດຈຳກັດ ແລະ ຍອດເງິນ
ຂໍ້ຜິດພາດ
| Status | ປະເພດ | ຂໍ້ຄວາມ | ເມື່ອໃດ |
|---|---|---|---|
400 | invalid_request_error | tokenize is available for the hosted open models; unknown model: <model> | /v1/tokenize ທີ່ມີ model ທີ່ບໍ່ແມ່ນ id ຂອງ hosted open-weight. |
400 | invalid_request_error | count_tokens is available for the hosted open models; unknown model: <model> | /v1/messages/count_tokens ທີ່ມີ model ທີ່ບໍ່ແມ່ນ id ຂອງ hosted open-weight, ຫຼື ບໍ່ມີ model. |
400 | invalid_request_error | send `text` or `messages` | /v1/tokenize ທີ່ບໍ່ມີທັງ text ແລະ messages. |
401 | authentication_error | Missing authentication / Invalid API key | ບໍ່ໄດ້ສົ່ງ key, ຫຼື key ບໍ່ຖືກຕ້ອງ. |
413 | invalid_request_error | text too long | text ຍາວກວ່າ 4,000,000 bytes. body ທີ່ເກີນ 32 MiB ກໍຖືກຕອບດ້ວຍ 413 ເຊັ່ນກັນ. |
415 | invalid_request_error | Expected request with `Content-Type: application/json` | ຄຳຮ້ອງຂໍບໍ່ມີ content type ເປັນ JSON. |
422 | invalid_request_error | Failed to deserialize the JSON body into the target type: … | ຂາດຟີລດ໌ທີ່ຈຳເປັນ (model ໃນ /v1/tokenize, messages ໃນ /v1/messages/count_tokens) ຫຼື ຟີລດ໌ມີປະເພດຜິດ. |
503 | api_error | token counting is temporarily unavailable for this model | ບໍ່ສາມາດນັບສຳລັບ model ນີ້ໄດ້ໃນເວລານີ້. ລອງໃໝ່ພາຍຫຼັງ. |
/v1/tokenize ສົ່ງຂໍ້ຜິດພາດໃນຮູບຮ່າງ OpenAI. ໃນ /v1/messages/count_tokens ຂໍ້ຜິດພາດຂອງ endpoint ເອງ (400 ສຳລັບ model, 503) ມາໃນຮູບຮ່າງ Anthropic, ແລະ 401, 413, 415 ແລະ 422 ມາໃນຮູບຮ່າງ OpenAI. ອ່ານ status code ກ່ອນ, ແລ້ວຈຶ່ງ error.type ແລະ error.message, ເຊິ່ງມີໃນທັງສອງຮູບຮ່າງ.
{
"error": {
"type": "invalid_request_error",
"message": "tokenize is available for the hosted open models; unknown model: shannon-3"
}
} {
"type": "error",
"error": {
"type": "invalid_request_error",
"message": "count_tokens is available for the hosted open models; unknown model: shannon-3"
}
}