curl --request POST \
--url https://api.siliconflow.com/v1/embeddings \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "Qwen/Qwen3-Embedding-8B",
"input": "Silicon flow embedding online: fast, affordable, and high-quality embedding services. come try it out!"
}
'import requests
url = "https://api.siliconflow.com/v1/embeddings"
payload = {
"model": "Qwen/Qwen3-Embedding-8B",
"input": "Silicon flow embedding online: fast, affordable, and high-quality embedding services. come try it out!"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'Qwen/Qwen3-Embedding-8B',
input: 'Silicon flow embedding online: fast, affordable, and high-quality embedding services. come try it out!'
})
};
fetch('https://api.siliconflow.com/v1/embeddings', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));{
"object": [
"list"
],
"model": "<string>",
"data": [
{
"object": "embedding",
"embedding": [
123
],
"index": 123
}
],
"usage": {
"prompt_tokens": 123,
"completion_tokens": 123,
"total_tokens": 123
}
}{
"code": 20012,
"message": "<string>",
"data": "<string>"
}"Invalid token""404 page not found"{
"message": "Request was rejected due to rate limiting. If you want more, please contact contact@siliconflow.com. Details:TPM limit reached.",
"data": "<string>"
}{
"code": 50505,
"message": "Model service overloaded. Please try again later.",
"data": "<string>"
}"<string>"创建嵌入请求
Creates an embedding vector representing the input text.
curl --request POST \
--url https://api.siliconflow.com/v1/embeddings \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"model": "Qwen/Qwen3-Embedding-8B",
"input": "Silicon flow embedding online: fast, affordable, and high-quality embedding services. come try it out!"
}
'import requests
url = "https://api.siliconflow.com/v1/embeddings"
payload = {
"model": "Qwen/Qwen3-Embedding-8B",
"input": "Silicon flow embedding online: fast, affordable, and high-quality embedding services. come try it out!"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
model: 'Qwen/Qwen3-Embedding-8B',
input: 'Silicon flow embedding online: fast, affordable, and high-quality embedding services. come try it out!'
})
};
fetch('https://api.siliconflow.com/v1/embeddings', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));{
"object": [
"list"
],
"model": "<string>",
"data": [
{
"object": "embedding",
"embedding": [
123
],
"index": 123
}
],
"usage": {
"prompt_tokens": 123,
"completion_tokens": 123,
"total_tokens": 123
}
}{
"code": 20012,
"message": "<string>",
"data": "<string>"
}"Invalid token""404 page not found"{
"message": "Request was rejected due to rate limiting. If you want more, please contact contact@siliconflow.com. Details:TPM limit reached.",
"data": "<string>"
}{
"code": 50505,
"message": "Model service overloaded. Please try again later.",
"data": "<string>"
}"<string>"Authorizations
Body
Corresponding Model Name. To better enhance service quality, we will make periodic changes to the models provided by this service, including but not limited to model on/offlining and adjustments to model service capabilities. We will notify you of such changes through appropriate means such as announcements or message pushes where feasible.
Qwen/Qwen3-Embedding-8B, Qwen/Qwen3-Embedding-4B, Qwen/Qwen3-Embedding-0.6B "Qwen/Qwen3-Embedding-8B"
Input text to embed must be provided as a string or an array of tokens. To process multiple inputs in a single request, pass an array of strings or an array of token arrays. The input length must not exceed the model's maximum token limit and should not be an empty string. The maximum input tokens for each model are as follows:
BAAI/bge-large-zh-v1.5, BAAI/bge-large-en-v1.5, netease-youdao/bce-embedding-base_v1: 512 BAAI/bge-m3: 8192 Qwen/Qwen3-Embedding-8B, Qwen/Qwen3-Embedding-4B, Qwen/Qwen3-Embedding-0.6B: 32768
"Silicon flow embedding online: fast, affordable, and high-quality embedding services. come try it out!"
The number of dimensions the resulting output embeddings should have. Only supported in Qwen/Qwen3 series. - Qwen/Qwen3-Embedding-8B: [64,128,256,512,768,1024,2048,4096] - Qwen/Qwen3-Embedding-4B:[64,128,256,512,768,1024,2048] - Qwen/Qwen3-Embedding-0.6B: [64,128,256,512,768,1024]
1024
Response
200
The object type, which is always "list".
list The name of the model used to generate the embedding.
The list of embeddings generated by the model.
Show child attributes
Show child attributes
The usage information for the request.
Show child attributes
Show child attributes