استنتاج انعطاف‌پذیر

«میانای برنامه‌سازی کاربردی Gemini Flex» یک سطح استنباطی است که درمقایسه با نرخ‌های استاندارد، ۵۰٪ کاهش هزینه ارائه می‌دهد، درعوض تأخیر متغیر و دردسترس بودن با تمام تلاش را ارائه می‌دهد. این API برای حجم‌های کاری با تأخیرپذیری طراحی شده است که به پردازش هم‌زمان نیاز دارند اما به عملکرد هم‌زمان API استاندارد نیاز ندارند.

نحوه استفاده از Flex

برای استفاده از سطح Flex، در درخواستتان service_tier را به‌عنوان flex مشخص کنید. به‌طور پیش‌فرض، اگر این فیلد حذف شود، درخواست‌ها از سطح استاندارد استفاده می‌کنند.

Python

from google import genai

client = genai.Client()

interaction = client.interactions.create(
    model="gemini-3.8-flash",
    input="Analyze this dataset for trends...",
    service_tier='flex'
)
print(interaction.output_text)

JavaScript

import { GoogleGenAI } from '@google/genai';

const client = new GoogleGenAI({});

async function main() {
    const interaction = await client.interactions.create({
        model: 'gemini-3.8-flash',
        input: 'Analyze this dataset for trends...',
        service_tier: 'flex'
    });
    console.log(interaction.output_text);
}
await main();

جاوا

import com.google.genai.Client;
import com.google.genai.gaos.models.interactions.CreateModelInteraction;
import com.google.genai.gaos.models.interactions.Interaction;
import com.google.genai.gaos.models.interactions.InteractionsInput;
import com.google.genai.gaos.models.interactions.Model;
import com.google.genai.gaos.models.interactions.ServiceTier;
import com.google.genai.gaos.models.operations.CreateInteractionRequestBody;

Client client = new Client();

CreateModelInteraction params =
    CreateModelInteraction.builder()
        .model(Model.of("gemini-3.8-flash"))
        .input(InteractionsInput.of("Analyze this dataset for trends..."))
        .serviceTier(ServiceTier.FLEX)
        .build();

Interaction interaction =
    client.interactions.create(CreateInteractionRequestBody.of(params)).interaction().get();

System.out.println(interaction.outputText().orElse(""));

رفتن

package main

import (
    "context"
    "fmt"
    "log"

    "google.golang.org/genai"
    "google.golang.org/genai/interactions/models/interactions"
    "google.golang.org/genai/interactions/models/operations"
)

func main() {
    ctx := context.Background()
    client, err := genai.NewClient(ctx, nil)
    if err != nil {
        log.Fatal(err)
    }

    res, err := client.Interactions.Create(ctx, operations.CreateInteractionRequest{
        Body: operations.NewCreateInteractionRequestBody(interactions.CreateModelInteraction{
            Model:       interactions.Model("gemini-3.8-flash"),
            Input:       interactions.NewInteractionsInput("Analyze this dataset for trends..."),
            ServiceTier: interactions.ServiceTierFlex.ToPointer(),
        }),
    })
    if err != nil {
        log.Fatal(err)
    }
    if res.Interaction.OutputText != nil {
        fmt.Println(*res.Interaction.OutputText)
    }
}

REST

curl -X POST "https://generativelanguage.googleapis.com/v1beta/interactions" \
  -H "Content-Type: application/json" \
  -H "x-goog-api-key: $GEMINI_API_KEY" \
  -d '{
      "model": "gemini-3.8-flash",
      "input": "Analyze this dataset for trends...",
      "service_tier": "flex"
  }'

نحوه عملکرد استنتاج انعطاف‌پذیر

استنباط Gemini Flex شکاف بین «میانای برنامه‌سازی کاربردی» استاندارد و زمان پاسخ ۲۴ ساعته میانای برنامه‌سازی کاربردی دسته‌ای را پر می‌کند. این سرویس از ظرفیت محاسباتی خارج از اوج مصرف و «قابل‌کاهش» استفاده می‌کند تا راه‌حلی مقرون‌به‌صرفه برای وظایف پس‌زمینه‌ای و گردش‌های کار ترتیبی ارائه دهد.

ویژگی به‌رخ کشیدن اولویت استاندارد دسته
قیمت‌گذاری ‫٪۵۰ تخفیف ‫۷۵ تا ۱۰۰٪ بیشتر از «استاندارد» قیمت کامل ‫٪۵۰ تخفیف
تأخیر دقیقه (هدف ۱ تا ۱۵ دقیقه) کم (ثانیه) ثانیه به دقیقه حداکثر ۲۴ ساعت
قابلیت اطمینان نهایت تلاش (قابل‌حذف) زیاد (غیرقابل ریزش) زیاد / متوسط روبه‌بالا بالا (برای توان گذرداد)
میانای هم‌زمان هم‌زمان هم‌زمان غیرهمگام

مزایای کلیدی

  • کارایی هزینه: صرفه‌جویی قابل‌توجه برای ارزیابی‌های غیرتولیدی، کارگزاران پس‌زمینه، و غنی‌سازی داده‌ها.
  • اصطکاک کم: به‌سادگی یک پارامتر به درخواست‌های موجود خود اضافه کنید.
  • گردش کارهای هم‌زمان: برای زنجیره‌های متوالی میانای برنامه‌سازی کاربردی که در آن درخواست بعدی به برونداد درخواست قبلی بستگی دارد ایده‌آل است و آن را انعطاف‌پذیرتر از «دسته‌ای» برای گردش کارهای عامل‌گرا می‌کند.

موارد استفاده

  • ارزیابی‌های آفلاین: اجرای آزمون‌های پس‌رفت «مدل زبانی بزرگ به‌عنوان داور» یا تابلوهای امتیازات.
  • عامل‌های پس‌زمینه: تکالیف ترتیبی مثل به‌روزرسانی‌های CRM، ساختن نمایه‌های شخصی، یا مدیریت محتوا که در آن‌ها چند دقیقه تأخیر قابل‌قبول است.
  • پژوهش با بودجه محدود: آزمایش‌های آکادمیک که به حجم بالای کد با بودجه محدود نیاز دارند.

محدودیت‌های نرخ

ترافیک استنباط انعطاف‌پذیر در حدود نرخ کلی شما محاسبه می‌شود؛ این ترافیک حدود نرخ گسترده‌ای مانند Batch API ارائه نمی‌دهد.

ظرفیت قابل‌کاهش

با ترافیک انعطاف‌پذیر با اولویت پایین‌تر رفتار می‌شود. اگر افزایش ناگهانی در ترافیک استاندارد وجود داشته باشد، درخواست‌های Flex ممکن است برای اطمینان از ظرفیت برای کاربران با اولویت بالا، پیش‌گیری یا حذف شوند. اگر به‌دنبال استنباط اولویت بالا هستید، استنباط اولویت‌دار را بررسی کنید

کدهای خطا

وقتی ظرفیت «انعطاف‌پذیر» دردسترس نباشد یا سیستم دچار ازدحام باشد، «میانای برنامه‌سازی کاربردی» کدهای خطای استاندارد را برمی‌گرداند:

  • ‫۵۰۳ سرویس دردسترس نیست: درحال‌حاضر، سیستم به حداکثر ظرفیت رسیده است.
  • ‫429 درخواست‌های بیش‌ازحد: محدودیت‌های نرخ یا اتمام منابع.

مسئولیت کارخواه

  • بدون بازگشت به سمت سرور: برای جلوگیری از هزینه‌های غیرمنتظره، اگر ظرفیت Flex تکمیل شده باشد، سیستم به‌طور خودکار درخواست Flex را به سطح «استاندارد» ارتقا نخواهد داد.
  • تلاش‌های مجدد: باید منطق تلاش مجدد سمت کارخواه خودتان را با پس‌گیری نمایی پیاده‌سازی کنید.
  • زمان‌های اتمام: ازآنجایی‌که درخواست‌های Flex ممکن است در صف قرار بگیرند، توصیه می‌کنیم زمان‌های اتمام سمت مشتری را به ۱۰ دقیقه یا بیشتر افزایش دهید تا از بسته شدن زودرس اتصال جلوگیری شود.

تنظیم کردن پنجره‌های مهلت زمانی

می‌توانید زمان‌های اتمام درخواست را برای REST API و کتابخانه‌های کارخواه پیکربندی کنید. همیشه مطمئن شوید که مهلت زمانی سمت مشتری شما پنجره صبر سرور موردنظر را پوشش می‌دهد (برای نمونه، ۶۰۰ ثانیه و بیشتر برای صف‌های انتظار Flex). کیت‌های توسعه نرم‌افزار مقادیر درنگ را برحسب میلی‌ثانیه انتظار دارند.

زمان‌های اتمام درخواست

Python

from google import genai

client = genai.Client(http_options={"timeout": 900000})

interaction = client.interactions.create(
    model="gemini-3.8-flash",
    input="why is the sky blue?",
    service_tier="flex",
)

JavaScript

import { GoogleGenAI } from '@google/genai';

const client = new GoogleGenAI({});

async function main() {
    const interaction = await client.interactions.create({
        model: "gemini-3.8-flash",
        input: "why is the sky blue?",
        service_tier: "flex",
    }, {timeout: 900000});
}

await main();

جاوا

import com.google.genai.Client;
import com.google.genai.gaos.models.interactions.CreateModelInteraction;
import com.google.genai.gaos.models.interactions.Interaction;
import com.google.genai.gaos.models.interactions.InteractionsInput;
import com.google.genai.gaos.models.interactions.Model;
import com.google.genai.gaos.models.interactions.ServiceTier;
import com.google.genai.gaos.models.operations.CreateInteractionRequestBody;
import com.google.genai.types.HttpOptions;

Client client =
    Client.builder()
        .httpOptions(HttpOptions.builder().timeout(900000).build())
        .build();

CreateModelInteraction params =
    CreateModelInteraction.builder()
        .model(Model.of("gemini-3.8-flash"))
        .input(InteractionsInput.of("why is the sky blue?"))
        .serviceTier(ServiceTier.FLEX)
        .build();

Interaction interaction =
    client.interactions.create(CreateInteractionRequestBody.of(params)).interaction().get();

رفتن

package main

import (
    "context"
    "log"
    "time"

    "google.golang.org/genai"
    "google.golang.org/genai/interactions/models/interactions"
    "google.golang.org/genai/interactions/models/operations"
)

func main() {
    ctx := context.Background()
    client, err := genai.NewClient(ctx, &genai.ClientConfig{
        HTTPOptions: genai.HTTPOptions{
            Timeout: genai.Ptr(15 * time.Minute),
        },
    })
    if err != nil {
        log.Fatal(err)
    }

    res, err := client.Interactions.Create(ctx, operations.CreateInteractionRequest{
        Body: operations.NewCreateInteractionRequestBody(interactions.CreateModelInteraction{
            Model:       interactions.Model("gemini-3.8-flash"),
            Input:       interactions.NewInteractionsInput("why is the sky blue?"),
            ServiceTier: interactions.ServiceTierFlex.ToPointer(),
        }),
    })
    if err != nil {
        log.Fatal(err)
    }
    _ = res
}

پیاده‌سازی تلاش‌های مجدد

ازآنجایی‌که Flex قابل‌حذف است و با خطاهای ۵۰۳ ازکار می‌افتد، در اینجا نمونه‌ای از پیاده‌سازی اختیاری منطق تلاش مجدد برای ادامه دادن با درخواست‌های ناموفق آورده شده است:

Python

import time
from google import genai

client = genai.Client()

def call_with_retry(max_retries=3, base_delay=5):
    for attempt in range(max_retries):
        try:
            return client.interactions.create(
                model="gemini-3.8-flash",
                input="Analyze this batch statement.",
                service_tier="flex",
            )
        except Exception as e:
            if attempt < max_retries - 1:
                delay = base_delay * (2 ** attempt) # Exponential Backoff
                print(f"Flex busy, retrying in {delay}s...")
                time.sleep(delay)
            else:
                print("Flex exhausted, falling back to Standard...")
                return client.interactions.create(
                    model="gemini-3.8-flash",
                    input="Analyze this batch statement."
                )

interaction = call_with_retry()
print(interaction.output_text)

JavaScript

import { GoogleGenAI } from '@google/genai';

const ai = new GoogleGenAI({});

async function sleep(ms) {
  return new Promise(resolve => setTimeout(resolve, ms));
}

async function callWithRetry(maxRetries = 3, baseDelay = 5) {
  for (let attempt = 0; attempt < maxRetries; attempt++) {
    try {
      console.log(`Attempt ${attempt + 1}: Calling Flex tier...`);
      const interaction = await ai.interactions.create({
        model: "gemini-3.8-flash",
        input: "Analyze this batch statement.",
        service_tier: 'flex',
      });
      return interaction;
    } catch (e) {
      if (attempt < maxRetries - 1) {
        const delay = baseDelay * (2 ** attempt);
        console.log(`Flex busy, retrying in ${delay}s...`);
        await sleep(delay * 1000);
      } else {
        console.log("Flex exhausted, falling back to Standard...");
        return await ai.interactions.create({
          model: "gemini-3.8-flash",
          input: "Analyze this batch statement.",
        });
      }
    }
  }
}

async function main() {
    const interaction = await callWithRetry();
    console.log(interaction.output_text);
}

await main();

جاوا

import com.google.genai.Client;
import com.google.genai.gaos.models.interactions.CreateModelInteraction;
import com.google.genai.gaos.models.interactions.Interaction;
import com.google.genai.gaos.models.interactions.InteractionsInput;
import com.google.genai.gaos.models.interactions.Model;
import com.google.genai.gaos.models.interactions.ServiceTier;
import com.google.genai.gaos.models.operations.CreateInteractionRequestBody;

Client client = new Client();

int maxRetries = 3;
int baseDelay = 5;
Interaction interaction = null;

for (int attempt = 0; attempt < maxRetries; attempt++) {
  try {
    CreateModelInteraction flexParams =
        CreateModelInteraction.builder()
            .model(Model.of("gemini-3.8-flash"))
            .input(InteractionsInput.of("Analyze this batch statement."))
            .serviceTier(ServiceTier.FLEX)
            .build();
    interaction =
        client.interactions.create(CreateInteractionRequestBody.of(flexParams)).interaction().get();
    break;
  } catch (Exception e) {
    if (attempt < maxRetries - 1) {
      int delay = baseDelay * (1 << attempt); // Exponential Backoff
      System.out.println("Flex busy, retrying in " + delay + "s...");
      Thread.sleep(delay * 1000L);
    } else {
      System.out.println("Flex exhausted, falling back to Standard...");
      CreateModelInteraction standardParams =
          CreateModelInteraction.builder()
              .model(Model.of("gemini-3.8-flash"))
              .input(InteractionsInput.of("Analyze this batch statement."))
              .build();
      interaction =
          client
              .interactions
              .create(CreateInteractionRequestBody.of(standardParams))
              .interaction()
              .get();
    }
  }
}

if (interaction != null) {
  System.out.println(interaction.outputText().orElse(""));
}

رفتن

package main

import (
    "context"
    "fmt"
    "log"
    "time"

    "google.golang.org/genai"
    "google.golang.org/genai/interactions/models/interactions"
    "google.golang.org/genai/interactions/models/operations"
)

func main() {
    ctx := context.Background()
    client, err := genai.NewClient(ctx, nil)
    if err != nil {
        log.Fatal(err)
    }

    maxRetries := 3
    baseDelay := 5
    var interaction *interactions.Interaction

    for attempt := 0; attempt < maxRetries; attempt++ {
        res, err := client.Interactions.Create(ctx, operations.CreateInteractionRequest{
            Body: operations.NewCreateInteractionRequestBody(interactions.CreateModelInteraction{
                Model:       interactions.Model("gemini-3.8-flash"),
                Input:       interactions.NewInteractionsInput("Analyze this batch statement."),
                ServiceTier: interactions.ServiceTierFlex.ToPointer(),
            }),
        })
        if err == nil {
            interaction = res.Interaction
            break
        }

        if attempt < maxRetries-1 {
            delay := baseDelay * (1 << attempt) // Exponential Backoff
            fmt.Printf("Flex busy, retrying in %ds...\n", delay)
            time.Sleep(time.Duration(delay) * time.Second)
        } else {
            fmt.Println("Flex exhausted, falling back to Standard...")
            stdRes, err := client.Interactions.Create(ctx, operations.CreateInteractionRequest{
                Body: operations.NewCreateInteractionRequestBody(interactions.CreateModelInteraction{
                    Model: interactions.Model("gemini-3.8-flash"),
                    Input: interactions.NewInteractionsInput("Analyze this batch statement."),
                }),
            })
            if err != nil {
                log.Fatal(err)
            }
            interaction = stdRes.Interaction
        }
    }

    if interaction != nil && interaction.OutputText != nil {
        fmt.Println(*interaction.OutputText)
    }
}

قیمت‌گذاری

قیمت استنباط انعطاف‌پذیر ۵۰٪ از میانای برنامه‌سازی کاربردی استاندارد است و براساس تعداد نشان صورت‌حساب می‌شود.

مدل‌های پشتیبانی‌شده

مدل‌های زیر از استنتاج انعطاف‌پذیر پشتیبانی می‌کنند:

مدل استنتاج انعطاف‌پذیر
Gemini 3.8 Flash ✔️
Gemini 3.6 Flash ✔️
Gemini 3.5 Flash-Lite ✔️
Gemini 3.1 Flash-Lite ✔️
پیش‌نمایش Gemini 3.1 Pro ✔️
پیش‌نمایش Gemini 3 Flash ✔️
Gemini 2.5 Pro ✔️
Gemini 2.5 Flash ✔️
Gemini 2.5 Flash-Lite ✔️

قدم بعدی چیست