> ## Documentation Index
> Fetch the complete documentation index at: https://developers.telnyx.com/llms.txt
> Use this file to discover all available pages before exploring further.

# Create gateway token group

> Create gateway token group within the authenticated Telnyx account. Foreign and unknown resource IDs return the same 404 response.



## OpenAPI

````yaml /openapi/source/external/inference/ai-gateway-resources.json post /llm_token_gateway/token_groups
openapi: 3.1.0
info:
  title: AI Gateway configuration
  version: 1.0.0
  description: >-
    Manage account-scoped AI Gateway groups and inference keys. All mutations
    require Idempotency-Key; updates and deletes require the current resource
    ETag in If-Match. Nullable fields clear on null and omitted PATCH fields are
    preserved.
servers:
  - url: https://api.telnyx.com/v2
security: []
paths:
  /llm_token_gateway/token_groups:
    post:
      tags:
        - AI Gateway
      summary: Create gateway token group
      description: >-
        Create gateway token group within the authenticated Telnyx account.
        Foreign and unknown resource IDs return the same 404 response.
      operationId: create_token_groups
      parameters:
        - $ref: '#/components/parameters/IdempotencyKey'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/TokenGroupCreate'
            example:
              name: support-backend
              allowed_models:
                - Kimi-K3
                - GLM-5.3-Flash
              max_parallel_requests: 20
              fallbacks:
                Kimi-K3:
                  - GLM-5.3-Flash
              retries:
                max_attempts: 2
                retry_delay_ms: 250
                backoff: exponential
                request_timeout_ms: 10000
      responses:
        '201':
          description: OK
          headers:
            X-Request-ID:
              schema:
                type: string
                format: uuid
              description: Server-generated correlation ID.
            Cache-Control:
              schema:
                const: no-store
              description: All API responses; application cache is server-side only.
            ETag:
              schema:
                type: string
                minLength: 1
                maxLength: 256
                pattern: ^[^\u0000-\u001f\u007f]+$
              description: Quoted integer resource version.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/TokenGroupResponse'
              example:
                data:
                  record_type: token_group
                  id: 11111111-1111-4111-8111-111111111111
                  created_at: '2026-10-07T08:00:00Z'
                  updated_at: '2026-10-07T08:00:00Z'
                  version: 1
                  name: support-backend
                  max_budget: null
                  budget_duration: 1d
                  tpm_limit: null
                  rpm_limit: null
                  allowed_models:
                    - Kimi-K3
                    - GLM-5.3-Flash
                  cache_ttl: null
                  provider_key_ids: []
                  guardrails: null
                  fallbacks: null
                  retries: null
                  blocked: false
                  spend: 0
                  reserved_spend: 0
                  budget_started_at: null
                  resets_at: null
                  max_parallel_requests: 20
        '400':
          $ref: '#/components/responses/GatewayError400'
        '401':
          $ref: '#/components/responses/GatewayError401'
        '403':
          $ref: '#/components/responses/GatewayError403'
        '404':
          $ref: '#/components/responses/GatewayError404'
        '409':
          $ref: '#/components/responses/GatewayError409'
        '412':
          $ref: '#/components/responses/GatewayError412'
        '428':
          $ref: '#/components/responses/GatewayError428'
        '429':
          $ref: '#/components/responses/GatewayError429'
        '502':
          $ref: '#/components/responses/GatewayError502'
        '503':
          $ref: '#/components/responses/GatewayError503'
        '504':
          $ref: '#/components/responses/GatewayError504'
      security:
        - telnyxApiKey: []
components:
  parameters:
    IdempotencyKey:
      name: Idempotency-Key
      in: header
      required: true
      schema:
        type: string
        minLength: 1
        maxLength: 128
        pattern: ^[A-Za-z0-9_-]+$
      description: >-
        24h, scoped by account+method+path. Same key/body returns same outcome;
        changed body 409; in-progress 409 with Retry-After. Token-create replay
        returns 200 metadata WITHOUT secret.
  schemas:
    TokenGroupCreate:
      type: object
      additionalProperties: false
      properties:
        name:
          type: string
          minLength: 1
          maxLength: 256
          pattern: ^[^\u0000-\u001f\u007f]+$
        max_budget:
          anyOf:
            - $ref: '#/components/schemas/Money'
            - type: 'null'
        budget_duration:
          $ref: '#/components/schemas/BudgetDuration'
        tpm_limit:
          anyOf:
            - type: integer
              minimum: 0
            - type: 'null'
        rpm_limit:
          anyOf:
            - type: integer
              minimum: 0
            - type: 'null'
        allowed_models:
          type: array
          items:
            type: string
            minLength: 1
            maxLength: 256
            pattern: ^[^\u0000-\u001f\u007f]+$
          uniqueItems: true
          maxItems: 1000
        max_parallel_requests:
          anyOf:
            - type: integer
              minimum: 1
              maximum: 1000
              description: >-
                Optional maximum concurrent reserved/dispatched cache misses
                across replicas. Null means no concurrency cap. Cache hits do
                not consume slots; unknown usage retains its budget hold without
                a slot.
            - type: 'null'
        cache_ttl:
          anyOf:
            - type: integer
              minimum: 1
              maximum: 86400
            - type: 'null'
        provider_key_ids:
          type: array
          items:
            type: string
            format: uuid
          uniqueItems: true
          maxItems: 1000
        guardrails:
          anyOf:
            - type: object
              additionalProperties: false
              properties:
                secrets:
                  type: object
                  additionalProperties: false
                  properties:
                    prompt:
                      type: string
                      enum:
                        - ignore
                        - flag
                        - block
                    response:
                      type: string
                      enum:
                        - ignore
                        - flag
                        - block
                dlp:
                  type: object
                  additionalProperties: false
                  properties:
                    profiles:
                      type: array
                      items:
                        type: string
                        enum:
                          - financial
                          - government_id
                          - contact
                      uniqueItems: true
                    prompt:
                      type: string
                      enum:
                        - ignore
                        - flag
                        - block
                    response:
                      type: string
                      enum:
                        - ignore
                        - flag
                        - block
                streaming:
                  type: string
                  enum:
                    - buffered
                    - passthrough
              description: >-
                Per-group guardrail policy. Missing actions default to ignore.
                profiles may be listed with both dlp actions ignore (staged,
                inactive) but a non-ignore dlp action requires at least one
                profile. streaming defaults to buffered when any response action
                is block and passthrough otherwise; passthrough with a response
                block is 400. Findings are codes and counts only; content is
                never stored. The stored form is echoed back canonicalised (all
                actions explicit, profiles sorted); {} is stored as an
                all-ignore policy, null disables.
            - type: 'null'
        fallbacks:
          anyOf:
            - type: object
              maxProperties: 1000
              additionalProperties:
                type: array
                items:
                  type: string
                  minLength: 1
                  maxLength: 256
                  pattern: ^[^\u0000-\u001f\u007f]+$
                minItems: 1
                maxItems: 3
                uniqueItems: true
              description: >-
                Per-group model fallbacks: primary alias -> ordered list of 1..3
                other aliases. Every key and listed alias must be in
                allowed_models (a PATCH removing a referenced alias is 400); a
                chain never lists its own key; wildcards are rejected. Chains
                are one level deep (a step's own chain is not expanded).
                Fallback is attempted only for managed routes, only before the
                first provider event is forwarded, and only to steps the key may
                use; each attempt is reserved and recorded. {} and null both
                disable fallbacks.
            - type: 'null'
        retries:
          anyOf:
            - type: object
              additionalProperties: false
              properties:
                max_attempts:
                  type: integer
                  minimum: 1
                  maximum: 5
                retry_delay_ms:
                  type: integer
                  minimum: 0
                  maximum: 5000
                  default: 250
                backoff:
                  type: string
                  enum:
                    - constant
                    - linear
                    - exponential
                  default: exponential
                request_timeout_ms:
                  anyOf:
                    - type: integer
                      minimum: 1000
                      maximum: 60000
                    - type: 'null'
              required:
                - max_attempts
              description: >-
                Per-group same-route retries. max_attempts counts the first try
                of each step (the primary and each fallback) and is the ceiling
                for the x-ltg-max-attempts request header. A step is retried
                only when its attempt definitely did no provider work
                (connection failure before dispatch, or 429/529) before the
                first provider event; an uncertain failure (timeout,
                408/500/502/503/504) moves to the next fallback step instead.
                max_attempts x (1 + the longest fallback chain) must not exceed
                8, else 400. The n-th retry waits retry_delay_ms (constant),
                retry_delay_ms*n (linear) or retry_delay_ms*2^(n-1)
                (exponential), at most 5 s and never past the request deadline,
                then jittered uniformly down to half that value so concurrent
                retries spread out; a provider Retry-After of at most 5 s is
                honoured and a longer one moves to the next step.
                request_timeout_ms applies to streaming requests only: it bounds
                each attempt's time to its first provider event and never
                applies after it. Non-stream requests keep the per-attempt hop
                and idle bounds (their headers arrive only with the finished
                completion). retry_delay_ms and backoff are floors for the
                request knobs. Each planned attempt is reserved (at most 8 per
                request); unused reservation is released at settlement. Managed
                routes only. The stored form is echoed back with defaults filled
                in; null disables retries.
            - type: 'null'
        blocked:
          type: boolean
      required:
        - name
        - allowed_models
    TokenGroupResponse:
      type: object
      additionalProperties: false
      properties:
        data:
          $ref: '#/components/schemas/TokenGroup'
      required:
        - data
    Money:
      type: number
      minimum: 0
      maximum: 1000000000
      multipleOf: 0.000001
      description: >-
        USD, decimal precision at most 6 places; parse exactly to integer
        micro-USD, never binary float accounting.
    BudgetDuration:
      type:
        - string
        - 'null'
      enum:
        - 1d
        - 7d
        - 30d
        - null
    TokenGroup:
      type: object
      additionalProperties: false
      properties:
        record_type:
          const: token_group
        id:
          type: string
          format: uuid
        created_at:
          type: string
          format: date-time
        updated_at:
          type: string
          format: date-time
        version:
          type: integer
          minimum: 1
        name:
          type: string
          minLength: 1
          maxLength: 256
          pattern: ^[^\u0000-\u001f\u007f]+$
        max_budget:
          anyOf:
            - $ref: '#/components/schemas/Money'
            - type: 'null'
        budget_duration:
          $ref: '#/components/schemas/BudgetDuration'
        tpm_limit:
          anyOf:
            - type: integer
              minimum: 0
            - type: 'null'
        rpm_limit:
          anyOf:
            - type: integer
              minimum: 0
            - type: 'null'
        allowed_models:
          type: array
          items:
            type: string
            minLength: 1
            maxLength: 256
            pattern: ^[^\u0000-\u001f\u007f]+$
          uniqueItems: true
          maxItems: 1000
        max_parallel_requests:
          anyOf:
            - type: integer
              minimum: 1
              maximum: 1000
              description: >-
                Optional maximum concurrent reserved/dispatched cache misses
                across replicas. Null means no concurrency cap. Cache hits do
                not consume slots; unknown usage retains its budget hold without
                a slot.
            - type: 'null'
        cache_ttl:
          anyOf:
            - type: integer
              minimum: 1
              maximum: 86400
            - type: 'null'
        provider_key_ids:
          type: array
          items:
            type: string
            format: uuid
          uniqueItems: true
          maxItems: 1000
        guardrails:
          anyOf:
            - type: object
              additionalProperties: false
              properties:
                secrets:
                  type: object
                  additionalProperties: false
                  properties:
                    prompt:
                      type: string
                      enum:
                        - ignore
                        - flag
                        - block
                    response:
                      type: string
                      enum:
                        - ignore
                        - flag
                        - block
                dlp:
                  type: object
                  additionalProperties: false
                  properties:
                    profiles:
                      type: array
                      items:
                        type: string
                        enum:
                          - financial
                          - government_id
                          - contact
                      uniqueItems: true
                    prompt:
                      type: string
                      enum:
                        - ignore
                        - flag
                        - block
                    response:
                      type: string
                      enum:
                        - ignore
                        - flag
                        - block
                streaming:
                  type: string
                  enum:
                    - buffered
                    - passthrough
              description: >-
                Per-group guardrail policy. Missing actions default to ignore.
                profiles may be listed with both dlp actions ignore (staged,
                inactive) but a non-ignore dlp action requires at least one
                profile. streaming defaults to buffered when any response action
                is block and passthrough otherwise; passthrough with a response
                block is 400. Findings are codes and counts only; content is
                never stored. The stored form is echoed back canonicalised (all
                actions explicit, profiles sorted); {} is stored as an
                all-ignore policy, null disables.
            - type: 'null'
        fallbacks:
          anyOf:
            - type: object
              maxProperties: 1000
              additionalProperties:
                type: array
                items:
                  type: string
                  minLength: 1
                  maxLength: 256
                  pattern: ^[^\u0000-\u001f\u007f]+$
                minItems: 1
                maxItems: 3
                uniqueItems: true
              description: >-
                Per-group model fallbacks: primary alias -> ordered list of 1..3
                other aliases. Every key and listed alias must be in
                allowed_models (a PATCH removing a referenced alias is 400); a
                chain never lists its own key; wildcards are rejected. Chains
                are one level deep (a step's own chain is not expanded).
                Fallback is attempted only for managed routes, only before the
                first provider event is forwarded, and only to steps the key may
                use; each attempt is reserved and recorded. {} and null both
                disable fallbacks.
            - type: 'null'
        retries:
          anyOf:
            - type: object
              additionalProperties: false
              properties:
                max_attempts:
                  type: integer
                  minimum: 1
                  maximum: 5
                retry_delay_ms:
                  type: integer
                  minimum: 0
                  maximum: 5000
                  default: 250
                backoff:
                  type: string
                  enum:
                    - constant
                    - linear
                    - exponential
                  default: exponential
                request_timeout_ms:
                  anyOf:
                    - type: integer
                      minimum: 1000
                      maximum: 60000
                    - type: 'null'
              required:
                - max_attempts
              description: >-
                Per-group same-route retries. max_attempts counts the first try
                of each step (the primary and each fallback) and is the ceiling
                for the x-ltg-max-attempts request header. A step is retried
                only when its attempt definitely did no provider work
                (connection failure before dispatch, or 429/529) before the
                first provider event; an uncertain failure (timeout,
                408/500/502/503/504) moves to the next fallback step instead.
                max_attempts x (1 + the longest fallback chain) must not exceed
                8, else 400. The n-th retry waits retry_delay_ms (constant),
                retry_delay_ms*n (linear) or retry_delay_ms*2^(n-1)
                (exponential), at most 5 s and never past the request deadline,
                then jittered uniformly down to half that value so concurrent
                retries spread out; a provider Retry-After of at most 5 s is
                honoured and a longer one moves to the next step.
                request_timeout_ms applies to streaming requests only: it bounds
                each attempt's time to its first provider event and never
                applies after it. Non-stream requests keep the per-attempt hop
                and idle bounds (their headers arrive only with the finished
                completion). retry_delay_ms and backoff are floors for the
                request knobs. Each planned attempt is reserved (at most 8 per
                request); unused reservation is released at settlement. Managed
                routes only. The stored form is echoed back with defaults filled
                in; null disables retries.
            - type: 'null'
        blocked:
          type: boolean
        spend:
          $ref: '#/components/schemas/Money'
        reserved_spend:
          $ref: '#/components/schemas/Money'
        budget_started_at:
          anyOf:
            - type: string
              format: date-time
            - type: 'null'
        resets_at:
          anyOf:
            - type: string
              format: date-time
            - type: 'null'
      required:
        - record_type
        - id
        - created_at
        - updated_at
        - version
        - name
        - max_budget
        - budget_duration
        - tpm_limit
        - rpm_limit
        - allowed_models
        - cache_ttl
        - provider_key_ids
        - guardrails
        - fallbacks
        - retries
        - blocked
        - spend
        - reserved_spend
        - budget_started_at
        - resets_at
    Error:
      type: object
      additionalProperties: false
      properties:
        errors:
          type: array
          items:
            $ref: '#/components/schemas/ErrorItem'
          minItems: 1
      required:
        - errors
    ErrorItem:
      type: object
      additionalProperties: false
      properties:
        code:
          type: string
          enum:
            - invalid_request
            - invalid_token_key
            - unauthorized
            - token_key_blocked
            - resource_blocked
            - budget_exceeded
            - end_user_budget_exceeded
            - model_not_in_catalog
            - rate_limit_exceeded
            - concurrency_limit_exceeded
            - not_found
            - conflict
            - idempotency_conflict
            - precondition_failed
            - precondition_required
            - enforcement_unavailable
            - upstream_error
            - upstream_timeout
            - limit_out_of_range
            - prompt_blocked
            - response_blocked
            - invalid_metadata
            - cursor_expired
        title:
          type: string
        detail:
          type: string
        meta:
          type: object
          additionalProperties: false
          properties:
            scope:
              $ref: '#/components/schemas/Scope'
            resets_at:
              anyOf:
                - type: string
                  format: date-time
                - type: 'null'
            request_id:
              type: string
              format: uuid
          required:
            - request_id
      required:
        - code
        - title
        - detail
        - meta
    Scope:
      type: string
      enum:
        - token_group
        - token_user
        - token_key
        - end_user
  responses:
    GatewayError400:
      description: >-
        Invalid fields or values (invalid_request), or token-key budget/rate
        limits outside their supported range (limit_out_of_range).
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: invalid_request
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError401:
      description: Wrong/missing credential for this plane.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
        WWW-Authenticate:
          schema:
            type: string
          description: >-
            Bearer challenge for this plane; no provider/master credential
            disclosure.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: unauthorized
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError403:
      description: >-
        Blocked/expired, exhausted budget or disallowed model; scope identifies
        exhausted level.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: unauthorized
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError404:
      description: Absent or foreign-account resource (indistinguishable).
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: not_found
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError409:
      description: Membership, idempotency or snapshot conflict.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
        Retry-After:
          schema:
            type: integer
            minimum: 1
          description: Seconds; retrying inference is not idempotent.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: idempotency_conflict
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError412:
      description: ETag mismatch.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: precondition_failed
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError428:
      description: If-Match required.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: precondition_required
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError429:
      description: Rate limited.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
        Retry-After:
          schema:
            type: integer
            minimum: 1
          description: Seconds; retrying inference is not idempotent.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: rate_limit_exceeded
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError502:
      description: >-
        upstream_error: every admitted provider attempt, including configured
        group retries and fallbacks, failed before the first provider event;
        streams return this real status instead of 200 plus an SSE error.
        Retrying is not idempotent.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
        Retry-After:
          schema:
            type: integer
            minimum: 1
          description: Seconds; retrying inference is not idempotent.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: invalid_request
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError503:
      description: Enforcement/secret/managed route unavailable; fail closed.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
        Retry-After:
          schema:
            type: integer
            minimum: 1
          description: Seconds; retrying inference is not idempotent.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: enforcement_unavailable
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
    GatewayError504:
      description: >-
        upstream_timeout: as 502 upstream_error, but the last admitted attempt
        timed out waiting for the provider (connect, read or write timeout, the
        attempt deadline, or x-ltg-request-timeout) before its first event.
        Retrying is not idempotent.
      headers:
        X-Request-ID:
          schema:
            type: string
            format: uuid
          description: Server-generated correlation ID.
        Cache-Control:
          schema:
            const: no-store
          description: All API responses; application cache is server-side only.
        Retry-After:
          schema:
            type: integer
            minimum: 1
          description: Seconds; retrying inference is not idempotent.
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
          example:
            errors:
              - code: invalid_request
                title: Request could not be completed
                detail: Check the request and the documented handling for this error.
                meta:
                  request_id: 11111111-1111-4111-8111-111111111111
  securitySchemes:
    telnyxApiKey:
      type: http
      scheme: bearer
      description: Management-only Telnyx account credential.

````