> ## Documentation Index
> Fetch the complete documentation index at: https://docs.costgraph.ai/llms.txt
> Use this file to discover all available pages before exploring further.

# List model-serving deployments

> Returns one row per model-serving deployment held by the tenant selected by the X-CostGraph-Tenant-ID header. Verdicts are recorded per replica, so each row carries the worst replica's verdict and how many replicas share it rather than an average across replicas. A deployment with no accelerator measurements is not returned.



## OpenAPI

````yaml /api-reference/costgraph/openapi.json get /api/v1/tenant/ai/deployments
openapi: 3.0.0
info:
  description: Read and manage your CostGraph organization, spend, alerts, and settings.
  title: CostGraph API
  contact: {}
  version: '1.0'
servers:
  - url: https://api.costgraph.ai
security: []
tags:
  - name: ai
    x-group: AI
  - name: ai-serving
    x-group: AI serving
  - name: anomalies
    x-group: Anomalies
  - name: auth
    x-group: Auth
  - name: billing
    x-group: Billing
  - name: billing-export
    x-group: Billing export
  - name: budgets
    x-group: Budgets
  - name: ci
    x-group: CI
  - name: compute-recommendations
    x-group: Compute recommendations
  - name: config
    x-group: Config
  - name: cost
    x-group: Cost
  - name: gpus
    x-group: GPUs
  - name: graphai
    x-group: Graph AI
  - name: infracost
    x-group: Infracost
  - name: integrations
    x-group: Integrations
  - name: invitations
    x-group: Invitations
  - name: kubernetes-clusters
    x-group: Kubernetes clusters
  - name: marketplace
    x-group: Marketplace
  - name: network-requests
    x-group: Network requests
  - name: notifications
    x-group: Notifications
  - name: oauth
    x-group: OAuth
  - name: oauth-clients
    x-group: OAuth clients
  - name: opencost
    x-group: OpenCost
  - name: organization
    x-group: Audit log
  - name: organizations
    x-group: Organizations
  - name: placement-alternatives
    x-group: Placement alternatives
  - name: reports
    x-group: Reports
  - name: service-map
    x-group: Service map
  - name: settings
    x-group: Settings
  - name: sso
    x-group: Single sign-on
  - name: tenants
    x-group: Tenants
  - name: user
    x-group: Users
  - name: virtual-machines
    x-group: Virtual machines
  - name: virtual-tags
    x-group: Virtual tags
  - name: workflows
    x-group: Workflows
paths:
  /api/v1/tenant/ai/deployments:
    get:
      tags:
        - ai
      summary: List model-serving deployments
      description: >-
        Returns one row per model-serving deployment held by the tenant selected
        by the X-CostGraph-Tenant-ID header. Verdicts are recorded per replica,
        so each row carries the worst replica's verdict and how many replicas
        share it rather than an average across replicas. A deployment with no
        accelerator measurements is not returned.
      parameters:
        - description: Tenant ID
          name: X-CostGraph-Tenant-ID
          in: header
          required: true
          schema:
            type: string
        - description: Filter to one served model
          name: model
          in: query
          schema:
            type: string
        - description: Filter to one gateway
          name: gateway
          in: query
          schema:
            type: string
        - description: Filter to one namespace
          name: namespace
          in: query
          schema:
            type: string
        - description: Filter to one cluster
          name: cluster_id
          in: query
          schema:
            type: string
        - description: Filter on the worst replica's utilisation band
          name: utilization_band
          in: query
          schema:
            type: string
        - description: Filter on the worst replica's verdict
          name: action
          in: query
          schema:
            type: string
        - description: Filter on the worst replica's urgency
          name: urgency
          in: query
          schema:
            type: string
        - description: Maximum rows to return
          name: limit
          in: query
          schema:
            type: integer
        - description: Rows to skip
          name: offset
          in: query
          schema:
            type: integer
      responses:
        '200':
          description: OK
          content:
            application/json:
              schema:
                allOf:
                  - $ref: '#/components/schemas/responses.SuccessResponse'
                  - type: object
                    properties:
                      data:
                        type: array
                        items:
                          $ref: '#/components/schemas/db.AIDeploymentListRow'
        '400':
          description: Bad Request
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/responses.ErrorResponse'
        '401':
          description: Unauthorized
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/responses.ErrorResponse'
        '403':
          description: Forbidden
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/responses.ErrorResponse'
        '500':
          description: Internal Server Error
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/responses.ErrorResponse'
      security:
        - BearerAuth: []
components:
  schemas:
    responses.SuccessResponse:
      type: object
      required:
        - message
        - status
      properties:
        data: {}
        message:
          type: string
          example: some message
        status:
          type: string
          example: success
    db.AIDeploymentListRow:
      type: object
      properties:
        action:
          type: string
        cards_current:
          type: number
        cards_recommended:
          type: number
        cluster_id:
          type: string
        cluster_name:
          type: string
        compute_cost_per_hour:
          type: number
        container_name:
          type: string
        cost_per_mtok:
          type: number
        data_age_hours:
          type: number
        deployment_name:
          type: string
        gateway:
          type: string
        gpu_memory_p95:
          type: number
        gpu_utilization_p95_worst:
          type: number
        kind:
          type: string
        model:
          type: string
        namespace:
          type: string
        reasons:
          type: array
          items:
            type: string
        replica_count:
          type: integer
        replicas_at_worst:
          type: integer
        replicas_measured:
          type: integer
        resource_id:
          type: string
        savings_per_month:
          type: number
        urgency:
          type: string
        utilization_band:
          type: string
        workload_report_id:
          type: string
        worst_replica_action:
          type: string
        worst_replica_band:
          type: string
        worst_replica_container_id:
          type: string
        worst_replica_urgency:
          type: string
    responses.ErrorResponse:
      type: object
      required:
        - message
        - status
      properties:
        message:
          type: string
          example: some message
        status:
          type: string
          example: error
  securitySchemes:
    BearerAuth:
      description: Enter "Bearer {token}"
      type: apiKey
      name: Authorization
      in: header

````

This documentation is built and hosted on [Mintlify](https://mintlify.com), a developer documentation platform.