diff --git a/aigw/api-reference/admin-api/authentication.mdx b/aigw/api-reference/admin-api/authentication.mdx new file mode 100644 index 00000000..da987901 --- /dev/null +++ b/aigw/api-reference/admin-api/authentication.mdx @@ -0,0 +1,244 @@ +--- +title: "Authentication" +description: "Authorise Admin API requests with a Strata Cloud Manager access token issued to a service account" +--- + +Admin API requests are authorised with a short-lived access token issued by Strata Cloud Manager, not with an AI Gateway API key. + + +A gateway API key, the key you send as `Authorization: Bearer $API_KEY` on inference requests, is **not** accepted on any Admin API endpoint. See [Inference API authentication](/aigw/api-reference/inference-api/authentication) for the inference path. + + +You obtain an Admin API token by authenticating a **service account** against the Palo Alto Networks authentication service. The token carries the ID of the tenant service group (TSG) it was scoped to, and every request made with it is routed to that tenant. + +```mermaid +sequenceDiagram + participant SCM as Strata Cloud Manager + participant App as Your job + participant Auth as Auth service + participant API as Admin API + + SCM-->>App: Client ID, Secret, TSG ID + App->>Auth: POST /oauth2/access_token + Auth-->>App: access_token, expires_in 900 + + loop Within 15 minutes + App->>API: Authorization: Bearer + API-->>App: One tenant's resources + end +``` + +## What you need + +Before you can request a token, three values must exist. All three come from Strata Cloud Manager. + +| Value | What it identifies | Where it comes from | +|:--|:--|:--| +| TSG ID | The tenant the token will act on. One TSG corresponds to one AI Gateway organisation. | Shown against the tenant in Strata Cloud Manager | +| Client ID | The service account | Issued when the service account is created | +| Client Secret | The service account's credential | Shown **once**, at creation | + + +A TSG must have a service account before you can make any API call against it. If a tenant has no service account of its own, a service account belonging to one of its ancestor TSGs can be used instead. See [Token scope within a TSG hierarchy](#token-scope-within-a-tsg-hierarchy). + +One TSG may have many service accounts, and one service account may have many tokens. + + + +*Tenant service group* and *tenant* are used interchangeably; there is no functional difference between them. + + +## Create a service account + +Service accounts are created in Strata Cloud Manager, through Common Services Identity & Access. A service account is not tied to a specific user. + + + + Sign in to [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) and go to **System Settings > Identity & Access**. + + {/* TODO(screenshot): SCM System Settings > Identity & Access */} + + + Choose the tenant the service account belongs to. + + A service account added to a parent tenant is automatically added to all of that tenant's children, which is how a parent manages its children. Add it to a child tenant instead if you do not want that inheritance. + + Creating service accounts in different tenant service groups lets you assign different roles for different access permissions, and keeps the audit trail readable. + + {/* TODO(screenshot): tenant selector */} + + + Select **Add** (or **Add Identity**), then set **Identity Type** to **Service Account**. + + Give it a unique and meaningful **Service Account Name**. Optionally add a **Service Account Contact** email and a **Description**. The contact person is not added as a user. + + {/* TODO(screenshot): Add Identity dialog with Identity Type set to Service Account */} + + + Select **Next**. The Client Credentials screen shows the **Client ID** and **Client Secret**. + + + The Client Secret is presented once. Copy both values, or select **Download CSV File**, before leaving the screen. If you lose the secret you must issue a new credential. + + + {/* TODO(screenshot): Client Credentials screen with Download CSV File */} + + + Select **Next**. The display name of the service account is formatted as `@.iam.panserviceaccount.com`. + + Every service account of a parent tenant carries the parent TSG ID, and every service account of a child tenant carries the child TSG ID. Take note of the `tsg_id`, because you pass it on every token request. + + {/* TODO(screenshot): display name showing the tsg_id */} + + + On the Assign Roles screen, select the scope (for example **All Apps & Services**) and assign the role the service account needs. A service account with no role assignment cannot obtain a token. + + Grant the narrowest role that covers the Admin API operations you intend to automate. A custom role needs `iam.service_account` and `iam.custom_role` permissions if the account will manage identities itself. + + {/* TODO(screenshot): Assign Roles screen */} + + + Save to create the service account. You now have the Client ID, Client Secret and TSG ID needed to request a token. + + + +## Request an access token + +Exchange the service account credentials for an access token with `POST /oauth2/access_token`. + + +The authentication service runs on a different FQDN from the rest of Strata Cloud Manager: +`https://auth.apps.paloaltonetworks.com` + + +The endpoint uses basic auth, with the Client ID as the username and the Client Secret as the password, and takes the TSG ID in the `scope` field: + +```sh Request an access token +curl -d "grant_type=client_credentials&scope=tsg_id:" \ + -u : \ + -H "Content-Type: application/x-www-form-urlencoded" \ + -X POST https://auth.apps.paloaltonetworks.com/oauth2/access_token +``` + +The service account you authenticate with must belong to the TSG named on `scope`, or to one of its ancestors. + +A successful response carries the token and its lifetime in seconds: + +```json +{ + "access_token": "eyJhbGciOi...", + "token_type": "Bearer", + "expires_in": 900 +} +``` + + +**Access tokens have a lifespan of 15 minutes.** Read `expires_in` rather than hard-coding the number, cache the token in memory for the life of that window, and refresh it about a minute before it lapses. Requesting a fresh token per API call is unnecessary; carrying one across a long-running job is what fails. + + +## Call an Admin API endpoint + +Send the token as a bearer token. Every Admin API endpoint takes the same header; the base URL and path for each one are shown on its own reference page. + +```sh +curl https://api.apps.paloaltonetworks.com/ai_gw/v2/configs \ + -H "Authorization: Bearer $ACCESS_TOKEN" \ + -H "Content-Type: application/json" +``` + +You do not pass the TSG ID on the request. The token already contains it, and the request is routed to that tenant on the strength of it. + +## Token scope within a TSG hierarchy + +A token issued for one TSG cannot be used against another. If you have a tenant, Tenant 1A, with a service account named `1a_svc`, then a token obtained through `1a_svc` reaches Tenant 1A and nothing else. + +When you run multiple tenants you organise them as a hierarchy of TSGs. Creating a dedicated service account for every TSG and tenant in that hierarchy is the simplest arrangement, but it is not necessary. **A service account belonging to a TSG can name any descendant of that TSG when it requests a token.** + +Consider a root TSG A with two tenants, and a child TSG B with two more: + +```mermaid +graph TD + TSGA["TSG A
a_svc"] --> T1A[Tenant 1A] + TSGA --> T2A[Tenant 2A] + TSGA --> TSGB["TSG B
b_svc"] + TSGB --> T1B[Tenant 1B] + TSGB --> T2B[Tenant 2B] +``` + +Assume `a_svc` and `b_svc` were created with the `superuser` role on TSG A and TSG B respectively. Then: + +- **`a_svc` can request a token for any TSG ID in the hierarchy**, because every TSG and tenant shown is a descendant of TSG A. +- **`b_svc` can request tokens for TSG B, Tenant 1B and Tenant 2B**, its own descendants. +- **`b_svc` cannot request a token for TSG A, Tenant 1A or Tenant 2A.** Those are its ancestor and its peers. +- Tenants 1A, 2A, 1B and 2B hold no service accounts of their own, so only the service accounts of their parent TSGs can obtain tokens for them. + + +The TSG IDs used in the examples on this page are deliberately fake. Real TSG IDs are 10-digit integers, such as `1000000001`. + + +### Grant cross-hierarchy access with an access policy + +`b_svc` cannot obtain a token for Tenant 1A, because Tenant 1A sits outside its subtree. Where you need exactly that, create an **access policy** on Tenant 1A naming the Client ID of `b_svc` as the principal. The policy overrides the hierarchy restriction for that one pairing. + +You can do this from the multitenant UI, or with the Identity and Access Management *create an access policy* API. The following grants `b_svc` superuser permissions on Tenant 1A, represented here by TSG ID `18`: + +```sh Grant b_svc access to Tenant 1A +curl -d '{"role":"superuser","resource":"prn:18::::","principal":"b_svc@15.iam.panserviceaccount.com"}' \ + -H "Authorization: Bearer $ACCESS_TOKEN" \ + -H "Content-Type: application/json" \ + -X POST https://api.strata.paloaltonetworks.com/iam/v1/access_policies +``` + +The token you authenticate this call with must itself be scoped to the TSG that owns the resource, `15` in the example above. + +The same endpoint grants a person a role on a tenant, with their email address as the principal. Three fields make up a policy: + +| Field | Format | +|:--|:--| +| `principal` | A user's email address, or a service account as `@.iam.panserviceaccount.com` | +| `resource` | `prn:::::`. Leave the last segment empty for the TSG root, which Strata Cloud Manager labels **All Apps & Services** | +| `role` | A built-in role is a bare slug, such as `superuser` or `view_only_admin`. A role defined by a tenant carries that tenant's ID, as `my_custom_role:1000000001` | + +`GET /iam/v1/access_policies` lists the policies on the tenant your token is scoped to, and `DELETE /iam/v1/access_policies/{id}` removes one. A duplicate `POST` returns `409`, so the call is safe to retry. + + +The access policy endpoint does not check that the principal exists. A `POST` naming an address that belongs to nobody returns `201` and a policy ID, and the policy sits in the listing doing nothing until somebody with that address appears. Validate addresses against your directory before creating policies in bulk, and review the IDs you get back. + + + +Older Palo Alto Networks documentation shows this endpoint as `https://api.sase.paloaltonetworks.com/access_policies`. `https://api.strata.paloaltonetworks.com/iam/v1/access_policies` is the current host and path. + + +## Check your token + +If the credentials baked into a token are not the ones you expect, the Admin API rejects the request and the error reports an invalid authorisation code. + +Paste the token into [jwt.io](https://jwt.io/) to decode it and read the claims back. The decoded payload shows the `tsg_id` the token was issued for, which is the fastest way to confirm you are hitting the tenant you think you are, and an `access` claim listing the roles the service account holds on it. + +{/* TODO(screenshot): jwt.io showing an encoded token beside its decoded payload, tsg_id highlighted */} + +### Common failures + +| Symptom | Cause | +|:--|:--| +| Token request rejected at `/oauth2/access_token` | The Client ID or Client Secret is wrong, or the service account has no role assignment | +| Token request rejected for the TSG on `scope` | The service account is not in that TSG or an ancestor of it. Create an [access policy](#grant-cross-hierarchy-access-with-an-access-policy) | +| Admin API returns an authorisation error with a valid-looking token | The token has expired, since they last 15 minutes, or it is scoped to a different tenant. Decode it and check `tsg_id` | +| Admin API rejects a key that works for inference | A gateway API key was sent instead of an access token. Only tokens from the authentication service are accepted | + +## Related + + + + What the Admin API manages, and the permissions model behind it + + + Gateway API keys and JWT authentication for inference requests + + + Admin API error codes and what they mean + + + Every administrative action, attributed to the principal that made it + + diff --git a/aigw/api-reference/admin-api/error.mdx b/aigw/api-reference/admin-api/error.mdx index 4c7c89a6..a1509e82 100644 --- a/aigw/api-reference/admin-api/error.mdx +++ b/aigw/api-reference/admin-api/error.mdx @@ -1,45 +1,64 @@ --- title: Errors -description: Error codes returned by the Admin API and how to resolve them. +description: Every error code the Admin API returns, and what to do about it. --- -# Admin API Error Codes +Admin API failures carry a code in the `AB` series alongside the HTTP status. The code is the precise reason; the status is the class. -Below is a list of error codes returned by Prisma AIRS AI Gateway Admin API. These help with debugging failed requests and ensuring proper authentication, permissions, and request formatting. +## Error reference -## Error Reference +| Code | Status | Message | Type | +|:--|:--|:--|:--| +| `AB01` | 400 | Request Validation Error | Client Error | +| `AB02` | 404 | Request Validation Error | Client Error | +| `AB03` | 403 | User not allowed to access the resource | Client Error | +| `AB04` | 500 | Internal Server Error | Server Error | +| `AB05` | 401 | Unauthorized access | Client Error | +| `AB06` | 429 | Rate limit exceeded | Client Error | +| `AB07` | 409 | Resource already exists | Client Error | +| `AB08` | 404 | Resource not found | Client Error | +| `AB09` | 402 | Subscription exhausted | Client Error | -| Error Code | HTTP Status | Message | Type | -|------------|-------------|----------------------------------------|--------------| -| AB01 | 400 | Request Validation Error | Client Error | -| AB02 | 404 | Request Validation Error | Client Error | -| AB03 | 403 | User not allowed to access the resource | Client Error | -| AB04 | 500 | Internal Server Error | Server Error | -| AB05 | 401 | Unauthorized access | Client Error | -| AB06 | 429 | Rate limit exceeded | Client Error | -| AB07 | 409 | Resource already exists | Client Error | -| AB08 | 404 | Resource not found | Client Error | -| AB09 | 402 | Subscription exhausted | Client Error | +## The four you will actually hit ---- + + + The request never got as far as being authorised. In order of likelihood: -## Notes on Common Errors - -**AB01 – Request Validation Error (400)** + 1. **A gateway API key was sent instead of an access token.** The Admin API does not accept the key your applications use for inference. Obtain a [Strata Cloud Manager access token](/aigw/api-reference/admin-api/authentication) instead. + 2. **The token expired.** Access tokens last 15 minutes. Request a fresh one. + 3. **The token is malformed**, truncated in an environment variable or carrying a stray newline. -This error usually happens when: -- You're either missing required parameters -- using incorrect data types (e.g., sending a number instead of a string) -- passing values that are not within the allowed set (enum violations). + Paste the token into [jwt.io](https://jwt.io/) to confirm what it actually contains. + + + The token is valid, but it does not reach this resource. - + - **Wrong tenant.** The token's `tsg_id` names a different tenant from the one that owns the resource. A token cannot cross a TSG boundary unless an [access policy](/aigw/api-reference/admin-api/authentication#grant-cross-hierarchy-access-with-an-access-policy) grants it. + - **Insufficient role.** The service account that obtained the token holds a role that does not cover this operation. + - **Wrong workspace.** The resource belongs to a workspace, and the `workspace_id` you passed points somewhere else. + + + The body did not match what the endpoint expects. Usually one of: - -**AB05 – Unauthorized Access (401)** + - a required parameter is missing + - a value has the wrong type, a number where a string belongs + - a value sits outside the allowed set for an enum -This indicates that the user is not authorized to access the resource, often due to incorrect API key permissions. + The endpoint's reference page lists every field and its type. + + + The path is right and the resource is not there. Two causes dominate: -**Common Cause**: Your API key does **not have the right permissions** for this request. + - **A slug was passed where an ID was expected**, or the reverse. Configs, integrations and providers take slugs; guardrails, MCP servers, policies and API keys take IDs. + - **The resource belongs to a workspace you did not name.** Pass `workspace_id` as a query parameter on `GET`, and in the body on `POST` and `PUT`. + + -**Fix**: Go to [stratacloudmanager.paloaltonetworks.com](https://stratacloudmanager.paloaltonetworks.com/) → **API Keys**, and check the permissioning for your key. Ensure it includes the required scopes for the endpoint you're calling. + +`AB07` on a create means the resource already exists. Most creates are not idempotent, so retrying a request that timed out can produce a `409` rather than a duplicate. Check before retrying blindly. + + + Most Admin API errors are authentication errors. Start here. + diff --git a/aigw/api-reference/admin-api/introduction.mdx b/aigw/api-reference/admin-api/introduction.mdx index 5d489e27..915240e2 100644 --- a/aigw/api-reference/admin-api/introduction.mdx +++ b/aigw/api-reference/admin-api/introduction.mdx @@ -1,199 +1,192 @@ --- title: "Introduction" -description: "Manage your Prisma AIRS AI Gateway organisation and workspaces programmatically" +description: "Everything Strata Cloud Manager configures, available as an API" --- -# AI Gateway Admin API +The Admin API manages the resources that make up your AI Gateway organisation: the integrations that connect model providers, the credentials behind them, the configs and guardrails applied to workspaces, the policies that cap usage, the MCP servers you expose, and the analytics you report on. -The AI Gateway Admin API provides programmatic access to manage your organisation, workspaces, and resources. Whether you're automating routine administration tasks, integrating the AI Gateway with your existing systems, or customizing your deployment at scale, this API gives you the tools to control every aspect of your AI Gateway implementation. +Anything an administrator sets up in Strata Cloud Manager can be set up here instead. -## Understanding the Admin API Ecosystem +```mermaid +graph TD + SA["Service account
Client ID + Client Secret"] + AUTH["Authentication service
auth.apps.paloaltonetworks.com"] + ADMIN["Admin API
api.apps.paloaltonetworks.com"] + ORG["Organisation resources
integrations, credentials,
deployments"] + WS["Workspace resources
configs, providers,
guardrails"] + + SA -->|"scope=tsg_id:TSG_ID"| AUTH + AUTH -->|"access token
valid 15 minutes"| ADMIN + ADMIN --> ORG + ADMIN --> WS +``` -The Admin API is organized around key capabilities that let you manage different aspects of your AI Gateway environment. Let's explore what you can build and automate: +Created in Strata Cloud Manager, the service account is the identity behind every administrative call. It exchanges its credentials for a short-lived token, the token names one tenant, and every request made with it lands on that tenant. -### Resource Management + +Admin API requests are authorised with a **Strata Cloud Manager access token**, not a gateway API key. A key that works for inference returns `401` here. See [Authentication](/aigw/api-reference/admin-api/authentication). + -At the foundation of the AI Gateway are the resources that define how your AI implementation works. These can all be managed programmatically: +## Your first request - - - Create and manage configuration profiles that define routing rules, model settings, and more. - - - Manage AI providers and credentials across workspaces. Replaces Virtual Keys. - - - Create and manage API keys for accessing AI Gateway services. - - + + + In Strata Cloud Manager, go to **System Settings > Identity & Access**, select your tenant, and add an identity of type **Service Account**. Save the Client ID and Client Secret it issues, and note the tenant service group ID (TSG ID). -### Analytics and Monitoring + Full walkthrough with roles and inheritance: [Create a service account](/aigw/api-reference/admin-api/authentication#create-a-service-account). + + + Exchange those three values for a token. -Once your resources are configured, you'll want to measure performance and usage. The Admin API gives you powerful tools to access analytics data: + ```sh + curl -d "grant_type=client_credentials&scope=tsg_id:" \ + -u : \ + -H "Content-Type: application/x-www-form-urlencoded" \ + -X POST https://auth.apps.paloaltonetworks.com/oauth2/access_token + ``` + + + Send the token as a bearer token. Every endpoint takes the same header. - - - Retrieve aggregated usage statistics and performance metrics. - - - Access detailed analytics organized by metadata, model, or user. - - - Monitor performance trends, costs, errors, feedback, and usage patterns over time. - - + ```sh + curl https://api.apps.paloaltonetworks.com/ai_gw/v2/configs \ + -H "Authorization: Bearer $ACCESS_TOKEN" + ``` -## Authentication Strategy + The response lists the configs belonging to the organisation the token is scoped to. You do not send a tenant identifier on the request, because the token carries it. + + + Tokens last **15 minutes**, reported as `expires_in: 900` on the token response. Cache one for that window and refresh it shortly before it lapses. + + -Now that you understand what the Admin API can do, let's explore how to authenticate your requests. The AI Gateway uses a sophisticated access control system with two types of API keys, each designed for different use cases: +## Base URLs - - - **Organisation-wide access** +Which base URL an endpoint uses follows the resource, not the sidebar group it sits in. - These keys grant access to administrative operations across your entire organisation. +| Base URL | Serves | +|:--|:--| +| `https://api.apps.paloaltonetworks.com/ai_gw/v2` | Configs, workspace guardrails, providers, API keys, usage and rate limit policies, MCP servers, analytics, feedback | +| `https://api.apps.paloaltonetworks.com/ai_gw/admin/v2` | Integrations, MCP integrations, secret references, deployments, organisation guardrails | +| `https://aigw.portkey.ai/v1` | `POST /logs` and `GET /logs/{logId}`. Self-hosted deployments substitute their own host | - Only Organisation Owners and Admins can create and manage Admin API keys. - - - **Workspace-specific access** +The organisation guardrail endpoints carry their prefix in the path instead of the base URL, as `/admin/v2/guardrails` against `https://api.apps.paloaltonetworks.com/ai_gw`. That resolves to the same place as the admin base above. - These keys provide targeted access to resources within a single workspace. +Each endpoint's reference page states its server. Where the two disagree, trust the reference page, which is generated from the specification. - Workspace Managers can create and manage Workspace API keys. - - +## How resources are scoped -The key you use determines which operations you can perform. For organisation-wide administrative tasks, you'll need an Admin API key. For workspace-specific operations, you can use a Workspace API key. +Every resource sits at one of two levels. Knowing which one you are addressing explains most `403` responses. -## Access Control and Permissions Model +```mermaid +graph LR + T["Access token
one tsg_id"] -The AI Gateway's hierarchical access control system governs who can use which APIs. Let's examine how roles, API keys, and permissions interact: + T --> OL["Organisation level
/ai_gw/admin/v2"] + T --> WL["Workspace level
/ai_gw/v2
plus workspace_id"] -```mermaid -graph TD - A[Organization] --> B[Owner] - A --> C[Org Admin] - A --> D[Workspaces] - B --> E[Admin API Key] - C --> E - D --> F[Workspace Manager] - D --> G[Workspace Member] - F --> H[Workspace API Key] - E --> I[Organization-wide Operations] - H --> J[Workspace-specific Operations] - - classDef entity fill:#4a5568,stroke:#ffffff,color:#ffffff - classDef roles fill:#805ad5,stroke:#ffffff,color:#ffffff - classDef keys fill:#3182ce,stroke:#ffffff,color:#ffffff - classDef operations fill:#38a169,stroke:#ffffff,color:#ffffff - - class A,D entity - class B,C,F,G roles - class E,H keys - class I,J operations + OL --> O["Integrations
MCP integrations
Secret references
Deployments
Organisation guardrails"] + WL --> W["Configs
Providers
Workspace guardrails
Analytics"] ``` -This access model follows a clear hierarchy: +| Level | Addressed by | Holds | +|:--|:--|:--| +| **Organisation** | The TSG ID inside your token. One TSG is one organisation. | Integrations, MCP integrations, secret references, deployments, organisation guardrails | +| **Workspace** | A `workspace_id`, passed explicitly | Configs, providers, workspace guardrails, analytics | -| Role | Can Create Admin API Key | Can Create Workspace API Key | Access Scope | -|:------|:--------------------------|:------------------------------|:--------------| -| Organisation Owner | ✅ | ✅ (any workspace) | All organisation resources | -| Organisation Admin | ✅ | ✅ (any workspace) | All organisation resources | -| Workspace Manager | ❌ | ✅ (managed workspace only) | Single workspace resources | -| Workspace Member | ❌ | ❌ | Limited workspace access | +The split matches the base URLs: organisation-level resources are served from the `admin/v2` base. API keys and limit policies span both levels, and [`POST /api-keys/{sub-type}`](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) takes the level from the calling token rather than from the path — `sub-type` chooses `user` or `service`, nothing more. -## Creating and Managing API Keys +To reach a workspace-level resource, pass `workspace_id` as a **query parameter** on `GET` and list requests, and in the **request body** on `POST` and `PUT`. A workspace UUID or a slug both work. -Now that you understand the permission model, let's look at how to create the API keys you'll need: - -### Through Strata Cloud Manager +```sh +curl "https://api.apps.paloaltonetworks.com/ai_gw/v2/configs?workspace_id=WORKSPACE_SLUG" \ + -H "Authorization: Bearer $ACCESS_TOKEN" +``` -The simplest way to create an API key is through Strata Cloud Manager: +Omit it and the call resolves against the organisation default. -### Through the API +### Slugs and IDs -You can also create keys programmatically: +Mixing these up is a common cause of `404`. - -```sh Creating Admin API Key {1,2} -curl -X POST https://aigw.portkey.ai/v1/api-keys/organisation/service - -H "Authorization: Bearer YOUR_EXISTING_ADMIN_KEY" \ - -H "Content-Type: application/json" \ - -d '{ - "name":"API_KEY_NAME_0809", - "scopes":[ - "logs.export", - "logs.list", - "logs.view" - ] - }' -``` +- **Slugs** identify the resources you name yourself: `/configs/{slug}`, `/integrations/{slug}`, `/providers/{slug}` +- **IDs** identify the resources the gateway names for you: `/guardrails/{guardrailId}`, `/mcp-servers/{mcpServerId}`, `/policies/usage-limits/{policyUsageLimitsId}`, `/api-keys/{id}` -```sh Creating Workspace API Key {1,2} -curl -X POST https://aigw.portkey.ai/v1/api-keys/workspace/user \ - -H "Authorization: Bearer YOUR_EXISTING_WORKSPACE_KEY" \ - -H "Content-Type: application/json" \ - -d '{ - "name":"API_KEY_NAME_0909", - "workspace_id":"WORKSPACE_ID", - "scopes":[ - "virtual_keys.create", - "virtual_keys.update", - ] - }' -``` - +### Guardrails sit at both levels -## Understanding API Key Capabilities +[`/guardrails`](/aigw/api-reference/guardrails/list-guardrails) manages the guardrails of a single workspace. [`/admin/v2/guardrails`](/aigw/api-reference/org-guardrails/list-org-guardrails) manages the organisation-wide ones, which apply to every workspace unless a workspace is explicitly excluded. See [Enforcing Org Level Guardrails](/aigw/product/administration/enforce-organisation-level-guardrails). -Both key types have different capabilities. This table clarifies which operations each key type can perform: +## Common tasks -| Operation | Admin API Key | Workspace API Key | -|:-----------|:---------------|:-------------------| -| Manage organisation settings | ✅ | ❌ | -| Create/manage workspaces | ✅ | ❌ | -| Manage users and permissions | ✅ | ❌ | -| Create/manage configs | ✅ (All workspaces) | ✅ (Single workspace) | -| Create/manage providers | ✅ (All workspaces) | ✅ (Single workspace) | -| Access Analytics | ✅ (All workspaces) | ✅ (Single workspace) | -| Create/update feedback | ❌ | ✅ | +Each of these is a sequence of calls, not a single endpoint. -## Security and Compliance: Audit Logs + + + 1. [`POST /integrations`](/aigw/api-reference/integrations/post-integrations) creates the integration and attaches the provider credential + 2. [`PUT /integrations/{slug}/models`](/aigw/api-reference/integrations/models/put-integrations-by-slug-models) chooses which models it exposes + 3. [`PUT /integrations/{slug}/workspaces`](/aigw/api-reference/integrations/workspaces/put-integrations-by-slug-workspaces) grants the workspaces that may use it -For security-conscious organisations, the AI Gateway provides comprehensive audit logging of all Admin API operations. These logs give you complete visibility into administrative actions: + Steps 2 and 3 are [model provisioning](/aigw/product/model-catalog/model-provisioning) and [workspace provisioning](/aigw/product/model-catalog/workspace-provisioning). Skip them and the integration exists but nobody can reach it. + + + 1. [`POST /mcp-servers`](/aigw/api-reference/mcp-servers/mcp-servers-create) registers the server + 2. [`GET /mcp-servers/{mcpServerId}/capabilities`](/aigw/api-reference/mcp-servers/capabilities/mcp-server-capabilities-list) reads the tools it advertises + 3. [`PUT /mcp-servers/{mcpServerId}/capabilities`](/aigw/api-reference/mcp-servers/capabilities/mcp-server-capabilities-bulk-update) enables only the tools you intend to expose + 4. [`PUT /mcp-servers/{mcpServerId}/user-access`](/aigw/api-reference/mcp-servers/user-access/mcp-server-user-access-bulk-update) decides who may call it + 5. [`POST /mcp-servers/{mcpServerId}/test`](/aigw/api-reference/mcp-servers/mcp-servers-test) confirms the connection before anyone depends on it -Every administrative action is recorded with: -- User identity -- Action type and target resource -- Timestamp -- IP address -- Request details + See the [MCP Gateway](/aigw/product/mcp-gateway) documentation for what each capability means. + + + 1. [`POST /policies/usage-limits`](/aigw/api-reference/usage-limit-policies/create-usage-limits-policy) defines the budget + 2. [`GET /policies/usage-limits/{policyUsageLimitsId}/entities`](/aigw/api-reference/usage-limit-policies/list-usage-limits-policy-entities) shows what the policy currently binds to + 3. [`PUT /policies/usage-limits/{policyUsageLimitsId}/entities/{entityId}/reset`](/aigw/api-reference/usage-limit-policies/reset-usage-limits-policy-entity) clears consumption for one entity -This audit trail helps maintain compliance and provides accountability for all administrative changes. + Rate limits work the same way under [`/policies/rate-limits`](/aigw/api-reference/rate-limit-policies/list-rate-limits-policies). Background in [Budget Limits](/aigw/product/policies/budget-limits) and [Rate Limits](/aigw/product/policies/rate-limits). + + + 1. [`POST /api-keys/{sub-type}`](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) creates the key with the scopes it needs, `user` or `service` + 2. [`POST /api-keys/{id}/rotate`](/aigw/api-reference/api-keys/post-api-keys-by-id-rotate) rotates it on a schedule or on suspicion - - Learn more about the AI Gateway's audit logging capabilities - + Available scopes are listed in [API Keys (AuthN and AuthZ)](/aigw/product/enterprise-offering/org-management/api-keys-authn-and-authz). + + + - [`GET /analytics/graphs/cost`](/aigw/api-reference/analytics/graphs/get-analytics-graphs-cost) returns spend over time + - [`GET /analytics/groups/ai-models`](/aigw/api-reference/analytics/groups/get-analytics-groups-ai-models) breaks it down by model + - [`GET /analytics/groups/metadata/{metadataKey}`](/aigw/api-reference/analytics/groups/get-analytics-groups-metadata-by-metadata-key) breaks it down by whatever your requests tag themselves with -## Getting Started with the Admin API + Tagging requests with [metadata](/aigw/product/observability/metadata) is what makes the last one useful. + + -Now that you understand the Admin API ecosystem, authentication, and permissions model, you're ready to start making requests. Here's what you'll need: +## When a call fails -1. **Appropriate role**: Ensure you have the right permissions (Org Owner/Admin for Admin API, Workspace Manager for Workspace API) -2. **API key**: Generate the appropriate key from Strata Cloud Manager -3. **Make your first request**: Use your key in the request header +| Status | Read it as | +|:--|:--| +| `401` | The token is expired, malformed, or is actually a gateway API key | +| `403` | The token is valid but scoped to a different tenant, or the service account's role does not cover this operation | +| `404` | A slug was passed where an ID was expected, or the resource belongs to a workspace you did not name | +| `409` | The resource already exists. Most creates are not idempotent | -For developers looking to integrate with the Admin API, we provide a complete OpenAPI specification that you can use with your API development tools: +Every code the Admin API returns is listed on the [Errors](/aigw/api-reference/admin-api/error) page. - - A single specification covers both the gateway and the Admin API - +## Audit -## Need Support? +Every administrative call is recorded with the principal that made it, the action, the target resource, a timestamp, an IP address and the request details. Automation is attributable to the service account that ran it, which is a good reason to give each one a name that says what it is for. -If you need help setting up or using the Admin API, our team is ready to assist: +## Next - - Schedule time with our team to get personalized help with the Admin API - + + + Service accounts, access tokens, and how scope works across a TSG hierarchy + + + Every error code, with the usual cause + + + The trail of every administrative change + + + The same settings, configured from Strata Cloud Manager + + diff --git a/aigw/api-reference/inference-api/authentication.mdx b/aigw/api-reference/inference-api/authentication.mdx index aac67bc6..88930f90 100644 --- a/aigw/api-reference/inference-api/authentication.mdx +++ b/aigw/api-reference/inference-api/authentication.mdx @@ -14,6 +14,10 @@ Based on your access level, you might see the relevant permissions on the API ke You can also authenticate the AI Gateway using JWT Tokens. Learn more here + +This key authenticates inference requests only. The Admin API does not accept it. Those requests are authorised with a Strata Cloud Manager access token issued to a service account. See [Admin API Authentication](/aigw/api-reference/admin-api/authentication). + + ## Using Your API Key ### REST API diff --git a/aigw/api-reference/inference-api/headers.mdx b/aigw/api-reference/inference-api/headers.mdx index 897960d3..50e8d0be 100644 --- a/aigw/api-reference/inference-api/headers.mdx +++ b/aigw/api-reference/inference-api/headers.mdx @@ -241,7 +241,7 @@ curl https://aigw.portkey.ai/v1/chat/completions \ ### Fetch Integrated Models -Applies to `GET /v1/models` only. When `true`, forces the endpoint to return the AI Gateway's integrated (Model Catalog) models even when a provider or config is passed on the request. Without this header, any provider signal causes the endpoint to proxy to the upstream provider's `/v1/models`. You can also set this as `fetch_integrated_models: true` inside an AI Gateway config. ([Docs](/api-reference/inference-api/models/models)) +Applies to `GET /v1/models` only. When `true`, forces the endpoint to return the AI Gateway's integrated (Model Catalog) models even when a provider or config is passed on the request. Without this header, any provider signal causes the endpoint to proxy to the upstream provider's `/v1/models`. You can also set this as `fetch_integrated_models: true` inside an AI Gateway config. ([Docs](/aigw/api-reference/models/list-models)) @@ -311,7 +311,7 @@ List of header names whose values the AI Gateway should mask (hash) in request, `x-portkey-sensitive-headers` only controls log masking. It does **not** forward headers to the upstream provider. -If you need both behaviors, use: +If you need both behaviours, use: - `x-portkey-forward-headers` to forward the header upstream - `x-portkey-sensitive-headers` to mask that header's value in AI Gateway logs diff --git a/aigw/api-reference/inference-api/introduction.mdx b/aigw/api-reference/inference-api/introduction.mdx index 7d171a1a..f7a5c4dd 100644 --- a/aigw/api-reference/inference-api/introduction.mdx +++ b/aigw/api-reference/inference-api/introduction.mdx @@ -1,44 +1,96 @@ --- title: "Introduction" -description: "This documentation provides detailed information about the various ways you can access and interact with Prisma AIRS AI Gateway - **a robust AI gateway** designed to simplify and enhance your experience with Large Language Models (LLMs) like OpenAI's GPT models. " +description: "One OpenAI-compatible endpoint in front of every model you have connected" --- -Whether you're integrating directly with OpenAI, using a framework like Langchain or LlamaIndex, or building standalone applications, the AI Gateway offers a flexible, secure, and efficient way to manage and deploy AI-powered features. +The Gateway API runs inference. The [Admin API](/aigw/api-reference/admin-api/introduction) configures what it is allowed to do. -## 2 Ways to Integrate +Point an OpenAI-compatible client at the gateway, name a model, and the request is routed, guarded, logged and billed according to the configuration your administrators have already put in place. Nothing about routing, credentials or policy belongs in the request itself. -The AI Gateway can be accessed through two primary methods, each catering to different use cases and integration requirements: +```sh +curl https://aigw.portkey.ai/v1/chat/completions \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer $API_KEY" \ + -d '{ + "model": "@openai-prod/gpt-4o", + "messages": [ + { "role": "user", "content": "Hello!" } + ] + }' +``` -### 1\. OpenAI SDK through the AI Gateway +Two things carry the routing: the `Authorization` header holds your gateway API key, and the `model` field is prefixed with the slug of the integration to use. `@openai-prod/gpt-4o` reads as *the `gpt-4o` model, through the integration called `openai-prod`*. -**Ideal for:** Python, Node.js, or any other application that already uses an OpenAI-compatible client. By changing the base URL and adding gateway-specific headers, you can integrate the AI Gateway's features into your existing setup. This is also how frameworks like Langchain and LlamaIndex connect to the gateway. + +The base URL is `https://aigw.portkey.ai/v1`. MCP Gateway serves `/m` and Agent Gateway serves `/agent`. + -#### Installing the SDK +## Two ways in -Choose the SDK that matches your development environment: + + + Change the base URL and the API key on a client you already have. This is also how Langchain, LlamaIndex and most agent frameworks connect. + + + Call the endpoints directly. Everything the SDKs do is available over plain HTTP. + + - - -```sh -npm install openai -``` +### Through an OpenAI SDK - - -```sh -pip install openai +Install the official client, then override two fields: + + +```python Python +from openai import OpenAI + +client = OpenAI( + api_key="PORTKEY_API_KEY", + base_url="https://aigw.portkey.ai/v1", +) + +response = client.chat.completions.create( + model="@openai-prod/gpt-4o", + messages=[{"role": "user", "content": "Say this is a test"}], +) ``` - - -#### Usage +```js NodeJS +import OpenAI from 'openai'; -Point the client at `https://aigw.portkey.ai/v1`, authenticate with your AI Gateway API key, and prefix the model name with your provider slug. +const client = new OpenAI({ + apiKey: "PORTKEY_API_KEY", + baseURL: "https://aigw.portkey.ai/v1", +}); -Learn more [here](/aigw/integrations/llms/openai). +const response = await client.chat.completions.create({ + model: '@openai-prod/gpt-4o', + messages: [{ role: 'user', content: 'Say this is a test' }], +}); +``` + -### 2\. REST API +Gateway-specific behaviour (a saved config, request metadata, a cache directive) travels in headers, which the SDKs expose as `default_headers` / `defaultHeaders`. -**Ideal for:** applications that prefer RESTful services. The base URL for all REST API requests is `https://aigw.portkey.ai/v1`, with an [authentication](/aigw/api-reference/inference-api/authentication) header. +## Start here -Learn more [here](/api-reference/inference-api/chat). + + + API keys, and JWT as an alternative + + + Every gateway header, and what each one changes + + + Routing, fallbacks, retries and caching as a single object + + + What can sit behind the gateway + + + What comes back, and where the gateway adds to it + + + Failures on the inference path + + diff --git a/aigw/changelog/enterprise.mdx b/aigw/changelog/enterprise.mdx index 6d097cc1..705462c6 100644 --- a/aigw/changelog/enterprise.mdx +++ b/aigw/changelog/enterprise.mdx @@ -43,7 +43,7 @@ Azure AI Foundry additionally accepts `service_tier` as a request parameter, whi - **Pricing**: Pricing adjustments can now be set per model on an integration, not just for the integration as a whole. A model-level adjustment fully replaces the integration-level one rather than merging with it. [Pricing Adjustments](/aigw/product/model-catalog/pricing-adjustments) - **Pricing**: Added bundled pricing data for ElevenLabs, so text-to-speech and speech-to-text costs are tracked without a pricing sync. [Model Catalog](/aigw/product/model-catalog) -- **Responses API**: Streaming responses now report reasoning and cached token counts in `usage`, which previously always returned `0`. [Responses API Reference](/api-reference/inference-api/responses/responses) +- **Responses API**: Streaming responses now report reasoning and cached token counts in `usage`, which previously always returned `0`. [Responses API Reference](/aigw/api-reference/responses/create-response) - **Guardrails**: Checks that are turned off inside a redaction or transformation guardrail — such as PII redaction — are now correctly skipped. Previously they still ran. Guardrail configs referencing an unknown check ID are also now ignored with a warning instead of being passed through. [Guardrails](/aigw/product/guardrails) - **MCP Gateway**: Fixed a 500 error when loading the OAuth consent screen at `GET /oauth/authorize`. Applies to Prisma AIRS AI Gateway deployments, where the consent screen is served as HTML. [MCP OAuth](/aigw/product/mcp-gateway/authentication/oauth) - **MCP Gateway**: A repeated OAuth callback no longer shows an `invalid_state` error when the connection has already succeeded. [MCP External OAuth](/aigw/product/mcp-gateway/authentication/external-oauth) @@ -184,7 +184,7 @@ OpenAI-compatible integrations can now authenticate to the upstream provider usi The `/v1/ocr` endpoint now supports Mistral OCR models served on Google Vertex AI, with pricing tracked per page processed. -[OCR API Reference](/api-reference/inference-api/ocr) · [Vertex AI Documentation](/aigw/integrations/llms/vertex-ai) +[OCR API Reference](/aigw/api-reference/ocr/create-ocr) · [Vertex AI Documentation](/aigw/integrations/llms/vertex-ai) ### Qwen: Responses & Messages APIs @@ -257,7 +257,7 @@ Per-request AWS session tags can now be passed through to Bedrock via STS, letti New `POST /v1/ocr` endpoint brings OCR requests under the gateway's retries, fallbacks, load balancing, caching, logging, and pricing. Currently supports Mistral AI and Azure AI Foundry. -[OCR API Reference](/api-reference/inference-api/ocr) · [Mistral AI Documentation](/aigw/integrations/llms/mistral-ai#ocr-document-processing) · [Azure AI Foundry Documentation](/aigw/integrations/llms/azure-foundry#ocr-document-processing) +[OCR API Reference](/aigw/api-reference/ocr/create-ocr) · [Mistral AI Documentation](/aigw/integrations/llms/mistral-ai#ocr-document-processing) · [Azure AI Foundry Documentation](/aigw/integrations/llms/azure-foundry#ocr-document-processing) ### Singulr Guardrail @@ -279,7 +279,7 @@ Guardrail checks can now be configured to forward specific client request header ### Provider Updates -- **OpenAI & Azure OpenAI**: `prompt_cache_options` parameter can now be passed through to control prompt caching behavior +- **OpenAI & Azure OpenAI**: `prompt_cache_options` parameter can now be passed through to control prompt caching behaviour - **Azure AI Foundry**: Input items sent via the Responses API now get an explicit `type: message` field, fixing rejected requests from clients (e.g. OpenCode) that omit it - **Bedrock (Mantle)**: OpenAI models now support the native Messages API, gated behind the `use-responses-api-2026-07-30` beta header - **Bedrock (Mantle)**: `propertyNames` is stripped from tool schemas for Gemma models to prevent validation errors @@ -306,7 +306,7 @@ Databricks provider now supports the Responses API, enabling OpenAI-compatible R ### Anthropic Extended Beta Parameters -New Anthropic beta parameters can now be passed through to the Anthropic provider: `inference_geo`, `diagnostics`, `fallbacks`, and `context_management`. These parameters enable geo-routing, diagnostic telemetry, fallback behavior, and managed-context workflows for Claude models. +New Anthropic beta parameters can now be passed through to the Anthropic provider: `inference_geo`, `diagnostics`, `fallbacks`, and `context_management`. These parameters enable geo-routing, diagnostic telemetry, fallback behaviour, and managed-context workflows for Claude models. [Anthropic Documentation](/aigw/integrations/llms/anthropic) @@ -342,7 +342,7 @@ New beta flag `use-responses-api-2026-07-30` enables the latest Responses API sc Anthropic Messages API requests can now be routed to models that use the OpenAI Responses API — such as OpenAI's codex models — with request and response transformed automatically in both directions. -[Responses API Documentation](/api-reference/inference-api/responses/responses) +[Responses API Documentation](/aigw/api-reference/responses/create-response) ### Lightning AI Provider @@ -417,7 +417,7 @@ New `scan_scope` and `strip_scaffolding` parameters for the Palo Alto Networks P `GET /v1/models` now routes through the gateway's provider-backed model listing handler. The endpoint calls the upstream provider, applies response transforms and returns the normalized model list — replacing the previous pass-through delegation. -[Models API Documentation](/api-reference/inference-api/models/models) +[Models API Documentation](/aigw/api-reference/models/list-models) ### Guardrails: Soft Deny (HTTP 200) @@ -546,7 +546,7 @@ Vertex AI integration now supports the `moonshotai` model provider. - **Anthropic**: Batch results download URLs validated before fetching - **AWS Batches**: Bucket names validated before use on batch routes -- **Metrics endpoint**: The `/dataservice/metrics` endpoint is now gated behind the `ENABLE_PROMETHEUS` env flag, matching the `/metrics` endpoint behavior +- **Metrics endpoint**: The `/dataservice/metrics` endpoint is now gated behind the `ENABLE_PROMETHEUS` env flag, matching the `/metrics` endpoint behaviour - **Inline-block**: `x-portkey-virtual-key` header is allowed when inline configs are blocked - **Security**: Strengthened request validation across inline-config blocking, batch routes, and provider auth - Updated dependencies to patch security vulnerabilities @@ -560,9 +560,9 @@ Vertex AI integration now supports the `moonshotai` model provider. ### AI Gateway Models Endpoint Override -A new `x-portkey-fetch-integrated-models` header and `fetch_integrated_models` config field force `GET /v1/models` to return gateway-configured models even when provider, virtual-key, or config routing signals are present. Without the flag, behavior is unchanged — provider signals still proxy `/v1/models` to the upstream provider. +A new `x-portkey-fetch-integrated-models` header and `fetch_integrated_models` config field force `GET /v1/models` to return gateway-configured models even when provider, virtual-key, or config routing signals are present. Without the flag, behaviour is unchanged — provider signals still proxy `/v1/models` to the upstream provider. -[Models API Documentation](/api-reference/inference-api/models/models) +[Models API Documentation](/aigw/api-reference/models/list-models) ### MCP Gateway: Metrics-Only Logging @@ -649,7 +649,7 @@ Anthropic chat completions now pass `context_management` through to the upstream ### Anthropic Skills: File Download Fix -`GET /v1/files/{file_id}` and `GET /v1/files/{file_id}/content` now route correctly for Anthropic, and binary responses (e.g., PPTX generated via Skills + Code Execution) pass through untouched. Previously these requests fell back to list-route behavior and corrupted binary downloads. +`GET /v1/files/{file_id}` and `GET /v1/files/{file_id}/content` now route correctly for Anthropic, and binary responses (e.g., PPTX generated via Skills + Code Execution) pass through untouched. Previously these requests fell back to list-route behaviour and corrupted binary downloads. [Anthropic Files Documentation](/aigw/integrations/llms/anthropic/files) @@ -926,7 +926,7 @@ Deployment hardening for **FIPS** environments (DHI-related paths) improves comp - **Messages API streaming**: Fixed **content block** tracking in the stream adapter so multi-block assistant output is logged and forwarded consistently. - **Security**: Enhanced authorization checks for internal service communication. - **Data plane / custom host**: Enforces a configurable **maximum response size** from custom-host upstreams to protect the gateway from oversized payloads. -- **Redis**: **Auth refresh** fixes for long-lived connections; token rotation and reconnect behavior are more reliable. +- **Redis**: **Auth refresh** fixes for long-lived connections; token rotation and reconnect behaviour are more reliable. @@ -964,7 +964,7 @@ New `bedrock-mantle` provider for AWS's OpenAI-compatible Bedrock inference engi `/v1/messages` requests now pass through Anthropic's top-level `speed` parameter (`fast` / `standard`). OpenAI's `service_tier` → Anthropic `speed` mapping continues to work for `/chat/completions`. -[Anthropic `service_tier` → `speed` mapping](/aigw/integrations/llms/anthropic#service_tier--anthropic-speed-mapping) +[Anthropic `service_tier` → `speed` mapping](/aigw/integrations/llms/anthropic#service-tier) ### Centralized `anthropic-beta` Header Filtering @@ -974,7 +974,7 @@ New `bedrock-mantle` provider for AWS's OpenAI-compatible Bedrock inference engi The [Required Metadata Key-Value Pairs](/aigw/product/guardrails/list-of-guardrail-checks#access-control-management) guardrail now accepts a `matchType` parameter to control how values are compared: -| `matchType` | Behavior | +| `matchType` | Behaviour | |---|---| | `exact` (default) | Metadata value must equal the expected value | | `contains` | Metadata value must contain at least one expected value | @@ -1111,7 +1111,7 @@ Custom models in the [Model Catalog](/aigw/product/model-catalog/custom-models) ### Anthropic Enhancements - **Data URL file inputs**: `/chat/completions` accepts base64 data URLs (`data:;base64,`) in `file.file_data` and forwards them as Anthropic document content -- **`cache_control` passthrough**: `/chat/completions` now forwards a top-level `cache_control` parameter to Anthropic, matching `/messages` behavior. For block-level caching, continue using [prompt caching](/aigw/integrations/llms/anthropic/prompt-caching) +- **`cache_control` passthrough**: `/chat/completions` now forwards a top-level `cache_control` parameter to Anthropic, matching `/messages` behaviour. For block-level caching, continue using [prompt caching](/aigw/integrations/llms/anthropic/prompt-caching) - **Strict structured outputs**: Object-type `response_format` JSON schemas automatically get `additionalProperties: false` ### MCP Gateway Reliability @@ -1445,7 +1445,7 @@ Prometheus histogram buckets and metric labels are now configurable to control c - Default bucket counts reduced for lower cardinality while preserving observability ### Provider Updates -- **Perplexity**: Added as a native [Responses API](/api-reference/inference-api/responses/responses) provider +- **Perplexity**: Added as a native [Responses API](/aigw/api-reference/responses/create-response) provider - **OpenRouter**: Added embeddings endpoint support - **ZhipuAI (Z.ai)**: Added cost attribution for chat and image generation models @@ -1507,7 +1507,7 @@ Together AI now supports reasoning/thinking models with `reasoning_effort` param [Together AI Documentation](/aigw/integrations/llms/together-ai#reasoning--thinking-support) ### Fixes and Improvements -- **Unified Messages & Responses API**: Fixed issues when a request config had a combination of native and non-native providers (e.g., load balancing across Anthropic and OpenAI). The adapter decision is now made per-provider, ensuring correct behavior for mixed configs. +- **Unified Messages & Responses API**: Fixed issues when a request config had a combination of native and non-native providers (e.g., load balancing across Anthropic and OpenAI). The adapter decision is now made per-provider, ensuring correct behaviour for mixed configs. - **Responses API**: Groq, OpenRouter, and xAI now use the native `/responses` endpoint supported by each provider for Responses API requests - **Anthropic**: Fixed `max_tokens` handling in chat completions — `max_tokens` now takes precedence over `max_completion_tokens` when both are provided - Updated model pricing configurations for Bedrock @@ -1592,7 +1592,7 @@ The OpenAI Responses API (`/v1/responses`) now works with **all 70+ providers**, **Note:** Responses API-only features like `previous_response_id` state management and built-in tools (`web_search`, `file_search`, `computer_use`) are only supported on OpenAI and Azure OpenAI. -[Responses API Documentation](/api-reference/inference-api/responses/responses) +[Responses API Documentation](/aigw/api-reference/responses/create-response) ### Provider Updates - **Vertex AI**: Added option to skip cost attribution for Provisioned Throughput (PTU) deployments. Configure via: @@ -1870,7 +1870,7 @@ Added new condition keys for fine-grained budget and rate limit policies: - [Documentation](/aigw/product/ai-gateway/load-balancing#sticky-load-balancing) ### Provider Updates -- **Gemini/Vertex AI**: Added `reasoning_effort` parameter support for controlling thinking behavior. Maps OpenAI's `reasoning_effort` (`minimal`/`low`/`medium`/`high`) to Gemini's `thinkingLevel` (`low`/`high`). +- **Gemini/Vertex AI**: Added `reasoning_effort` parameter support for controlling thinking behaviour. Maps OpenAI's `reasoning_effort` (`minimal`/`low`/`medium`/`high`) to Gemini's `thinkingLevel` (`low`/`high`). - [Documentation](/aigw/integrations/llms/gemini#using-reasoning_effort-parameter) - **Azure OpenAI**: Added support for v1 preview API version for Azure OpenAI endpoints - **Azure OpenAI**: Added pricing support for batch `/responses` endpoint with deployment @@ -2005,7 +2005,7 @@ Added new condition keys for fine-grained budget and rate limit policies: ### Usage and Rate Limit Policy - Introduced usage limits and rate limit policies, which allow organisations to apply flexible budget and rate limit controls based on dynamic conditions (API keys, metadata, workspace, etc.). -- More details: [Documentation](/aigw/product/enterprise-offering/budget-policies) and [API Reference](/api-reference/admin-api/control-plane/policies) +- More details: [Documentation](/aigw/product/enterprise-offering/budget-policies) and [API Reference](/aigw/api-reference/usage-limit-policies/list-usage-limits-policies) ### Logging Enhancements - Added support for OpenTelemetry W3C trace context headers (`traceparent` and `baggage`) to enable integration with distributed tracing systems. @@ -2423,7 +2423,7 @@ Added new condition keys for fine-grained budget and rate limit policies: ### Unified Models Endpoint - Released unified models API which follows OpenAI API specification to list all available models that can be used through the AI Gateway -- [Documentation Link](/api-reference/inference-api/models/models) +- [Documentation Link](/aigw/api-reference/models/list-models) @@ -2737,7 +2737,7 @@ Added new condition keys for fine-grained budget and rate limit policies: ### AWS Bedrock Prompt Caching - Added support for AWS Bedrock prompt caching. -- [Docs Link](/aigw/integrations/llms/bedrock/prompt-caching#prompt-caching-on-bedrock) +- [Docs Link](/aigw/integrations/llms/bedrock/prompt-caching#how-bedrock-prompt-caching-works) ### VertexAI Gemini 2.5 Thinking Param Support - Added support for the thinking settings parameters for VertexAI. @@ -2950,8 +2950,8 @@ We are redacting this release and will be releasing a patch with out Workspace B - Updated the unified API signature for Extended thinking which was introduced in v1.10.12 to ensure that OpenAI compliant field of the response remain untouched regardless of strict_open_ai_compliance flag. - More Details: - [Anthropic](/aigw/integrations/llms/anthropic#extended-thinking-reasoning-models) - - [AWS Bedrock](/aigw/integrations/llms/bedrock/aws-bedrock#extended-thinking-reasoning-models) - - [VertexAI](/aigw/integrations/llms/vertex-ai#extended-thinking-reasoning-models) + - [AWS Bedrock](/aigw/integrations/llms/bedrock/aws-bedrock#extended-thinking-reasoning-models-beta) + - [VertexAI](/aigw/integrations/llms/vertex-ai#extended-thinking-reasoning-models-beta) ### Unified Batches API Improvements - ```custom_id``` will be preserved in the VertexAI batch output. @@ -3002,8 +3002,8 @@ We are redacting this release and will be releasing a patch with out Workspace B - Introduced a unified API signature to support single-turn and multi-turn conversations with Anthropic Extended Reasoning across Anthropic, AWS Bedrock and VertexAI. - More Details: - [Anthropic](/aigw/integrations/llms/anthropic#extended-thinking-reasoning-models) - - [AWS Bedrock](/aigw/integrations/llms/bedrock/aws-bedrock#extended-thinking-reasoning-models) - - [VertexAI](/aigw/integrations/llms/vertex-ai#extended-thinking-reasoning-models) + - [AWS Bedrock](/aigw/integrations/llms/bedrock/aws-bedrock#extended-thinking-reasoning-models-beta) + - [VertexAI](/aigw/integrations/llms/vertex-ai#extended-thinking-reasoning-models-beta) ### Prometheus Metric Updates - Added a new metric (```llm_last_byte_diff_duration_milliseconds```) to track LLM last byte latency for chunked JSON responses. diff --git a/aigw/help-center/mcp-gateway-troubleshooting.mdx b/aigw/help-center/mcp-gateway-troubleshooting.mdx index 15562f5b..7c6644f5 100644 --- a/aigw/help-center/mcp-gateway-troubleshooting.mdx +++ b/aigw/help-center/mcp-gateway-troubleshooting.mdx @@ -19,7 +19,7 @@ MCP Gateway authenticates in two places: your client → Prisma AIRS AI Gateway DCR errors, wrong scope, redirect issues - Updated config still shows old behavior + Updated config still shows old behaviour Gateway URL, redirect URI, base URL @@ -142,7 +142,7 @@ You updated an MCP server's configuration (URL, auth, scopes), but the gateway s **Fix:** 1. Wait a short while for the change to propagate, then retry. 2. Disconnect and reconnect the server from your client, and re-run the sign-in so a fresh connection is created. -3. If it still uses the old behavior, confirm you edited the integration in the **same workspace/organisation** your client is connecting through. +3. If it still uses the old behaviour, confirm you edited the integration in the **same workspace/organisation** your client is connecting through. Creating the server again under a brand-new slug always picks up the latest config—useful as a quick confirmation that the issue was a stale connection. Prefer reconnecting first; a new slug also changes your client URL. diff --git a/aigw/help-center/you-do-not-have-enough-permissions.mdx b/aigw/help-center/you-do-not-have-enough-permissions.mdx index 9f59c439..54d6b194 100644 --- a/aigw/help-center/you-do-not-have-enough-permissions.mdx +++ b/aigw/help-center/you-do-not-have-enough-permissions.mdx @@ -84,17 +84,17 @@ Before fixing the error, understand the two key types and how scopes work. -Admin API keys access workspace-level resources (providers, configs, prompts, integrations) by passing `workspace_id` as a query parameter (for `GET`/list requests) or in the request body (for `POST`/`PUT` requests). Use a workspace UUID or slug. Workspace API keys are locked to their own workspace and can't access other workspaces. +Admin API requests reach workspace-level resources (providers, configs, prompts, integrations) by passing `workspace_id` as a query parameter (for `GET`/list requests) or in the request body (for `POST`/`PUT` requests). Use a workspace UUID or slug. Workspace API keys are locked to their own workspace and can't access other workspaces. -```sh Listing configs (GET — query parameter) -curl "https://aigw.portkey.ai/v1/configs?workspace_id=ws-abc123" \ - -H "Authorization: Bearer YOUR_ADMIN_KEY" +```sh Listing configs (GET, query parameter) +curl "https://api.apps.paloaltonetworks.com/ai_gw/v2/configs?workspace_id=ws-abc123" \ + -H "Authorization: Bearer $ACCESS_TOKEN" ``` -```sh Creating a provider (POST — request body) -curl -X POST https://aigw.portkey.ai/v1/providers \ - -H "Authorization: Bearer YOUR_ADMIN_KEY" \ +```sh Creating a provider (POST, request body) +curl -X POST "https://api.apps.paloaltonetworks.com/ai_gw/v2/providers" \ + -H "Authorization: Bearer $ACCESS_TOKEN" \ -H "Content-Type: application/json" \ -d '{"workspace_id": "ws-abc123", "name": "My Provider"}' ``` @@ -102,7 +102,7 @@ curl -X POST https://aigw.portkey.ai/v1/providers \ -Data-plane APIs (`/v1/chat/completions`, `/v1/responses`, etc.) require **Workspace API keys**. Org-level operations (audit logs, user management) require **Admin API keys**. +Inference APIs (`/v1/chat/completions`, `/v1/responses`, etc.) require an **API key**. Every Admin API operation, including the ones above, requires a **Strata Cloud Manager access token** instead. An API key returns `401`. See [Admin API Authentication](/aigw/api-reference/admin-api/authentication). ### How permission scopes work diff --git a/aigw/integrations/agents/agno-ai.mdx b/aigw/integrations/agents/agno-ai.mdx index d990b019..d7a83152 100644 --- a/aigw/integrations/agents/agno-ai.mdx +++ b/aigw/integrations/agents/agno-ai.mdx @@ -13,7 +13,7 @@ The AI Gateway transforms your Agno agents into production-ready systems by prov - **Access to 3,000+ LLMs** through a unified interface - **Built-in reliability** with fallbacks, retries, and load balancing - **Cost tracking and optimization** across all agent operations -- **Advanced guardrails** for safe and compliant agent behavior +- **Advanced guardrails** for safe and compliant agent behaviour - **Enterprise governance** with budget controls and access management @@ -654,11 +654,11 @@ Create User-specific API keys that automatically: Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) -- [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) +- [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using Python SDK: -For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). +For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/autogen.mdx b/aigw/integrations/agents/autogen.mdx index 47da5796..8b7cce60 100644 --- a/aigw/integrations/agents/autogen.mdx +++ b/aigw/integrations/agents/autogen.mdx @@ -493,11 +493,11 @@ Create User-specific API keys that automatically: Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) -- [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) +- [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using Python SDK: -For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). +For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/bring-your-own-agents.mdx b/aigw/integrations/agents/bring-your-own-agents.mdx index ea72b0a6..46f6ff2c 100644 --- a/aigw/integrations/agents/bring-your-own-agents.mdx +++ b/aigw/integrations/agents/bring-your-own-agents.mdx @@ -123,7 +123,7 @@ llm2 = ChatOpenAI( ### 4\. [Logs](/aigw/product/observability/logs) -Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behavior, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. +Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behaviour, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. The AI Gateway offers comprehensive logging features that capture detailed information about every action and decision made by your AI agents. Access a dedicated section to view records of agent executions, including parameters, outcomes, function calls, and errors. Filter logs based on multiple parameters such as trace ID, model, tokens used, and metadata. diff --git a/aigw/integrations/agents/control-flow.mdx b/aigw/integrations/agents/control-flow.mdx index 5ba9fbb5..eca2a59e 100644 --- a/aigw/integrations/agents/control-flow.mdx +++ b/aigw/integrations/agents/control-flow.mdx @@ -119,7 +119,7 @@ llm2 = ChatOpenAI( ### 4\. [Logs](/aigw/product/observability/logs) -Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behavior, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. +Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behaviour, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. The AI Gateway offers comprehensive logging features that capture detailed information about every action and decision made by your AI agents. Access a dedicated section to view records of agent executions, including parameters, outcomes, function calls, and errors. Filter logs based on multiple parameters such as trace ID, model, tokens used, and metadata. diff --git a/aigw/integrations/agents/crewai.mdx b/aigw/integrations/agents/crewai.mdx index 524797b7..3f926e5f 100644 --- a/aigw/integrations/agents/crewai.mdx +++ b/aigw/integrations/agents/crewai.mdx @@ -13,7 +13,7 @@ The AI Gateway enhances CrewAI with production-readiness features, turning your - **Built-in reliability** with fallbacks, retries, and load balancing - **Cost tracking and optimization** to manage your AI spend - **Access to 3,000+ LLMs** through a single integration -- **Guardrails** to keep agent behavior safe and compliant +- **Guardrails** to keep agent behaviour safe and compliant - **Version-controlled prompts** for consistent agent performance @@ -426,11 +426,11 @@ Here's a basic configuration to route requests to OpenAI, specifically using GPT Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) - - [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) + - [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using Python SDK: - For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). + For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/langchain-agents.mdx b/aigw/integrations/agents/langchain-agents.mdx index a5c01219..55ca9578 100644 --- a/aigw/integrations/agents/langchain-agents.mdx +++ b/aigw/integrations/agents/langchain-agents.mdx @@ -120,7 +120,7 @@ llm2 = ChatOpenAI( ### 4\. [Logs](/aigw/product/observability/logs) -Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behavior, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. +Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behaviour, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. The AI Gateway offers comprehensive logging features that capture detailed information about every action and decision made by your AI agents. Access a dedicated section to view records of agent executions, including parameters, outcomes, function calls, and errors. Filter logs based on multiple parameters such as trace ID, model, tokens used, and metadata. @@ -149,7 +149,7 @@ With the AI Gateway tracing, you can encapsulate the complete execution of your ### 6\. Guardrails -LLMs are brittle - not just in API uptimes or their inexplicable `400`/`500` errors, but also in their core behavior. You can get a response with a `200` status code that completely errors out for your app's pipeline due to mismatched output. With the AI Gateway's Guardrails, we now help you enforce LLM behavior in real-time with our _Guardrails on the Gateway_ pattern. +LLMs are brittle - not just in API uptimes or their inexplicable `400`/`500` errors, but also in their core behaviour. You can get a response with a `200` status code that completely errors out for your app's pipeline due to mismatched output. With the AI Gateway's Guardrails, we now help you enforce LLM behaviour in real-time with our _Guardrails on the Gateway_ pattern. Using the AI Gateway's Guardrail platform, you can now verify your LLM inputs AND outputs to be adhering to your specifed checks; and since Guardrails are built into the gateway itself, you can orchestrate your request exactly the way you want - with actions ranging from _denying the request_, _logging the guardrail result_, _creating an evals dataset_, _falling back to another LLM or prompt_, _retrying the request_, and more. diff --git a/aigw/integrations/agents/langgraph.mdx b/aigw/integrations/agents/langgraph.mdx index 0fe6425c..d93fd5d8 100644 --- a/aigw/integrations/agents/langgraph.mdx +++ b/aigw/integrations/agents/langgraph.mdx @@ -13,7 +13,7 @@ The AI Gateway enhances LangGraph with production-readiness features, turning yo - **Built-in reliability** with fallbacks, retries, and load balancing - **Cost tracking and optimization** to manage your AI spend - **Access to 3,000+ LLMs** through a single integration -- **Guardrails** to keep agent behavior safe and compliant +- **Guardrails** to keep agent behaviour safe and compliant - **Version-controlled prompts** for consistent agent performance @@ -680,11 +680,11 @@ Here's a basic configuration to route requests to OpenAI, specifically using GPT Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) - - [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) + - [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using Python SDK: - For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). + For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/livekit.mdx b/aigw/integrations/agents/livekit.mdx index b4a10673..e45a2eef 100644 --- a/aigw/integrations/agents/livekit.mdx +++ b/aigw/integrations/agents/livekit.mdx @@ -243,11 +243,11 @@ Create User-specific API keys that automatically: Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) -- [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) +- [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using Python SDK: -For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). +For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/llama-agents.mdx b/aigw/integrations/agents/llama-agents.mdx index 61c3b98e..311e9fb6 100644 --- a/aigw/integrations/agents/llama-agents.mdx +++ b/aigw/integrations/agents/llama-agents.mdx @@ -116,7 +116,7 @@ llm = OpenAI( ### 4\. [Logs](/aigw/product/observability/logs) -Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behavior, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. +Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behaviour, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. The AI Gateway offers comprehensive logging features that capture detailed information about every action and decision made by your AI agents. Access a dedicated section to view records of agent executions, including parameters, outcomes, function calls, and errors. Filter logs based on multiple parameters such as trace ID, model, tokens used, and metadata. diff --git a/aigw/integrations/agents/mastra-agents.mdx b/aigw/integrations/agents/mastra-agents.mdx index e54bda03..ccf45803 100644 --- a/aigw/integrations/agents/mastra-agents.mdx +++ b/aigw/integrations/agents/mastra-agents.mdx @@ -13,7 +13,7 @@ The AI Gateway turns your experimental Mastra agents into production-ready syste - **Built-in reliability** with fallbacks, retries, and load balancing - **Cost tracking and optimization** to manage your AI spend - **Access to 3,000+ LLMs** through a single integration -- **Guardrails** to keep agent behavior safe and compliant +- **Guardrails** to keep agent behaviour safe and compliant - **Version-controlled prompts** for consistent agent performance @@ -619,12 +619,12 @@ Create User-specific API keys that automatically: Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) -- [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) +- [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using Node.js SDK: -For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). +For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/openai-agents-ts.mdx b/aigw/integrations/agents/openai-agents-ts.mdx index 435c88fd..2d93231f 100644 --- a/aigw/integrations/agents/openai-agents-ts.mdx +++ b/aigw/integrations/agents/openai-agents-ts.mdx @@ -12,7 +12,7 @@ The AI Gateway turns your experimental OpenAI Agents into production-ready syste - **Built-in reliability** with fallbacks, retries, and load balancing - **Cost tracking and optimization** to manage your AI spend - **Access to 3,000+ LLMs** through a single integration -- **Guardrails** to keep agent behavior safe and compliant +- **Guardrails** to keep agent behaviour safe and compliant - **Version-controlled prompts** for consistent agent performance @@ -704,11 +704,11 @@ Create User-specific API keys that automatically: Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) -- [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) +- [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using TypeScript SDK: -For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). +For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/openai-agents.mdx b/aigw/integrations/agents/openai-agents.mdx index db38ce5b..4c8e061a 100644 --- a/aigw/integrations/agents/openai-agents.mdx +++ b/aigw/integrations/agents/openai-agents.mdx @@ -11,7 +11,7 @@ The AI Gateway turns your experimental OpenAI Agents into production-ready syste - **Built-in reliability** with fallbacks, retries, and load balancing - **Cost tracking and optimization** to manage your AI spend - **Access to 3,000+ LLMs** through a single integration -- **Guardrails** to keep agent behavior safe and compliant +- **Guardrails** to keep agent behaviour safe and compliant - **Version-controlled prompts** for consistent agent performance @@ -857,11 +857,11 @@ Create User-specific API keys that automatically: Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) -- [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) +- [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using Python SDK: -For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). +For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/openai-swarm.mdx b/aigw/integrations/agents/openai-swarm.mdx index 5a929a67..7f6f1bc6 100644 --- a/aigw/integrations/agents/openai-swarm.mdx +++ b/aigw/integrations/agents/openai-swarm.mdx @@ -139,7 +139,7 @@ Add trace IDs to track specific workflows: ## 5. [Logs and Traces](/aigw/product/observability/logs) -Logs are essential for understanding agent behavior, diagnosing issues, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. +Logs are essential for understanding agent behaviour, diagnosing issues, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. Access a dedicated section to view records of agent executions, including parameters, outcomes, function calls, and errors. Filter logs based on multiple parameters such as trace ID, model, tokens used, and metadata. diff --git a/aigw/integrations/agents/phidata.mdx b/aigw/integrations/agents/phidata.mdx index ac25c2f4..13304058 100644 --- a/aigw/integrations/agents/phidata.mdx +++ b/aigw/integrations/agents/phidata.mdx @@ -128,7 +128,7 @@ llm2 = ChatOpenAI( ### 4\. [Logs](/aigw/product/observability/logs) -Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behavior, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. +Agent runs are complex. Logs are essential for diagnosing issues, understanding agent behaviour, and improving performance. They provide a detailed record of agent activities and tool use, which is crucial for debugging and optimizing processes. The AI Gateway offers comprehensive logging features that capture detailed information about every action and decision made by your AI agents. Access a dedicated section to view records of agent executions, including parameters, outcomes, function calls, and errors. Filter logs based on multiple parameters such as trace ID, model, tokens used, and metadata. diff --git a/aigw/integrations/agents/pydantic-ai.mdx b/aigw/integrations/agents/pydantic-ai.mdx index 7835a900..652f8624 100644 --- a/aigw/integrations/agents/pydantic-ai.mdx +++ b/aigw/integrations/agents/pydantic-ai.mdx @@ -13,7 +13,7 @@ The AI Gateway enhances PydanticAI with production-readiness features, turning y - **Built-in reliability** with fallbacks, retries, and load balancing - **Cost tracking and optimization** to manage your AI spend - **Access to 3,000+ LLMs** through a single integration -- **Guardrails** to keep agent behavior safe and compliant +- **Guardrails** to keep agent behaviour safe and compliant - **OpenTelemetry integration** for comprehensive monitoring @@ -977,11 +977,11 @@ Here's a basic configuration to route requests to OpenAI, specifically using GPT Create API keys through: - [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) - - [API Key Management API](/api-reference/admin-api/control-plane/api-keys/create-api-key) + - [API Key Management API](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) Example using Python SDK: - For detailed key management instructions, see our [API Keys documentation](/api-reference/admin-api/control-plane/api-keys/create-api-key). + For detailed key management instructions, see our [API Keys documentation](/aigw/api-reference/api-keys/post-api-keys-by-sub-type). diff --git a/aigw/integrations/agents/strands.mdx b/aigw/integrations/agents/strands.mdx index 55741e62..db99fed4 100644 --- a/aigw/integrations/agents/strands.mdx +++ b/aigw/integrations/agents/strands.mdx @@ -145,7 +145,7 @@ The agent will automatically use both tools as needed, and every step will be lo ### 1. Enhanced Observability -The AI Gateway provides comprehensive visibility into your agent's behavior without requiring any code changes. +The AI Gateway provides comprehensive visibility into your agent's behaviour without requiring any code changes. @@ -306,7 +306,7 @@ The AI Gateway's guardrails can: -Configure different behavior for development, staging, and production: +Configure different behaviour for development, staging, and production: @@ -500,7 +500,7 @@ Also check the Logs section in Strata Cloud Manager and filter by your metadata. Now that you have the AI Gateway integrated with your Strands agents: -1. **Monitor your agents** in the [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) to understand their behavior +1. **Monitor your agents** in the [Strata Cloud Manager](https://stratacloudmanager.paloaltonetworks.com/) to understand their behaviour 2. **Set up fallbacks** for critical production agents using multiple providers 3. **Add custom metadata** to track different agent types or user segments 4. **Configure budgets and alerts** if you're deploying multiple agents diff --git a/aigw/integrations/guardrails/acuvity.mdx b/aigw/integrations/guardrails/acuvity.mdx index da5ee3ea..0d413396 100644 --- a/aigw/integrations/guardrails/acuvity.mdx +++ b/aigw/integrations/guardrails/acuvity.mdx @@ -24,7 +24,7 @@ To get started with Acuvity, visit their website: * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) Here's your updated table with just the parameter names: diff --git a/aigw/integrations/guardrails/akto.mdx b/aigw/integrations/guardrails/akto.mdx index 6fc1b20e..f57fd9bb 100644 --- a/aigw/integrations/guardrails/akto.mdx +++ b/aigw/integrations/guardrails/akto.mdx @@ -24,7 +24,7 @@ To get started with Akto, visit their website: * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | diff --git a/aigw/integrations/guardrails/azure-guardrails.mdx b/aigw/integrations/guardrails/azure-guardrails.mdx index 62de85b6..331ee46c 100644 --- a/aigw/integrations/guardrails/azure-guardrails.mdx +++ b/aigw/integrations/guardrails/azure-guardrails.mdx @@ -62,7 +62,7 @@ Once authentication is set up, you can add Azure guardrail checks to your AI Gat 5. Save your configuration and create the guardrail - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) ## Azure Content Safety diff --git a/aigw/integrations/guardrails/bedrock-guardrails.mdx b/aigw/integrations/guardrails/bedrock-guardrails.mdx index 7dca8b88..89f8d2ac 100644 --- a/aigw/integrations/guardrails/bedrock-guardrails.mdx +++ b/aigw/integrations/guardrails/bedrock-guardrails.mdx @@ -15,7 +15,7 @@ To get started with AWS Bedrock Guardrails, visit their documentation: * Navigate to `AWS Bedrock` -> `Guardrails` -> `Create guardrail` * Configure the guardrail according to your requirements -* For `PII redaction`, we recommend setting the Guardrail behavior as **BLOCK** for the required entity types. This is necessary because Bedrock does not apply PII checks on input (request message) if the behavior is set to MASK +* For `PII redaction`, we recommend setting the Guardrail behaviour as **BLOCK** for the required entity types. This is necessary because Bedrock does not apply PII checks on input (request message) if the behaviour is set to MASK * Once the guardrail is created, note the **ID** and **version** displayed on the console - you'll need these to enable the guardrail in the AI Gateway ### 2. Enable Bedrock Plugin on the AI Gateway @@ -33,7 +33,7 @@ To get started with AWS Bedrock Guardrails, visit their documentation: * Set any actions you want on your guardrail check, and click `Create` - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) ### 4. Add Guardrail ID to a Config and Make Your Request diff --git a/aigw/integrations/guardrails/bring-your-own-guardrails.mdx b/aigw/integrations/guardrails/bring-your-own-guardrails.mdx index 3dae013e..f889738d 100644 --- a/aigw/integrations/guardrails/bring-your-own-guardrails.mdx +++ b/aigw/integrations/guardrails/bring-your-own-guardrails.mdx @@ -507,7 +507,7 @@ When triggered on a proxy request, the webhook payload includes `method`, `path` 2. **Independent Verdict and Transformation**: The `verdict` and any transformations are independent. You can return `verdict: false` while still returning transformations. -3. **Default Behavior**: If your webhook fails to respond within the timeout period, the AI Gateway will default to `verdict: true`. +3. **Default Behaviour**: If your webhook fails to respond within the timeout period, the AI Gateway will default to `verdict: true`. 4. **Event Type Awareness**: When implementing transformations, ensure your webhook checks the `eventType` field to determine whether it's being called before or after the LLM request. diff --git a/aigw/integrations/guardrails/crowdstrike-aidr.mdx b/aigw/integrations/guardrails/crowdstrike-aidr.mdx index 85888c61..f90646ba 100644 --- a/aigw/integrations/guardrails/crowdstrike-aidr.mdx +++ b/aigw/integrations/guardrails/crowdstrike-aidr.mdx @@ -23,7 +23,7 @@ description: "CrowdStrike AI Detection and Response (AIDR) integration for scann * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | diff --git a/aigw/integrations/guardrails/f5-guardrails.mdx b/aigw/integrations/guardrails/f5-guardrails.mdx index 11462379..2bb03c14 100644 --- a/aigw/integrations/guardrails/f5-guardrails.mdx +++ b/aigw/integrations/guardrails/f5-guardrails.mdx @@ -28,7 +28,7 @@ description: "F5 Guardrails (powered by CalypsoAI) provides advanced content mod * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | diff --git a/aigw/integrations/guardrails/headroom.mdx b/aigw/integrations/guardrails/headroom.mdx index dbb83b91..43556c12 100644 --- a/aigw/integrations/guardrails/headroom.mdx +++ b/aigw/integrations/guardrails/headroom.mdx @@ -58,7 +58,7 @@ Verify the deployment with `curl http://your-headroom-host:8787/health`. * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Supported Hooks | diff --git a/aigw/integrations/guardrails/javelin.mdx b/aigw/integrations/guardrails/javelin.mdx index a6aefa1d..6034ceb6 100644 --- a/aigw/integrations/guardrails/javelin.mdx +++ b/aigw/integrations/guardrails/javelin.mdx @@ -28,7 +28,7 @@ Javelin's unified guardrails approach automatically applies all enabled guardrai * Set any actions you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | @@ -245,7 +245,7 @@ The `application` field is **required** as it determines which guardrails policy The unified guardrails approach offers several advantages: ### Centralized Policy Management -- Configure all guardrail rules, thresholds, and behaviors in the Javelin platform +- Configure all guardrail rules, thresholds, and behaviours in the Javelin platform - Changes to your security policy are instantly reflected across all applications - No need to update code or configurations when adjusting security parameters diff --git a/aigw/integrations/guardrails/lasso.mdx b/aigw/integrations/guardrails/lasso.mdx index 43cf1386..0b2ce3db 100644 --- a/aigw/integrations/guardrails/lasso.mdx +++ b/aigw/integrations/guardrails/lasso.mdx @@ -25,7 +25,7 @@ To get started with Lasso Security, visit their documentation: * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) ### Available Checks @@ -110,7 +110,7 @@ Your requests are now guarded by Lasso Security's protective measures, and you c Lasso Security's Deputies analyze content for various security risks across multiple categories: -1. **Prompt Injections**: Detects attempts to manipulate AI behavior through crafted inputs +1. **Prompt Injections**: Detects attempts to manipulate AI behaviour through crafted inputs 2. **Data Leaks**: Prevents sensitive information from being exposed through AI interactions 3. **Jailbreak Attempts**: Identifies attempts to bypass AI safety mechanisms 4. **Custom Policy Violations**: Enforces your organisation's specific security policies diff --git a/aigw/integrations/guardrails/mistral.mdx b/aigw/integrations/guardrails/mistral.mdx index b8742ba7..aff1a745 100644 --- a/aigw/integrations/guardrails/mistral.mdx +++ b/aigw/integrations/guardrails/mistral.mdx @@ -34,7 +34,7 @@ To get started with Mistral, visit their documentation: * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | diff --git a/aigw/integrations/guardrails/palo-alto-panw-prisma.mdx b/aigw/integrations/guardrails/palo-alto-panw-prisma.mdx index 975cd500..f2a0bc12 100644 --- a/aigw/integrations/guardrails/palo-alto-panw-prisma.mdx +++ b/aigw/integrations/guardrails/palo-alto-panw-prisma.mdx @@ -42,7 +42,7 @@ Before integrating with the AI Gateway: * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | @@ -150,7 +150,7 @@ Prisma AIRS provides multi-layered protection against various AI-specific threat ### Security Threats Detected 1. **Malicious URL Detection**: Detects and block Malicious URLs in your LLM requests/response -2. **Prompt Injections**: Detects and blocks attempts to manipulate AI behavior through malicious prompts +2. **Prompt Injections**: Detects and blocks attempts to manipulate AI behaviour through malicious prompts 3. **Sensitive Data Leakage**: Prevents PII, secrets, and confidential information from being exposed 4. **Insecure Outputs**: Blocks responses containing malware, malicious URLs, or harmful content 5. **Model DoS Attacks**: Protects against attempts to overwhelm or disable AI models diff --git a/aigw/integrations/guardrails/pangea.mdx b/aigw/integrations/guardrails/pangea.mdx index d9158788..88e199b6 100644 --- a/aigw/integrations/guardrails/pangea.mdx +++ b/aigw/integrations/guardrails/pangea.mdx @@ -24,7 +24,7 @@ To get started with Pangea, visit their documentation: * Set any actions you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | diff --git a/aigw/integrations/guardrails/patronus-ai.mdx b/aigw/integrations/guardrails/patronus-ai.mdx index 8d93aec6..80c21614 100644 --- a/aigw/integrations/guardrails/patronus-ai.mdx +++ b/aigw/integrations/guardrails/patronus-ai.mdx @@ -2,7 +2,7 @@ title: "Patronus AI" description: "Patronus excels in industry-specific guardrails for RAG workflows." --- - It has a SOTA hallucination detection model Lynx, which is also [open source](https://www.patronus.ai/blog/lynx-state-of-the-art-open-source-hallucination-detection-model). The Prisma AIRS AI Gateway integrates with multiple Patronus evaluators to help you enforce LLM behavior. + It has a SOTA hallucination detection model Lynx, which is also [open source](https://www.patronus.ai/blog/lynx-state-of-the-art-open-source-hallucination-detection-model). The Prisma AIRS AI Gateway integrates with multiple Patronus evaluators to help you enforce LLM behaviour. Browse Patronus' docs for more info: diff --git a/aigw/integrations/guardrails/prompt-security.mdx b/aigw/integrations/guardrails/prompt-security.mdx index deee7efa..c1e5b2c3 100644 --- a/aigw/integrations/guardrails/prompt-security.mdx +++ b/aigw/integrations/guardrails/prompt-security.mdx @@ -30,7 +30,7 @@ To get started with Prompt Security, visit their website: * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | diff --git a/aigw/integrations/guardrails/qualifire.mdx b/aigw/integrations/guardrails/qualifire.mdx index e3595371..8e199e14 100644 --- a/aigw/integrations/guardrails/qualifire.mdx +++ b/aigw/integrations/guardrails/qualifire.mdx @@ -25,7 +25,7 @@ To get started with Qualifire, visit their website: * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) ## Available Guardrail Checks diff --git a/aigw/integrations/guardrails/request-parameters-check.mdx b/aigw/integrations/guardrails/request-parameters-check.mdx index 02488e4d..1fddd07c 100644 --- a/aigw/integrations/guardrails/request-parameters-check.mdx +++ b/aigw/integrations/guardrails/request-parameters-check.mdx @@ -35,7 +35,7 @@ This guardrail runs on **input only** (`beforeRequestHook`). * Set any `actions` you want on your check, and create the Guardrail! -Guardrail Actions let you orchestrate your guardrail's behaviour (deny, feedback, etc.). Learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions). +Guardrail Actions let you orchestrate your guardrail's behaviour (deny, feedback, etc.). Learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions). | Check Name | Description | Parameters | Supported Hooks | diff --git a/aigw/integrations/guardrails/tavily.mdx b/aigw/integrations/guardrails/tavily.mdx index 7b2b0f79..b21a794f 100644 --- a/aigw/integrations/guardrails/tavily.mdx +++ b/aigw/integrations/guardrails/tavily.mdx @@ -54,7 +54,7 @@ Requires Backend `v1.16.0+`. * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Supported Hooks | diff --git a/aigw/integrations/guardrails/zscaler.mdx b/aigw/integrations/guardrails/zscaler.mdx index 988d8ce2..74e62abd 100644 --- a/aigw/integrations/guardrails/zscaler.mdx +++ b/aigw/integrations/guardrails/zscaler.mdx @@ -25,7 +25,7 @@ description: "Zscaler AI Guard integration for enforcing security policies on LL * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) | Check Name | Description | Parameters | Supported Hooks | diff --git a/aigw/integrations/libraries/claude-desktop-developers.mdx b/aigw/integrations/libraries/claude-desktop-developers.mdx index b26257f3..3fafccaa 100644 --- a/aigw/integrations/libraries/claude-desktop-developers.mdx +++ b/aigw/integrations/libraries/claude-desktop-developers.mdx @@ -125,7 +125,7 @@ Send any prompt in Claude Desktop, then run through these two checks. - Model routing is controlled by your admin's config in the AI Gateway. If you're getting unexpected model behavior, your team's config likely points to a different model than you expect. Check with your platform team. + Model routing is controlled by your admin's config in the AI Gateway. If you're getting unexpected model behaviour, your team's config likely points to a different model than you expect. Check with your platform team. diff --git a/aigw/integrations/libraries/codex.mdx b/aigw/integrations/libraries/codex.mdx index bc4635c0..28455bc2 100644 --- a/aigw/integrations/libraries/codex.mdx +++ b/aigw/integrations/libraries/codex.mdx @@ -141,7 +141,7 @@ wire_api = "responses" ### Adding Model Capabilities **Reasoning, output, and tools (top-level)** -These top-level keys apply to the current session model and control reasoning, output, and tool behavior: +These top-level keys apply to the current session model and control reasoning, output, and tool behaviour: | Key | Values | Description | | --- | ------ | ----------- | diff --git a/aigw/integrations/libraries/langchain-js.mdx b/aigw/integrations/libraries/langchain-js.mdx index 78ede6cf..89e08931 100644 --- a/aigw/integrations/libraries/langchain-js.mdx +++ b/aigw/integrations/libraries/langchain-js.mdx @@ -395,7 +395,7 @@ console.log(vectors); ``` -The AI Gateway supports OpenAI embeddings via `OpenAIEmbeddings`. For other providers (Cohere, Voyage), call the [embeddings endpoint](/aigw/api-reference/inference-api/embeddings) directly. +The AI Gateway supports OpenAI embeddings via `OpenAIEmbeddings`. For other providers (Cohere, Voyage), call the [embeddings endpoint](/aigw/api-reference/embeddings/create-embedding) directly. ## Migration from Direct OpenAI diff --git a/aigw/integrations/libraries/langchain-python.mdx b/aigw/integrations/libraries/langchain-python.mdx index 50860869..aa4a90e7 100644 --- a/aigw/integrations/libraries/langchain-python.mdx +++ b/aigw/integrations/libraries/langchain-python.mdx @@ -364,7 +364,7 @@ All routing decisions are tracked in the AI Gateway with full observability—se - ✅ A/B testing with traffic distribution **Use fixed models** when you need: -- ✅ Simple, predictable behavior +- ✅ Simple, predictable behaviour - ✅ Consistent model across all requests - ✅ Easier debugging @@ -405,7 +405,7 @@ vectors = embeddings.embed_documents(["Hello world", "Goodbye world"]) ``` -The AI Gateway supports OpenAI embeddings via `OpenAIEmbeddings`. For other providers (Cohere, Voyage), call the [embeddings endpoint](/aigw/api-reference/inference-api/embeddings) directly. +The AI Gateway supports OpenAI embeddings via `OpenAIEmbeddings`. For other providers (Cohere, Voyage), call the [embeddings endpoint](/aigw/api-reference/embeddings/create-embedding) directly. ## Prompt Management diff --git a/aigw/integrations/libraries/openai-agent-builder-python.mdx b/aigw/integrations/libraries/openai-agent-builder-python.mdx index 5fafd1af..df9b7b82 100644 --- a/aigw/integrations/libraries/openai-agent-builder-python.mdx +++ b/aigw/integrations/libraries/openai-agent-builder-python.mdx @@ -9,7 +9,7 @@ OpenAI Agent Builder is a visual canvas for creating multi-step agent workflows. - Cost tracking and optimization - Reliability features (fallbacks, retries) - Access to 3,000+ LLMs -- Guardrails for safe agent behavior +- Guardrails for safe agent behaviour ## Quick Start diff --git a/aigw/integrations/libraries/openai-agent-builder.mdx b/aigw/integrations/libraries/openai-agent-builder.mdx index 431c7da3..78526a6b 100644 --- a/aigw/integrations/libraries/openai-agent-builder.mdx +++ b/aigw/integrations/libraries/openai-agent-builder.mdx @@ -9,7 +9,7 @@ OpenAI Agent Builder is a visual canvas for creating multi-step agent workflows. - Cost tracking and optimization - Reliability features (fallbacks, retries) - Access to 3,000+ LLMs -- Guardrails for safe agent behavior +- Guardrails for safe agent behaviour ## Quick Start diff --git a/aigw/integrations/libraries/vercel.mdx b/aigw/integrations/libraries/vercel.mdx index b471eb0c..333ba400 100644 --- a/aigw/integrations/libraries/vercel.mdx +++ b/aigw/integrations/libraries/vercel.mdx @@ -318,7 +318,7 @@ console.log('Steps taken:', result.steps.length); ### Custom Parameters -Fine-tune model behavior with temperature, tokens, and retries: +Fine-tune model behaviour with temperature, tokens, and retries: ```typescript import { generateText } from 'ai'; diff --git a/aigw/integrations/llms/anthropic.mdx b/aigw/integrations/llms/anthropic.mdx index 537adea5..1fda93b9 100644 --- a/aigw/integrations/llms/anthropic.mdx +++ b/aigw/integrations/llms/anthropic.mdx @@ -220,7 +220,7 @@ Performance: There is zero overhead when the setting is disabled. When enabled, In your [config](/aigw/product/ai-gateway/configs), add `529` to the retry `on_status_codes` (or fallback `on_status_codes`). This supports all existing config combinations. - Attach the updated config to your API key so the new behavior applies to all routed requests. + Attach the updated config to your API key so the new behaviour applies to all routed requests. diff --git a/aigw/integrations/llms/azure-openai/fine-tuning.mdx b/aigw/integrations/llms/azure-openai/fine-tuning.mdx index 0336b584..99baff5a 100644 --- a/aigw/integrations/llms/azure-openai/fine-tuning.mdx +++ b/aigw/integrations/llms/azure-openai/fine-tuning.mdx @@ -77,6 +77,6 @@ print(fine_tune_job) -For more detailed examples and other fine-tuning operations (listing jobs, retrieving job details, canceling jobs, and getting job events), please refer to the [OpenAI fine-tuning documentation](/aigw/integrations/llms/openai/fine-tuning). +For more detailed examples and other fine-tuning operations (listing jobs, retrieving job details, cancelling jobs, and getting job events), please refer to the [OpenAI fine-tuning documentation](/aigw/integrations/llms/openai/fine-tuning). The Azure OpenAI fine-tuning API documentation is available at [Azure OpenAI API](https://learn.microsoft.com/en-us/rest/api/azureopenai/fine-tuning/create?view=rest-azureopenai-2025-01-01-preview&tabs=HTTP). diff --git a/aigw/integrations/llms/byollm.mdx b/aigw/integrations/llms/byollm.mdx index fda5321a..85d46248 100644 --- a/aigw/integrations/llms/byollm.mdx +++ b/aigw/integrations/llms/byollm.mdx @@ -188,7 +188,7 @@ The AI Gateway provides comprehensive observability for your private LLM deploym | Connection Errors | Incorrect URL, network issues, firewall rules | Verify URL format, check network connectivity, confirm firewall allows traffic | | Authentication Failures | Invalid credentials, incorrect header format | Check credentials, ensure headers are correctly formatted and forwarded | | Timeout Errors | LLM server overloaded, request too complex | Adjust timeout settings, implement load balancing, simplify requests | -| Inconsistent Responses | Different model versions, configuration differences | Standardize model versions, document expected behavior differences | +| Inconsistent Responses | Different model versions, configuration differences | Standardize model versions, document expected behaviour differences | ## FAQs diff --git a/aigw/integrations/llms/deepseek.mdx b/aigw/integrations/llms/deepseek.mdx index 6e206d8f..b9fa1cee 100644 --- a/aigw/integrations/llms/deepseek.mdx +++ b/aigw/integrations/llms/deepseek.mdx @@ -102,7 +102,7 @@ DeepSeek supports tool calling (function calling) on the `deepseek-chat` model. ## Reasoning (deepseek-reasoner) -The `deepseek-reasoner` model supports chain-of-thought reasoning. Use the `reasoning_effort` parameter to control reasoning behavior: +The `deepseek-reasoner` model supports chain-of-thought reasoning. Use the `reasoning_effort` parameter to control reasoning behaviour: When streaming, `reasoning_content` is included in the delta for `deepseek-reasoner` responses. diff --git a/aigw/integrations/llms/gemini.mdx b/aigw/integrations/llms/gemini.mdx index b0535147..49f5735c 100644 --- a/aigw/integrations/llms/gemini.mdx +++ b/aigw/integrations/llms/gemini.mdx @@ -96,7 +96,7 @@ Save your configuration. Your provider slug will be `@google` (or a custom name -The AI Gateway supports the `system_instructions` parameter for Google Gemini 1.5 - allowing you to control the behavior and output of your Gemini-powered applications with ease. +The AI Gateway supports the `system_instructions` parameter for Google Gemini 1.5 - allowing you to control the behaviour and output of your Gemini-powered applications with ease. Simply include your Gemini system prompt as part of the `{"role":"system"}` message within the `messages` array of your request body. AI Gateway will automatically transform your message to ensure seamless compatibility with the Google Gemini API. diff --git a/aigw/integrations/llms/local-ai.mdx b/aigw/integrations/llms/local-ai.mdx index c7fba5bc..80cfb486 100644 --- a/aigw/integrations/llms/local-ai.mdx +++ b/aigw/integrations/llms/local-ai.mdx @@ -50,9 +50,9 @@ ngrok http 8080 | Endpoint | Resource | | :----------------------------------------------- | :---------------------------------------------------------------- | -| /chat/completions (Chat, Vision, Tools support) | [Doc](/api-reference/inference-api/chat) | -| /images/generations | [Doc](/api-reference/inference-api/images/create-image) | -| /embeddings | [Doc](/api-reference/inference-api/embeddings) | +| /chat/completions (Chat, Vision, Tools support) | [Doc](/aigw/api-reference/chat/create-chat-completion) | +| /images/generations | [Doc](/aigw/api-reference/images/create-image) | +| /embeddings | [Doc](/aigw/api-reference/embeddings/create-embedding) | | /audio/transcriptions | [Doc](/aigw/product/ai-gateway/multimodal-capabilities/speech-to-text) | --- diff --git a/aigw/integrations/llms/recraft-ai.mdx b/aigw/integrations/llms/recraft-ai.mdx index 91009085..5f2b1912 100644 --- a/aigw/integrations/llms/recraft-ai.mdx +++ b/aigw/integrations/llms/recraft-ai.mdx @@ -106,7 +106,7 @@ The AI Gateway uses the OpenAI image generation signature for Recraft AI, allowi Cache generated images - + Complete image generation API docs diff --git a/aigw/integrations/llms/segmind.mdx b/aigw/integrations/llms/segmind.mdx index daff5fba..30c1c3df 100644 --- a/aigw/integrations/llms/segmind.mdx +++ b/aigw/integrations/llms/segmind.mdx @@ -72,7 +72,7 @@ The AI Gateway uses the OpenAI image generation signature for Segmind, allowing Cache generated images - + Complete image generation API docs diff --git a/aigw/integrations/llms/together-ai.mdx b/aigw/integrations/llms/together-ai.mdx index 954dd7df..3ac3bc8e 100644 --- a/aigw/integrations/llms/together-ai.mdx +++ b/aigw/integrations/llms/together-ai.mdx @@ -85,7 +85,7 @@ console.log(response.choices[0].message.content) ## Reasoning / Thinking Support -Together AI supports reasoning models that expose their internal chain of thought. Use the `reasoning_effort` parameter to control reasoning behavior, and set `strict_open_ai_compliance=False` to receive the thinking content in `content_blocks`. +Together AI supports reasoning models that expose their internal chain of thought. Use the `reasoning_effort` parameter to control reasoning behaviour, and set `strict_open_ai_compliance=False` to receive the thinking content in `content_blocks`. diff --git a/aigw/integrations/llms/vertex-ai.mdx b/aigw/integrations/llms/vertex-ai.mdx index 5efd37f6..5a8f9e0d 100644 --- a/aigw/integrations/llms/vertex-ai.mdx +++ b/aigw/integrations/llms/vertex-ai.mdx @@ -863,7 +863,7 @@ curl https://aigw.portkey.ai/v1/images/generations \ ``` -[Image Generation API Reference](/api-reference/inference-api/images/create-image) +[Image Generation API Reference](/aigw/api-reference/images/create-image) ### List of Supported Imagen Models - `imagen-3.0-generate-001` @@ -883,7 +883,7 @@ Unlike single-call video APIs (for example, [OpenAI’s Sora API](https://platfo 1. **Step 1 – Start generation:** Send a `predictLongRunning` request with your prompt (and optional image/video inputs). The API returns immediately with an **operation name** (no video yet). 2. **Step 2 – Poll until done:** Call `fetchPredictOperation` with that operation name repeatedly (e.g., every 30–60 seconds) until `done` is `true`. The final response contains the generated video(s), either as URIs (if you set `storageUri`) or as base64-encoded bytes. -For full request/response shapes, parameters, and polling behavior, see the [Veo on Vertex AI video generation API reference](https://docs.cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation). +For full request/response shapes, parameters, and polling behaviour, see the [Veo on Vertex AI video generation API reference](https://docs.cloud.google.com/vertex-ai/generative-ai/docs/model-reference/veo-video-generation). ### Implementation: Request Then Poll diff --git a/aigw/integrations/llms/vertex-ai/batches.mdx b/aigw/integrations/llms/vertex-ai/batches.mdx index 85e6f1f8..1b6f0e32 100644 --- a/aigw/integrations/llms/vertex-ai/batches.mdx +++ b/aigw/integrations/llms/vertex-ai/batches.mdx @@ -12,7 +12,7 @@ With Prisma AIRS AI Gateway, you can perform batch inference operations with Ver 1. **AI Gateway API key** and a **Vertex AI provider** configured in Model Catalog. 2. A **GCS bucket** in the same region as your model + `aiplatform-service-agent` permission on the file. 3. *(Only for gateway-native batching)* A **AI Gateway File** (`input_file_id`). -4. Familiarity with the [Create Batch OpenAPI spec](/api-reference/inference-api/batch/create-batch). +4. Familiarity with the [Create Batch OpenAPI spec](/aigw/api-reference/batch/create-batch). The AI Gateway supports **two modes** on Vertex: diff --git a/aigw/integrations/llms/vertex-ai/embeddings.mdx b/aigw/integrations/llms/vertex-ai/embeddings.mdx index 3249eb3a..801f1234 100644 --- a/aigw/integrations/llms/vertex-ai/embeddings.mdx +++ b/aigw/integrations/llms/vertex-ai/embeddings.mdx @@ -401,7 +401,7 @@ You can combine multiple input types in a single request: ### Setting Task Type and Dimensions -You can optionally specify `task_type` and `dimensions` to control the embedding behavior: +You can optionally specify `task_type` and `dimensions` to control the embedding behaviour: ```json { diff --git a/aigw/integrations/plugins/exa.mdx b/aigw/integrations/plugins/exa.mdx index c5229c30..0867e7cb 100644 --- a/aigw/integrations/plugins/exa.mdx +++ b/aigw/integrations/plugins/exa.mdx @@ -45,7 +45,7 @@ This process allows any LLM to respond with up-to-date knowledge without retrain * Set any `actions` you want on your check, and create the Guardrail! - Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions) + Guardrail Actions allow you to orchestrate your guardrails logic. You can learn more about them [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions) * Save your guardrail diff --git a/aigw/integrations/plugins/tavily.mdx b/aigw/integrations/plugins/tavily.mdx index 6453cdc1..4d05191c 100644 --- a/aigw/integrations/plugins/tavily.mdx +++ b/aigw/integrations/plugins/tavily.mdx @@ -34,7 +34,7 @@ If the request has no usable text, or Tavily returns no results, the request pas * Open the `Guardrails` page and click `Create`. * Search for **Tavily Online Search** and click `Add`. -* Configure the search behavior for your use case. +* Configure the search behaviour for your use case. #### Core settings @@ -75,7 +75,7 @@ If the request has no usable text, or Tavily returns no results, the request pas * Set any `actions` you want on the guardrail, then save it. - Guardrail Actions let you compose multiple checks into one workflow. Learn more [here](/aigw/product/guardrails#there-are-6-types-of-guardrail-actions). + Guardrail Actions let you compose multiple checks into one workflow. Learn more [here](/aigw/product/guardrails#there-are-7-types-of-guardrail-actions). ### 3. Add the Guardrail to a Config @@ -198,7 +198,7 @@ Tavily-enriched requests are visible in Strata Cloud Manager. You can inspect: - Use `autoParameters` when you want Tavily to choose the best topic and depth automatically. Leave it off when you need predictable behavior or tighter control over latency and cost. + Use `autoParameters` when you want Tavily to choose the best topic and depth automatically. Leave it off when you need predictable behaviour or tighter control over latency and cost. diff --git a/aigw/integrations/tracing-providers/arize.mdx b/aigw/integrations/tracing-providers/arize.mdx index fb230ac3..2ad22764 100644 --- a/aigw/integrations/tracing-providers/arize.mdx +++ b/aigw/integrations/tracing-providers/arize.mdx @@ -13,7 +13,7 @@ Use Arize AX for production AI observability and evaluation. If you want an open ## Why the AI Gateway + Arize AX? -Thanks to OpenInference instrumentation, the AI Gateway can emit structured traces automatically. This gives you visibility into each LLM call routed through the gateway, making it easier to debug behavior, inspect token usage, and evaluate production traffic in Arize AX. For production evaluation patterns, see Arize's [agent evaluation guide](https://arize.com/guides/ai-agent-handbook/agent-evaluation/) and [LLM evaluation guide](https://arize.com/resources/llm-evaluation/). +Thanks to OpenInference instrumentation, the AI Gateway can emit structured traces automatically. This gives you visibility into each LLM call routed through the gateway, making it easier to debug behaviour, inspect token usage, and evaluate production traffic in Arize AX. For production evaluation patterns, see Arize's [agent evaluation guide](https://arize.com/guides/ai-agent-handbook/agent-evaluation/) and [LLM evaluation guide](https://arize.com/resources/llm-evaluation/). diff --git a/aigw/integrations/tracing-providers/ml-flow.mdx b/aigw/integrations/tracing-providers/ml-flow.mdx index 2c4b78e8..da6706e3 100644 --- a/aigw/integrations/tracing-providers/ml-flow.mdx +++ b/aigw/integrations/tracing-providers/ml-flow.mdx @@ -3,7 +3,7 @@ title: "MLflow Tracing" description: "Enhance LLM observability with automatic tracing and intelligent gateway routing" --- -[MLflow Tracing](https://mlflow.org/docs/latest/llms/tracing/index.html) is a feature that enhances LLM observability in your Generative AI (GenAI) applications by capturing detailed information about the execution of your application's services. Tracing provides a way to record the inputs, outputs, and metadata associated with each intermediate step of a request, enabling you to easily pinpoint the source of bugs and unexpected behaviors. +[MLflow Tracing](https://mlflow.org/docs/latest/llms/tracing/index.html) is a feature that enhances LLM observability in your Generative AI (GenAI) applications by capturing detailed information about the execution of your application's services. Tracing provides a way to record the inputs, outputs, and metadata associated with each intermediate step of a request, enabling you to easily pinpoint the source of bugs and unexpected behaviours. MLflow offers automatic, no-code-added integrations with over 20 popular GenAI libraries, providing immediate observability with just a single line of code. Combined with Prisma AIRS AI Gateway's intelligent gateway, you get comprehensive tracing enriched with routing decisions and performance optimizations. diff --git a/aigw/integrations/tracing-providers/phoenix.mdx b/aigw/integrations/tracing-providers/phoenix.mdx index 0a1aafb0..2dce66d0 100644 --- a/aigw/integrations/tracing-providers/phoenix.mdx +++ b/aigw/integrations/tracing-providers/phoenix.mdx @@ -15,13 +15,13 @@ Phoenix's OpenInference instrumentation combined with the AI Gateway's intellige -Powerful UI for exploring traces, spans, and debugging LLM behavior +Powerful UI for exploring traces, spans, and debugging LLM behaviour Industry-standard semantic conventions for AI/LLM observability -Built-in tools for evaluating model performance and behavior +Built-in tools for evaluating model performance and behaviour The AI Gateway adds caching, fallbacks, and load balancing to every request diff --git a/aigw/introduction/feature-overview.mdx b/aigw/introduction/feature-overview.mdx index 17e42517..f24fd4a6 100644 --- a/aigw/introduction/feature-overview.mdx +++ b/aigw/introduction/feature-overview.mdx @@ -77,7 +77,7 @@ Gain real-time insights, track key metrics, and streamline debugging with our Op ## Guardrails -Enforce Real-Time LLM Behavior with 50+ state-of-the-art AI guardrails, so that you can synchronously run Guardrails on your requests and route them with precision. +Enforce Real-Time LLM Behaviour with 50+ state-of-the-art AI guardrails, so that you can synchronously run Guardrails on your requests and route them with precision. diff --git a/aigw/product/administration/enforce-default-config.mdx b/aigw/product/administration/enforce-default-config.mdx index 69c0a2f9..806b16dd 100644 --- a/aigw/product/administration/enforce-default-config.mdx +++ b/aigw/product/administration/enforce-default-config.mdx @@ -67,12 +67,11 @@ You can also programmatically attach config when creating or updating API keys u ```bash -curl -X POST https://aigw.portkey.ai/v1/admin/api-keys \ +curl -X POST "https://api.apps.paloaltonetworks.com/ai_gw/v2/api-keys/service" \ -H "Content-Type: application/json" \ - -H "Authorization: Bearer YOUR_ADMIN_API_KEY" \ + -H "Authorization: Bearer $ACCESS_TOKEN" \ -d '{ "name": "engineering-team", - "type": "organisation", "workspace_id": "YOUR_WORKSPACE_ID", "defaults": { "config_id": "pc-your-config-id", @@ -91,11 +90,11 @@ curl -X POST https://aigw.portkey.ai/v1/admin/api-keys \ For detailed information on API key management, refer to our API documentation: - + Learn how to create API keys with default configs - + Learn how to update existing API keys with new default configs @@ -152,7 +151,7 @@ If you use [JWT-based authentication](/aigw/product/enterprise-offering/org-mana ## Config Precedence -When using API keys with default configs, the AI Gateway provides flexible options for controlling config behavior: +When using API keys with default configs, the AI Gateway provides flexible options for controlling config behaviour: - The default config attached to the API key will be automatically applied to all requests made with that key - By default, if a user explicitly specifies a config ID in their request, that config will override the default config attached to the API key diff --git a/aigw/product/administration/enforce-organisation-level-guardrails.mdx b/aigw/product/administration/enforce-organisation-level-guardrails.mdx index 9ee9e6c8..152f218f 100644 --- a/aigw/product/administration/enforce-organisation-level-guardrails.mdx +++ b/aigw/product/administration/enforce-organisation-level-guardrails.mdx @@ -27,17 +27,17 @@ Once configured, these guardrails will be enforced on all API requests across th ## Excluding Workspaces -Some workspaces — an internal evaluation sandbox, for example — may need to run without the organisation's default guardrails. You can exempt them individually while every other workspace stays covered. +Some workspaces, an internal evaluation sandbox for example, may need to run without the organisation's default guardrails. You can exempt them individually while every other workspace stays covered. -Exclusions are managed per guardrail direction through the Admin API, and require an organisation service API key with the `organisation_exclusions.update` and `organisation_exclusions.list` scopes. +Exclusions are managed per guardrail direction through the Admin API, so the call is authorised with a [Strata Cloud Manager access token](/aigw/api-reference/admin-api/authentication) whose service account can administer the organisation. Workspace exclusions are currently available via the API only. Support for managing them from Strata Cloud Manager is coming shortly. ```bash -curl -X PUT "https://aigw.portkey.ai/v1/workspace-exclusions/input-guardrails" \ - -H "Authorization: Bearer YOUR_ADMIN_API_KEY" \ +curl -X PUT "https://api.apps.paloaltonetworks.com/ai_gw/admin/v2/workspace-exclusions/input-guardrails" \ + -H "Authorization: Bearer $ACCESS_TOKEN" \ -H "Content-Type: application/json" \ -d '{ "organisation_id": "ORGANISATION_ID", diff --git a/aigw/product/administration/enforce-saved-only-config.mdx b/aigw/product/administration/enforce-saved-only-config.mdx index 7d14d64f..f77dab05 100644 --- a/aigw/product/administration/enforce-saved-only-config.mdx +++ b/aigw/product/administration/enforce-saved-only-config.mdx @@ -32,7 +32,7 @@ When a request arrives, the Gateway resolves whether saved-only mode is active f Because the check runs at a single point before routing, blocked requests never reach an upstream provider, and every rejection is emitted from one consistent place, making the errors easy to detect, log, and alert on. -## Default Behavior and Rollout +## Default Behaviour and Rollout How this setting is initialized depends on when your organisation was created: diff --git a/aigw/product/administration/enforce-workspace-budget-and-rate-limits.mdx b/aigw/product/administration/enforce-workspace-budget-and-rate-limits.mdx index a83162fb..06e88c5e 100644 --- a/aigw/product/administration/enforce-workspace-budget-and-rate-limits.mdx +++ b/aigw/product/administration/enforce-workspace-budget-and-rate-limits.mdx @@ -97,7 +97,7 @@ Workspace budget limits are particularly useful for: - **Departmental Allocations**: Assign specific AI budgets to different departments (Marketing, Customer Support, R&D) - **Project Management**: Allocate resources based on project priority and requirements -- **Cost Center Tracking**: Monitor and control spending across different cost centers +- **Cost Centre Tracking**: Monitor and control spending across different cost centres - **Phased Rollouts**: Gradually increase limits as teams demonstrate value and mature their AI use cases ### Set Workspace Budget and Rate Limits using AI Gateway Admin API diff --git a/aigw/product/ai-gateway/batches.mdx b/aigw/product/ai-gateway/batches.mdx index 33bb26a1..aaaca903 100644 --- a/aigw/product/ai-gateway/batches.mdx +++ b/aigw/product/ai-gateway/batches.mdx @@ -25,7 +25,7 @@ Have the following ready to start making batch requests: 2. [Data Service](/aigw/changelog/data-service) to be enabled — required for **AI Gateway Managed Batching** or when **cost-attribution** is needed. 3. **Provider credentials** for each downstream model (OpenAI key, Bedrock IAM role, etc.). 4. A **AI Gateway File** (`input_file_id`) - **required only when using the AI Gateway Batch API (Mode #2)**. See [Files](/aigw/product/ai-gateway/files) to upload one. -5. Optional: Familiarity with the [Create Batch OpenAPI spec](/aigw/api-reference/inference-api/batch/create-batch). +5. Optional: Familiarity with the [Create Batch OpenAPI spec](/aigw/api-reference/batch/create-batch). --- @@ -33,7 +33,7 @@ Have the following ready to start making batch requests: Used to run batch jobs with the provider's native batch endpoint. Providers usually offer a cheaper rate for batch jobs, but you'll be limited by the provider's quota and limits. Most completion windows are about 24 hours. -**Polling for batch status**: The AI Gateway is stateless and does not poll for completion status of batches on the provider side. You must poll the batch status manually using the unified API with the same signature for all supported providers. See [Retrieve Batch](/api-reference/inference-api/batch/retrieve-batch) for details. +**Polling for batch status**: The AI Gateway is stateless and does not poll for completion status of batches on the provider side. You must poll the batch status manually using the unified API with the same signature for all supported providers. See [Retrieve Batch](/aigw/api-reference/batch/retrieve-batch) for details. ### Quickstart (OpenAI example) @@ -55,7 +55,7 @@ curl -X POST https://aigw.portkey.ai/v1/batches \ -> 🔗 Full schema: see the [OpenAPI reference](/api-reference/inference-api/batch/create-batch). +> 🔗 Full schema: see the [OpenAPI reference](/aigw/api-reference/batch/create-batch). ### Supported Providers & Endpoints @@ -189,4 +189,4 @@ AI Gateway Files are files uploaded to the gateway that are then automatically u * Custom `batch_size`, `batch_interval`, `max_retries` (Q3 2025) * Real‑time progress webhooks -* UI for canceling or pausing jobs +* UI for cancelling or pausing jobs diff --git a/aigw/product/ai-gateway/beta-features.mdx b/aigw/product/ai-gateway/beta-features.mdx index e5092118..d2ca9d1a 100644 --- a/aigw/product/ai-gateway/beta-features.mdx +++ b/aigw/product/ai-gateway/beta-features.mdx @@ -30,8 +30,8 @@ Disables the automatic transformation of `/v1/messages` requests to Chat Complet **When to use it** - You want to use the Responses API format internally while still calling the `/v1/messages` endpoint -- You need consistency with Responses API behavior across all your requests -- You're migrating from Messages API to Responses API and want to test the behavior without changing endpoints +- You need consistency with Responses API behaviour across all your requests +- You're migrating from Messages API to Responses API and want to test the behaviour without changing endpoints **How it works** diff --git a/aigw/product/ai-gateway/cache-simple-and-semantic.mdx b/aigw/product/ai-gateway/cache-simple-and-semantic.mdx index 9342674b..cd7ebd89 100644 --- a/aigw/product/ai-gateway/cache-simple-and-semantic.mdx +++ b/aigw/product/ai-gateway/cache-simple-and-semantic.mdx @@ -153,7 +153,7 @@ To enable semantic caching on a self-hosted AI Gateway, configure the embedding - **`VECTOR_STORE_COLLECTION_NAME`** — Omit this; it is not used for Pinecone. - **`VECTOR_STORE_ADDRESS`** — Set to your **Pinecone index name** (not a generic host string). - **`SEMANTIC_CACHE_EMBEDDING_DIMENSIONS`** — Must match the **dimension** configured on the index (same as your embedding vectors). - - In the Pinecone console, create or use an index with **cosine** as the similarity metric so it matches the AI Gateway's semantic cache behavior. + - In the Pinecone console, create or use an index with **cosine** as the similarity metric so it matches the AI Gateway's semantic cache behaviour. diff --git a/aigw/product/ai-gateway/chat-completions.mdx b/aigw/product/ai-gateway/chat-completions.mdx index 5531e4b5..7617b95a 100644 --- a/aigw/product/ai-gateway/chat-completions.mdx +++ b/aigw/product/ai-gateway/chat-completions.mdx @@ -234,13 +234,13 @@ curl https://aigw.portkey.ai/v1/chat/completions \ ## API Reference -- [Chat Completions](/api-reference/inference-api/chat) -- `POST /v1/chat/completions` +- [Chat Completions](/aigw/api-reference/chat/create-chat-completion) -- `POST /v1/chat/completions` OpenAI specification - + AI Gateway Chat Completions reference diff --git a/aigw/product/ai-gateway/circuit-breaker.mdx b/aigw/product/ai-gateway/circuit-breaker.mdx index aee826d1..63739848 100644 --- a/aigw/product/ai-gateway/circuit-breaker.mdx +++ b/aigw/product/ai-gateway/circuit-breaker.mdx @@ -65,7 +65,7 @@ Circuit breaker tracks per strategy path: **Circuit closes (CLOSED)** automatically after `cooldown_interval` passes. -## Runtime Behavior +## Runtime Behaviour ```mermaid flowchart TD diff --git a/aigw/product/ai-gateway/configs.mdx b/aigw/product/ai-gateway/configs.mdx index b534c5ee..0bb24a78 100644 --- a/aigw/product/ai-gateway/configs.mdx +++ b/aigw/product/ai-gateway/configs.mdx @@ -130,7 +130,7 @@ If you specify a config in a request (via headers or SDK parameters), it will ov Each config target can shape the request body before it reaches the upstream provider with three parameter fields: -| Field | Behavior | +| Field | Behaviour | |---|---| | `default_params` | Injects parameters into the request body **only when not already present**. Unlike `override_params`, this respects values set by the client. | | `override_params` | Always overwrites the matching field on the request body. | diff --git a/aigw/product/ai-gateway/custom-hosts.mdx b/aigw/product/ai-gateway/custom-hosts.mdx index 9fd769f9..1e6436eb 100644 --- a/aigw/product/ai-gateway/custom-hosts.mdx +++ b/aigw/product/ai-gateway/custom-hosts.mdx @@ -219,9 +219,9 @@ The AI Gateway maintains a trusted hosts allowlist via the `TRUSTED_CUSTOM_HOSTS `TRUSTED_CUSTOM_HOSTS` is available only on [self-hosted](/aigw/self-hosting/hybrid-deployments/architecture) hybrid and air-gapped enterprise deployments of the AI Gateway. -### Default behavior +### Default behaviour -| Environment | `TRUSTED_CUSTOM_HOSTS` unset | Behavior | +| Environment | `TRUSTED_CUSTOM_HOSTS` unset | Behaviour | |-------------|------------------------------|----------| | **Non-production** (`NODE_ENV` ≠ `production`) | Yes | Defaults to `localhost`, `127.0.0.1`, `::1`, and `host.docker.internal` | | **Production** (`NODE_ENV` = `production`) | Yes | **Empty allowlist** — localhost and private IPs are blocked until you opt in | @@ -278,7 +278,7 @@ Even for trusted hosts, the AI Gateway enforces: ## Common scenarios -| Scenario | Default behavior | Action needed | +| Scenario | Default behaviour | Action needed | |----------|-----------------|---------------| | **Local development** (e.g., Ollama on `localhost`) | Allowed in non-production — `localhost` and `127.0.0.1` are trusted by default | None in dev. In production, add to `TRUSTED_CUSTOM_HOSTS`. See the [Ollama integration guide](/aigw/integrations/llms/ollama). | | **Docker containers** (`host.docker.internal`) | Allowed in non-production — trusted by default | In production, add to `TRUSTED_CUSTOM_HOSTS` | diff --git a/aigw/product/ai-gateway/load-balancing.mdx b/aigw/product/ai-gateway/load-balancing.mdx index 73c7348a..760d2a48 100644 --- a/aigw/product/ai-gateway/load-balancing.mdx +++ b/aigw/product/ai-gateway/load-balancing.mdx @@ -102,7 +102,7 @@ The `@provider-slug/model-name` format automatically routes to the correct provi Sticky load balancing ensures that requests with the same identifier are consistently routed to the same target. This is useful for: - Maintaining conversation context across multiple requests -- Ensuring consistent model behavior for A/B testing +- Ensuring consistent model behaviour for A/B testing - Session-based routing for user-specific experiences ### Configuration @@ -136,7 +136,7 @@ Add `sticky` to your load balancing strategy: | Parameter | Type | Description | |-----------|------|-------------| -| `enabled` | boolean | Turns sticky routing on or off. Must be `true` to activate sticky behavior. If omitted or `false`, sticky routing is disabled and normal weighted load balancing is used. | +| `enabled` | boolean | Turns sticky routing on or off. Must be `true` to activate sticky behaviour. If omitted or `false`, sticky routing is disabled and normal weighted load balancing is used. | | `hash_fields` | array | Fields to use for generating the sticky session identifier. Supports dot notation for nested fields (e.g., `metadata.user_id`, `metadata.session_id`) | | `ttl` | number | Time-to-live in seconds for the sticky session. After this period, a new target may be selected. Default: 3600 (1 hour) | diff --git a/aigw/product/ai-gateway/nitro-mode.mdx b/aigw/product/ai-gateway/nitro-mode.mdx index 1c76b33e..70334e94 100644 --- a/aigw/product/ai-gateway/nitro-mode.mdx +++ b/aigw/product/ai-gateway/nitro-mode.mdx @@ -4,7 +4,7 @@ description: "Forward the request body directly to the provider without any tran --- -Nitro mode is currently in **beta**. Behavior may change based on feedback. Contact the Prisma AIRS AI Gateway account team to get access. +Nitro mode is currently in **beta**. Behaviour may change based on feedback. Contact the Prisma AIRS AI Gateway account team to get access. The AI Gateway supports **Nitro mode** through the `x-portkey-nitro-mode` header. When enabled, the gateway forwards your request body directly to the upstream provider **without reading or transforming it**. This is useful when your request and response structure already matches the provider's expected format and you don't need the gateway to modify the payload. @@ -206,4 +206,4 @@ Caching requires reading the full request body to build cache keys. If your conf If your configuration violates any of the constraints above (except caching, which is silently ignored), the gateway returns an HTTP 4xx error describing the issue. -Fix the configuration as indicated in the error message, or remove the `x-portkey-nitro-mode` header to use standard gateway behavior. +Fix the configuration as indicated in the error message, or remove the `x-portkey-nitro-mode` header to use standard gateway behaviour. diff --git a/aigw/product/ai-gateway/responses-api.mdx b/aigw/product/ai-gateway/responses-api.mdx index ebf250ef..752188e9 100644 --- a/aigw/product/ai-gateway/responses-api.mdx +++ b/aigw/product/ai-gateway/responses-api.mdx @@ -133,7 +133,7 @@ Supported roles: `user`, `assistant`, `developer` (maps to system), `system`, an ### Generation Parameters -Control generation behavior with optional parameters: +Control generation behaviour with optional parameters: @@ -195,7 +195,7 @@ curl https://aigw.portkey.ai/v1/responses \ Control tool usage with `tool_choice`: -| Value | Behavior | +| Value | Behaviour | |-------|----------| | `"auto"` | Model decides whether to call a tool (default) | | `"none"` | Model will not call any tools | @@ -400,16 +400,16 @@ Complete list of parameters supported by the Responses API and how they map inte ### API Endpoints -- [Create a Response](/api-reference/inference-api/responses/responses) -- `POST /v1/responses` -- [Retrieve a Response](/api-reference/inference-api/responses/retrieve-response) -- `GET /v1/responses/{response_id}` -- [Delete a Response](/api-reference/inference-api/responses/delete-response) -- `DELETE /v1/responses/{response_id}` -- [List Input Items](/api-reference/inference-api/responses/retrieve-inputs) -- `GET /v1/responses/{response_id}/input_items` +- [Create a Response](/aigw/api-reference/responses/create-response) -- `POST /v1/responses` +- [Retrieve a Response](/aigw/api-reference/responses/get-response) -- `GET /v1/responses/{response_id}` +- [Delete a Response](/aigw/api-reference/responses/delete-response) -- `DELETE /v1/responses/{response_id}` +- [List Input Items](/aigw/api-reference/responses/list-input-items) -- `GET /v1/responses/{response_id}/input_items` Full specification - + Responses API reference diff --git a/aigw/product/ai-gateway/universal-api.mdx b/aigw/product/ai-gateway/universal-api.mdx index 35da6670..55ab8dfc 100644 --- a/aigw/product/ai-gateway/universal-api.mdx +++ b/aigw/product/ai-gateway/universal-api.mdx @@ -384,22 +384,22 @@ The AI Gateway blocks requests to private and reserved IP ranges by default to p - **[Chat Completions](/aigw/product/ai-gateway/chat-completions)** — OpenAI-compatible text generation with streaming, function calling, and multimodal inputs - **[Responses API](/aigw/product/ai-gateway/responses-api)** — Next-gen format with built-in tool use and reasoning - **[Messages API](/aigw/product/ai-gateway/messages-api)** — Anthropic-compatible endpoint across all providers -- **[Images](/api-reference/inference-api/images/create-image)** — Generate, edit, and create image variations (DALL-E, gpt-image-1, Stable Diffusion) -- **[Audio](/api-reference/inference-api/audio/create-speech)** — Speech-to-text and text-to-speech -- **[OCR](/api-reference/inference-api/ocr)** — Extract text and structured content from PDFs and images +- **[Images](/aigw/api-reference/images/create-image)** — Generate, edit, and create image variations (DALL-E, gpt-image-1, Stable Diffusion) +- **[Audio](/aigw/api-reference/audio/create-speech)** — Speech-to-text and text-to-speech +- **[OCR](/aigw/api-reference/ocr/create-ocr)** — Extract text and structured content from PDFs and images ### Advanced Capabilities - **[Fine-tuning](/aigw/product/ai-gateway/fine-tuning)** — Customize models on specific datasets - **[Batch Processing](/aigw/product/ai-gateway/batches)** — Process large request volumes efficiently - **[Files](/aigw/product/ai-gateway/files)** — Upload and manage files for fine-tuning and batch operations -- **[Moderations](/api-reference/inference-api/moderations)** — Content safety and compliance checks +- **[Moderations](/aigw/api-reference/moderations/create-moderation)** — Content safety and compliance checks ### Additional Endpoints - **Gateway to Other APIs** — Proxy requests to any provider endpoint -- **[Assistants API](/api-reference/inference-api/assistants-api/assistants/create-assistant)** — OpenAI Assistants with persistent threads -- **[Completions](/api-reference/inference-api/completions)** — Legacy text completion endpoint +- **[Assistants API](/aigw/api-reference/assistants/create-assistant)** — OpenAI Assistants with persistent threads +- **[Completions](/aigw/api-reference/completions/create-completion)** — Legacy text completion endpoint ### Multimodal Capabilities diff --git a/aigw/product/enterprise-offering/budget-policies.mdx b/aigw/product/enterprise-offering/budget-policies.mdx index b4ee0ace..77f34a65 100644 --- a/aigw/product/enterprise-offering/budget-policies.mdx +++ b/aigw/product/enterprise-offering/budget-policies.mdx @@ -816,7 +816,7 @@ GET https://aigw.portkey.ai/v1/policies/usage-limits/{policyId}/entities?page_si **Headers:** ``` -Authorization: Bearer YOUR_ADMIN_API_KEY +Authorization: Bearer $ACCESS_TOKEN ``` This returns entities with their `id`, `value_key` (e.g. `metadata._user:username`), and `current_usage`. Use the `search` query param to filter results. @@ -831,7 +831,7 @@ PUT https://aigw.portkey.ai/v1/policies/usage-limits/{policyId}/entities/{entity **Headers:** ``` -Authorization: Bearer YOUR_ADMIN_API_KEY +Authorization: Bearer $ACCESS_TOKEN ``` This resets the entity's usage counter to zero, allowing it to consume the full credit limit again. @@ -859,17 +859,17 @@ This resets the entity's usage counter to zero, allowing it to consume the full For detailed API documentation, see the following endpoints: ### Usage Limits Policies - - - - - - - + + + + + + + ### Rate Limits Policies - - - - - + + + + + diff --git a/aigw/product/enterprise-offering/org-management/api-key-rotation.mdx b/aigw/product/enterprise-offering/org-management/api-key-rotation.mdx index f8eac396..31ca558a 100644 --- a/aigw/product/enterprise-offering/org-management/api-key-rotation.mdx +++ b/aigw/product/enterprise-offering/org-management/api-key-rotation.mdx @@ -78,7 +78,7 @@ flowchart TD Triggered via API call. The caller receives the new key in the response. -**Endpoint:** [`POST /v2/api-keys/:apiKeyId/rotate`](/api-reference/admin-api/control-plane/api-keys/rotate-api-key) +**Endpoint:** [`POST /v2/api-keys/:apiKeyId/rotate`](/aigw/api-reference/api-keys/post-api-keys-by-id-rotate) **Request body (optional):** @@ -119,7 +119,7 @@ Automatic rotation is handled by a background worker that runs on a recurring sc ## Rotation Policy Configuration -A rotation policy can be attached to any API key at [**creation**](/api-reference/admin-api/control-plane/api-keys/create-api-key) or via [**update**](/api-reference/admin-api/control-plane/api-keys/update-api-key). +A rotation policy can be attached to any API key at [**creation**](/aigw/api-reference/api-keys/post-api-keys-by-sub-type) or via [**update**](/aigw/api-reference/api-keys/put-api-keys-by-id). | Field | Type | Required | Description | |---|---|---|---| @@ -180,7 +180,7 @@ These notifications are triggered during automatic rotation runs.: ## Reading Rotation State -[`GET /v2/api-keys/:apiKeyId`](/api-reference/admin-api/control-plane/api-keys/retrieve-an-api-key) returns the rotation policy alongside the key details: +[`GET /v2/api-keys/:apiKeyId`](/aigw/api-reference/api-keys/get-api-keys-by-id) returns the rotation policy alongside the key details: ```json { @@ -249,10 +249,10 @@ Every rotation (manual and automatic) produces an audit log entry containing: ## Related - - - - + + + + diff --git a/aigw/product/enterprise-offering/org-management/api-keys-authn-and-authz.mdx b/aigw/product/enterprise-offering/org-management/api-keys-authn-and-authz.mdx index b2d39481..4cb2710c 100644 --- a/aigw/product/enterprise-offering/org-management/api-keys-authn-and-authz.mdx +++ b/aigw/product/enterprise-offering/org-management/api-keys-authn-and-authz.mdx @@ -5,6 +5,12 @@ description: "Discover how Admin and Workspace API Keys are used to manage acces ## API Keys + +API keys authenticate requests to the **inference API**. They are not accepted on the Admin API, which is authorised with a Strata Cloud Manager access token issued to a service account. See [Admin API Authentication](/aigw/api-reference/admin-api/authentication). + +The scopes below describe what an API key may do on the inference path, and what the Admin API may do on that key's behalf. + + The AI Gateway uses two types of API keys to manage access to resources and operations: **Admin API Keys** and **Workspace API Keys**. These keys play crucial roles in authenticating and authorizing various operations within your [organisation](/aigw/product/enterprise-offering/org-management/organizations) and [workspaces](/aigw/product/enterprise-offering/org-management/workspaces). ### Admin API Keys @@ -457,6 +463,7 @@ Both types of API keys play important roles in the AI Gateway's security model, ### Related Topics + diff --git a/aigw/product/enterprise-offering/org-management/directory-sync/cie-directory-sync.mdx b/aigw/product/enterprise-offering/org-management/directory-sync/cie-directory-sync.mdx index d87ac534..9601f4ab 100644 --- a/aigw/product/enterprise-offering/org-management/directory-sync/cie-directory-sync.mdx +++ b/aigw/product/enterprise-offering/org-management/directory-sync/cie-directory-sync.mdx @@ -5,17 +5,17 @@ description: "Sync users and groups from Palo Alto Networks Cloud Identity Engin # CIE Directory Sync -CIE (Cloud Identity Engine) Directory Sync allows you to pull users and groups from your organization's identity provider directories — such as **Entra ID (Azure AD)**, **Okta**, or **On-Premises Active Directory** — into SCM via Palo Alto's Cloud Identity Engine. Once synced, you can map CIE groups to SCM's AI Gateway workspaces so that users are **automatically provisioned** into the correct workspaces. +CIE (Cloud Identity Engine) Directory Sync allows you to pull users and groups from your organisation's identity provider directories — such as **Entra ID (Azure AD)**, **Okta**, or **On-Premises Active Directory** — into SCM via Palo Alto's Cloud Identity Engine. Once synced, you can map CIE groups to SCM's AI Gateway workspaces so that users are **automatically provisioned** into the correct workspaces. --- ## Overview -CIE Directory Sync is available for organizations running in **SCM (Strata Cloud Manager)**. It replaces the need for manual user provisioning or standalone SCIM integration by leveraging CIE as the centralized identity source. +CIE Directory Sync is available for organisations running in **SCM (Strata Cloud Manager)**. It replaces the need for manual user provisioning or standalone SCIM integration by leveraging CIE as the centralized identity source. ### How It Works -1. **CIE aggregates directories** — Your organization's identity providers (Entra ID, Okta, on-prem AD) are connected to CIE via the Strata Cloud Manager. CIE syncs and caches user/group data from these directories. +1. **CIE aggregates directories** — Your organisation's identity providers (Entra ID, Okta, on-prem AD) are connected to CIE via the Strata Cloud Manager. CIE syncs and caches user/group data from these directories. 2. **Admin maps groups to workspaces** — An admin selects which CIE directory to connect, then maps CIE groups to AI Gateway workspaces. 3. **Users are auto-provisioned** — Background sync periodically pulls group membership changes from CIE and provisions/deprovisions users in the mapped workspaces automatically. @@ -24,7 +24,7 @@ CIE Directory Sync is available for organizations running in **SCM (Strata Cloud | Concept | Description | |---------|-------------| | **Domain (Connected Directory)** | An identity provider directory synced into CIE. Each domain represents a separate directory source. | -| **Tenant ID** | The CIE tenant identifier for your organization, auto-provisioned during Onboarding. You never need to enter this manually. | +| **Tenant ID** | The CIE tenant identifier for your organisation, auto-provisioned during Onboarding. You never need to enter this manually. | | **Group** | A directory group from CIE (e.g., a security group). Groups contain users that can be mapped to workspaces. | | **Group-Workspace Mapping** | A 1:1 link between a CIE group and an AI Gateway workspace. All members of the mapped group are automatically provisioned into that workspace. | | **User Identity Attribute** | The CIE user attribute used as the email address — either **UPN (User Principal Name)** or **Mail (Primary Email)**. | @@ -36,9 +36,9 @@ CIE Directory Sync is available for organizations running in **SCM (Strata Cloud Before configuring CIE Directory Sync in SCM's AI Gateway, ensure the following: -1. **CIE is provisioned for your organization** — Your Strata Cloud Manager tenant must have CIE activated with a Directory Sync instance. This is set up during Onboarding. +1. **CIE is provisioned for your organisation** — Your Strata Cloud Manager tenant must have CIE activated with a Directory Sync instance. This is set up during Onboarding. 2. **At least one directory is connected in CIE** — Navigate to CIE and verify that at least one directory (Entra ID, Okta, or On-Premises) has been added and has a successful sync status. -3. **You have SCM admin access** — Only organization admins can configure Directory Sync in SCM's AI Gateway. +3. **You have SCM admin access** — Only organisation admins can configure Directory Sync in SCM's AI Gateway. CIE Directory Sync is only available for SCM Tenants. It is not available in standalone deployments. For non-SCM deployments, use [SCIM Provisioning](/aigw/product/enterprise-offering/org-management/scim/scim) instead. @@ -193,9 +193,9 @@ Once Directory Sync is configured and group mappings are in place, users from CI ### Viewing Workspaces -Navigate to **AI Gateway → Workspace Control** to see all workspaces in your organization. +Navigate to **AI Gateway → Workspace Control** to see all workspaces in your organisation. -![Workspace Control — list of all workspaces in the organization](/images/directory-sync/workspace-control-list.jpg) +![Workspace Control — list of all workspaces in the organisation](/images/directory-sync/workspace-control-list.jpg) This page shows all workspaces along with their slug, creation date, and last update time. Workspaces that have CIE groups mapped to them will have directory-provisioned members automatically added. @@ -315,7 +315,7 @@ To **re-enable** sync after disabling: | **No domains shown** in Connected Directory dropdown | CIE not provisioned for this org, or no directories added in CIE | Ensure CIE is activated for your SCM tenant. Add directories in CIE. | | **Users not provisioned** after mapping | Selected User Identity Attribute is missing for those users | Check CIE to confirm users have the UPN or Mail attribute populated. Switch attribute if needed. | | **Users not removed** after deleting mapping | User belongs to another group also mapped to the same workspace | This is by design — users who have access through another mapping are not removed. | -| **Delta sync falling back to full** | CIE cache was rebuilt, or cursor expired | This is expected behavior. CIE periodically rebuilds its cache, which triggers a full resync. This acts as a self-healing mechanism. | +| **Delta sync falling back to full** | CIE cache was rebuilt, or cursor expired | This is expected behaviour. CIE periodically rebuilds its cache, which triggers a full resync. This acts as a self-healing mechanism. | --- diff --git a/aigw/product/enterprise-offering/secret-references.mdx b/aigw/product/enterprise-offering/secret-references.mdx index 106a36ba..c50b428f 100644 --- a/aigw/product/enterprise-offering/secret-references.mdx +++ b/aigw/product/enterprise-offering/secret-references.mdx @@ -569,7 +569,7 @@ This setup: - If the entity is workspace-scoped, the secret reference must be accessible to that workspace (either `allow_all_workspaces: true` or explicitly mapped). - On create, `target_field` values with the `configurations.` prefix are auto-normalized — you can pass just the field name without the prefix and it will be prepended. -### Behavior +### Behaviour - At gateway runtime, mapped fields are resolved from the external secret manager using the referenced secret reference's `auth_config`, `secret_path`, and the mapping's `secret_key` (or the secret reference's default `secret_key`). - When a `target_field` of `key` is mapped, the `key` field on the entity becomes optional during creation. @@ -600,8 +600,8 @@ When you retrieve a secret reference via the API, sensitive `auth_config` fields ## API Reference - - - - - + + + + + diff --git a/aigw/product/guardrails.mdx b/aigw/product/guardrails.mdx index 977dc872..08ff5108 100644 --- a/aigw/product/guardrails.mdx +++ b/aigw/product/guardrails.mdx @@ -10,7 +10,7 @@ This feature is available on all plans. * **Enterprise**: Access to **all** Guardrails plus `custom` Guardrails. -LLMs are brittle - not just in API uptimes or their inexplicable `400`/`500` errors, but also in their core behavior. You can get a response with a `200` status code that completely errors out for your app's pipeline due to mismatched output. With the AI Gateway's Guardrails, we now help you enforce LLM behavior in real-time with our _Guardrails on the Gateway_ pattern. +LLMs are brittle - not just in API uptimes or their inexplicable `400`/`500` errors, but also in their core behaviour. You can get a response with a `200` status code that completely errors out for your app's pipeline due to mismatched output. With the AI Gateway's Guardrails, we now help you enforce LLM behaviour in real-time with our _Guardrails on the Gateway_ pattern. Use the AI Gateway's Guardrails to verify your LLM inputs AND outputs, adhering to your specifed checks. Since Guardrails are built into the gateway itself, you can orchestrate your request - with actions ranging from _denying the request_, _logging the guardrail result_, _creating an evals dataset_, _falling back to another LLM or prompt_, _retrying the request_, and more. @@ -780,7 +780,7 @@ If you already have a custom guardrail pipeline where you send your inputs/outpu ## Examples of When to Deny Requests with Guardrails -1. **Prompt Injection Checks**: Preventing inputs that could alter the behavior of the AI model or manipulate its responses. +1. **Prompt Injection Checks**: Preventing inputs that could alter the behaviour of the AI model or manipulate its responses. 2. **Moderation Checks**: Ensuring responses do not contain offensive, harmful, or inappropriate content. 3. **Compliance Checks**: Verifying that inputs and outputs comply with regulatory requirements or organisational policies. 4. **Security Checks**: Blocking requests that contain potentially harmful content, such as SQL injection attempts or cross-site scripting (XSS) payloads. diff --git a/aigw/product/guardrails/capabilities.mdx b/aigw/product/guardrails/capabilities.mdx index aa4932ff..29540a5c 100644 --- a/aigw/product/guardrails/capabilities.mdx +++ b/aigw/product/guardrails/capabilities.mdx @@ -45,7 +45,7 @@ The following Inference API endpoints **do not run guardrails**: ## Sync vs. Async -| Mode | `async` | Behavior | Latency | +| Mode | `async` | Behaviour | Latency | |:---|:---|:---|:---| | **Async** (default) | `true` | Runs in parallel with the LLM call. Results are logged only — no effect on the response. | None | | **Sync** | `false` | Runs before forwarding the request (input) or before returning the response (output). Can deny or modify. | Adds guardrail check latency | @@ -53,7 +53,7 @@ The following Inference API endpoints **do not run guardrails**: Use async to observe and log without blocking. Use sync when the guardrail result must gate the request. - With `async=false`, the AI Gateway returns `246` (failed, but allowed) or `446` (failed, denied) instead of the standard provider status codes. See [Guardrail behavior on the gateway](/aigw/product/guardrails#guardrail-behavior-on-the-gateway). + With `async=false`, the AI Gateway returns `246` (failed, but allowed) or `446` (failed, denied) instead of the standard provider status codes. See [Guardrail behaviour on the gateway](/aigw/product/guardrails#guardrail-behaviour-on-the-gateway). --- diff --git a/aigw/product/guardrails/guardrails-for-batches.mdx b/aigw/product/guardrails/guardrails-for-batches.mdx index ef88138a..5bcee90f 100644 --- a/aigw/product/guardrails/guardrails-for-batches.mdx +++ b/aigw/product/guardrails/guardrails-for-batches.mdx @@ -10,7 +10,7 @@ Data Service >= `1.8.0` Imagine usecases where you want to redact PII, remove some requests and redact responses when making a bulk batch inference call to providers like OpenAI, Azure OpenAI, Bedrock, Vertex AI, etc. -The AI Gateway supports running [guardrails](/aigw/product/guardrails) on [provider batch requests](/api-reference/inference-api/batch/create-batch). You pass guardrails through a [config](/aigw/product/ai-gateway/configs), +The AI Gateway supports running [guardrails](/aigw/product/guardrails) on [provider batch requests](/aigw/api-reference/batch/create-batch). You pass guardrails through a [config](/aigw/product/ai-gateway/configs), and the AI Gateway applies them asynchronously — pre-processing inputs before sending to the provider, and post-processing outputs after the provider completes. ```mermaid diff --git a/aigw/product/guardrails/pii-redaction.mdx b/aigw/product/guardrails/pii-redaction.mdx index da14ffef..e7842427 100644 --- a/aigw/product/guardrails/pii-redaction.mdx +++ b/aigw/product/guardrails/pii-redaction.mdx @@ -155,7 +155,7 @@ This feature can help organisations meet various compliance requirements by: If you experience issues: 1. Verify the feature is enabled in your guardrails configuration 2. Check the `transformed` flag and `check_results` for specific transformation details -3. Review logs for any error messages or unexpected behavior +3. Review logs for any error messages or unexpected behaviour 4. For custom regex patterns, validate the regex syntax and test with sample data 5. [Contact us](mailto:support@portkey.ai) for additional assistance diff --git a/aigw/product/mcp-gateway/advanced-configuration.mdx b/aigw/product/mcp-gateway/advanced-configuration.mdx index 13f8a2e9..4656774a 100644 --- a/aigw/product/mcp-gateway/advanced-configuration.mdx +++ b/aigw/product/mcp-gateway/advanced-configuration.mdx @@ -12,7 +12,7 @@ It unlocks everything the UI fields don't cover: - **Forward runtime headers** like trace IDs or tenant context to the upstream server - **Tell the MCP server who's calling** by forwarding authenticated user identity - **Validate tokens at the edge** before requests ever reach the upstream server -- **Customize OAuth behavior** when the upstream server's defaults don't work +- **Customize OAuth behaviour** when the upstream server's defaults don't work - **Bring your own auth** with an external identity provider for enterprise SSO --- diff --git a/aigw/product/mcp-gateway/authentication/forwarding-headers.mdx b/aigw/product/mcp-gateway/authentication/forwarding-headers.mdx index 2a1506be..62740074 100644 --- a/aigw/product/mcp-gateway/authentication/forwarding-headers.mdx +++ b/aigw/product/mcp-gateway/authentication/forwarding-headers.mdx @@ -40,7 +40,7 @@ Other headers from the agent request are dropped. ### Allowlist mode -The same behavior with explicit configuration. Entries in the `headers` array can be: +The same behaviour with explicit configuration. Entries in the `headers` array can be: - **A string** — forwards the header as-is. - **An object** `{ "from": "", "to": "" }` — forwards the header with its name changed. @@ -104,7 +104,7 @@ These headers are **never forwarded** regardless of your configuration: | `x-user-claims` | Identity spoofing prevention | | `x-user-jwt` | Identity spoofing prevention | -In `all-except` mode, these are automatically added to the blocklist. You cannot override this behavior. +In `all-except` mode, these are automatically added to the blocklist. You cannot override this behaviour. ### Header mapping validation diff --git a/aigw/product/mcp-gateway/authentication/jwt.mdx b/aigw/product/mcp-gateway/authentication/jwt.mdx index 1f35ae71..ef80dfa8 100644 --- a/aigw/product/mcp-gateway/authentication/jwt.mdx +++ b/aigw/product/mcp-gateway/authentication/jwt.mdx @@ -251,7 +251,7 @@ If a specified claim exists in both the header and payload, they must have ident --- -## Caching Behavior +## Caching Behaviour ### JWKS Caching diff --git a/aigw/product/mcp-gateway/authentication/oauth-client-metadata.mdx b/aigw/product/mcp-gateway/authentication/oauth-client-metadata.mdx index 8cca92cf..7c44b2db 100644 --- a/aigw/product/mcp-gateway/authentication/oauth-client-metadata.mdx +++ b/aigw/product/mcp-gateway/authentication/oauth-client-metadata.mdx @@ -220,7 +220,7 @@ Providing a `client_id` disables Dynamic Client Registration entirely for that i Set the OAuth App's **Authorization callback URL** to the AI Gateway's callback, `/oauth/upstream-callback`. On the AI Gateway's managed gateway that is `https://aigw.portkey.ai/m/oauth/upstream-callback`; on a self-hosted or hybrid deployment, use your own gateway host (the value of `MCP_GATEWAY_BASE_URL`). This URL is always controlled by the gateway, so a `redirect_uri` set in your OAuth metadata is ignored. -See the [GitHub MCP server guide](/aigw/integrations/mcp-servers/github-mcp-server#connect-via-portkey-mcp-gateway) for complete setup instructions. +See the [GitHub MCP server guide](/aigw/integrations/mcp-servers/github-mcp-server#connect-via-prisma-airs-ai-gateway-mcp-gateway) for complete setup instructions. --- diff --git a/aigw/product/mcp-gateway/external-mcp-servers.mdx b/aigw/product/mcp-gateway/external-mcp-servers.mdx index 4eaa25b1..1a935ccf 100644 --- a/aigw/product/mcp-gateway/external-mcp-servers.mdx +++ b/aigw/product/mcp-gateway/external-mcp-servers.mdx @@ -83,7 +83,7 @@ Some services (like **GitHub**) don't support DCR. For these: 2. Set the callback URL to `https://aigw.portkey.ai/m/oauth/upstream-callback` 3. Add the `client_id`, `client_secret`, and `redirect_uri` in the AI Gateway's **Advanced Configuration** -See the [GitHub MCP server guide](/aigw/integrations/mcp-servers/github-mcp-server#connect-via-portkey-mcp-gateway) for a detailed walkthrough. +See the [GitHub MCP server guide](/aigw/integrations/mcp-servers/github-mcp-server#connect-via-prisma-airs-ai-gateway-mcp-gateway) for a detailed walkthrough. ### Provision Access diff --git a/aigw/product/mcp-gateway/guardrails.mdx b/aigw/product/mcp-gateway/guardrails.mdx index a983cf15..bb9856a8 100644 --- a/aigw/product/mcp-gateway/guardrails.mdx +++ b/aigw/product/mcp-gateway/guardrails.mdx @@ -63,7 +63,7 @@ Every guardrail has a `target` field: | Target | Description | |--------|-------------| -| `llm` | Applied to LLM API requests (default, existing behavior) | +| `llm` | Applied to LLM API requests (default, existing behaviour) | | `mcp_tools` | Applied to MCP tool calls | When creating a guardrail for MCP, set `target` to `"mcp_tools"`. diff --git a/aigw/product/mcp-gateway/rate-limits.mdx b/aigw/product/mcp-gateway/rate-limits.mdx index 8934c068..988d4df4 100644 --- a/aigw/product/mcp-gateway/rate-limits.mdx +++ b/aigw/product/mcp-gateway/rate-limits.mdx @@ -293,7 +293,7 @@ When a rate limit is exceeded, the AI Gateway returns a **429 Too Many Requests* Full policy reference with conditions, group-by, and validation rules. - + API reference for creating and managing rate limit policies. diff --git a/aigw/product/model-catalog/custom-models.mdx b/aigw/product/model-catalog/custom-models.mdx index 3e6d8ec7..93199d1e 100644 --- a/aigw/product/model-catalog/custom-models.mdx +++ b/aigw/product/model-catalog/custom-models.mdx @@ -58,7 +58,7 @@ Each integration model can carry its own routing config — useful when several - Model-level `custom_headers` override integration-level `custom_headers` for that model. - A request-time / header-level `custom_host` still takes priority over the model-level value. -Configure via the [Update Model Access](/api-reference/admin-api/control-plane/integrations/models/update-model-access) API (`models[].configurations`), or set the fields when editing the model in Model Provisioning. +Configure via the [Update Model Access](/aigw/api-reference/integrations/models/put-integrations-by-slug-models) API (`models[].configurations`), or set the fields when editing the model in Model Provisioning. ```bash curl -X PUT "https://aigw.portkey.ai/v1/integrations/YOUR_INTEGRATION_SLUG/models" \ diff --git a/aigw/product/model-catalog/pricing-adjustments.mdx b/aigw/product/model-catalog/pricing-adjustments.mdx index 73f4dda9..7dfb247b 100644 --- a/aigw/product/model-catalog/pricing-adjustments.mdx +++ b/aigw/product/model-catalog/pricing-adjustments.mdx @@ -116,7 +116,7 @@ Models without their own `pricing_adjustments` continue to use the Integration-l ## Setting Pricing Adjustments via the API -You can also configure Pricing Adjustments programmatically through the [Integrations API](/api-reference/admin-api/control-plane/integrations/create-integration). Pass `pricing_adjustments` when creating or updating an Integration. Send `null` to clear all adjustments. +You can also configure Pricing Adjustments programmatically through the [Integrations API](/aigw/api-reference/integrations/post-integrations). Pass `pricing_adjustments` when creating or updating an Integration. Send `null` to clear all adjustments. diff --git a/aigw/product/observability/analytics.mdx b/aigw/product/observability/analytics.mdx index 810ab0fc..68171a3e 100644 --- a/aigw/product/observability/analytics.mdx +++ b/aigw/product/observability/analytics.mdx @@ -22,7 +22,7 @@ This is a good starting point to then dive deeper. The users tab provides an overview of the user information associated with your AI Gateway requests. This data is derived from the `user` parameter in OpenAI SDK requests or the special `_user` key in the AI Gateway [metadata header](/aigw/product/observability/metadata). -The AI Gateway currently does not provide analytics on usage patterns for individual team members in your AI Gateway organisation. The users tab is designed to track end-user behavior in your application, not internal team usage. +The AI Gateway currently does not provide analytics on usage patterns for individual team members in your AI Gateway organisation. The users tab is designed to track end-user behaviour in your application, not internal team usage. ### Errors diff --git a/aigw/product/observability/logs-export.mdx b/aigw/product/observability/logs-export.mdx index fe38b3e8..e1b361b4 100644 --- a/aigw/product/observability/logs-export.mdx +++ b/aigw/product/observability/logs-export.mdx @@ -104,7 +104,7 @@ Each export includes: With your exported logs data, you can: - Generate custom reports for stakeholders - Feed data into business intelligence tools -- Identify patterns in user behavior and model performance +- Identify patterns in user behaviour and model performance You can analyze your API usage patterns, monitor performance, optimize costs, and make data-driven decisions for your business or team. diff --git a/aigw/product/observability/opentelemetry.mdx b/aigw/product/observability/opentelemetry.mdx index be1f027e..12b4cf27 100644 --- a/aigw/product/observability/opentelemetry.mdx +++ b/aigw/product/observability/opentelemetry.mdx @@ -3,7 +3,7 @@ title: OpenTelemetry for LLM Observability description: Leverage OpenTelemetry with Prisma AIRS AI Gateway for comprehensive LLM application observability, combining gateway insights with full-stack telemetry. --- -[OpenTelemetry (OTel)](https://opentelemetry.io/) is a Cloud Native Computing Foundation (CNCF) open-source framework. It provides a standardized way to collect, process, and export telemetry data (traces, metrics, and logs) from your applications. This is vital for monitoring performance, debugging issues, and understanding complex system behavior. +[OpenTelemetry (OTel)](https://opentelemetry.io/) is a Cloud Native Computing Foundation (CNCF) open-source framework. It provides a standardized way to collect, process, and export telemetry data (traces, metrics, and logs) from your applications. This is vital for monitoring performance, debugging issues, and understanding complex system behaviour. Many popular AI development tools and SDKs, like the Vercel AI SDK, LlamaIndex, OpenLLMetry, and Logfire, utilize OpenTelemetry for observability. The AI Gateway now embraces OTel, allowing you to send telemetry data from any OTel-compatible source directly into the AI Gateway's observability platform. @@ -15,7 +15,7 @@ The AI Gateway's strength lies in its unique combination of an intelligent **LLM - **Holistic View with OpenTelemetry:** By adding an OTel endpoint, the AI Gateway now ingests traces and logs from your *entire* application stack, not just the LLM calls. Instrument your frontend, backend services, databases, and any other component with OTel, and send that data to the AI Gateway. -This combination provides an unparalleled, end-to-end view of your LLM application's performance, cost, and behavior. You can correlate application-level events with specific LLM interactions managed by the AI Gateway. +This combination provides an unparalleled, end-to-end view of your LLM application's performance, cost, and behaviour. You can correlate application-level events with specific LLM interactions managed by the AI Gateway. ## How OpenTelemetry Data Flows to the AI Gateway diff --git a/aigw/product/observability/traces.mdx b/aigw/product/observability/traces.mdx index eacdc978..2555e2a8 100644 --- a/aigw/product/observability/traces.mdx +++ b/aigw/product/observability/traces.mdx @@ -362,11 +362,11 @@ The logger endpoint supports inserting a single log as well as log array, and he ## Tracing for Gateway Features -Tracing also works very well to capture the Gateway behavior on retries, fallbacks, and other routing mechanisms on AI Gateway. +Tracing also works very well to capture the Gateway behaviour on retries, fallbacks, and other routing mechanisms on AI Gateway. The AI Gateway automatically groups all the requests that were part of a single fallback or retry config and shows the failed and succeeded requests chronologically as "spans" inside a "trace". -This is especially useful when you want to understand the total latency and behavior of your app when retry or fallbacks were triggered. +This is especially useful when you want to understand the total latency and behaviour of your app when retry or fallbacks were triggered. For more, check out the [Fallback](/aigw/product/ai-gateway/fallbacks) & [Automatic Retries](/aigw/product/ai-gateway/automatic-retries) docs. diff --git a/aigw/scripts/check_anchors.py b/aigw/scripts/check_anchors.py new file mode 100644 index 00000000..e912461b --- /dev/null +++ b/aigw/scripts/check_anchors.py @@ -0,0 +1,142 @@ +#!/usr/bin/env python3 +"""Check that every in-repo anchor link under aigw/ points at a heading that exists. + +`mint broken-links` resolves pages and stops there. It does not look at the fragment, so a +link to a section that was renamed or deleted stays green forever. Phase 2's UK rebrand +renamed headings without updating the links into them -- `#guardrail-behavior-on-the-gateway` +has been dead since then and nothing noticed. The spelling sweep on 2026-09-18 renamed five +more headings, which is what prompted writing this. + + python3 aigw/scripts/check_anchors.py # report, exit 1 on anything unlisted + python3 aigw/scripts/check_anchors.py --all # include the KNOWN baseline in output + +ON SLUGS: this does not reimplement Mintlify's slugifier, because getting it wrong invents +failures. Headings and link fragments are compared with every non-alphanumeric character +removed, so `can-t-connect` and `cant-connect`, or `reasoning--thinking` and +`reasoning-thinking`, are treated as the same anchor. Punctuation disagreements are where a +hand-rolled slugifier and a real one part company; word disagreements are where documentation +actually breaks. Only the second kind is reported. The cost is that a genuine +punctuation-only break would be missed, which is the right trade -- it is invisible to a +reader anyway, since the browser scrolls to nothing either way and they see the top of a page +that does contain their section. +""" + +import re +import sys +import pathlib +import collections + +ROOT = pathlib.Path(__file__).resolve().parents[1] +REPO = ROOT.parent + +# Anchors that are broken and are not ours to fix in passing. Each needs a destination +# decision, not a correction -- the section it wanted is gone rather than renamed. A +# shrinking baseline: an entry that stops matching anything fails the check, same as the +# endpoint check, so these cannot quietly rot once someone resolves them. +KNOWN = { + # Inherited. "3. Enterprise Governance" does not exist in the Latest version of these + # pages either, so the links arrived broken; we did not break them. 24 links, 14 pages. + "3-enterprise-governance": "section absent upstream too -- needs a destination", + "enterprise-governance": "section absent upstream too -- needs a destination", + # Sections removed by our own phases, with the links into them left behind. + "auto-instrumentation": "SDK feature removed in Phase 3", + "access-control-management": "section gone from list-of-guardrail-checks", + "supported-integrations": "section gone from opentelemetry", + "integration-approaches": "section gone from openai-agents-ts", + "processing-pdfs-with-claude": "section gone from anthropic", + "batches": "changelog links at a #batches heading the page never had", + # Ambiguous: more than one heading could be meant, so guessing would be inventing. + "portkey-batch-api-mode": "three links; 'AI Gateway Custom Batching' vs 'Provider Batch " + "API Mode' -- the text says provider-agnostic and immediate, " + "which points at Custom, but confirm before repointing", + "legacy-single-provider-with-api-key": "'Single Provider with Provider API Key' or " + "'Legacy: Single Provider with Virtual Key'", + "how-to-enable-request-tracing": "traces.mdx has 'Enabling Tracing' and 'Why Use Tracing'", + "using-reasoning_effort-parameter": "gemini.mdx does not mention reasoning_effort at all, " + "so 'Extended Thinking' is the nearest heading but " + "not the promised content -- ask product", +} + + +def key(s): + """Compare on letters and digits only -- see ON SLUGS above.""" + return re.sub(r"[^a-z0-9]", "", s.lower()) + + +def slug(heading): + h = re.sub(r"\[([^\]]*)\]\([^)]*\)", r"\1", heading) + h = re.sub(r"[`*_]", "", h).strip() + h = re.sub(r"[^a-z0-9\s-]", "", h.lower()).strip() + return re.sub(r"-+", "-", h.replace(" ", "-")) + + +def headings(path): + out, fence = set(), False + for line in path.read_text().splitlines(): + if line.strip().startswith(("```", "~~~")): + fence = not fence + continue + if fence: + continue + m = re.match(r"^(#{1,6})\s+(.*)", line) + if m: + out.add(slug(m.group(2))) + return out + + +def main(): + show_all = "--all" in sys.argv + pages = {} + for p in sorted(ROOT.rglob("*.mdx")): + if "_project" in p.parts: + continue + pages["/" + str(p.relative_to(REPO).with_suffix(""))] = headings(p) + + broken, listed = [], collections.Counter() + link = re.compile(r"\]\((/[^)#\s]*)?#([A-Za-z0-9][A-Za-z0-9_-]*)\)") + for p in sorted(ROOT.rglob("*.mdx")): + if "_project" in p.parts: + continue + here = "/" + str(p.relative_to(REPO).with_suffix("")) + for i, line in enumerate(p.read_text().splitlines(), 1): + for m in link.finditer(line): + target, anchor = m.group(1) or here, m.group(2) + if target not in pages: + continue # page-level; mint broken-links already owns this + if key(anchor) in {key(h) for h in pages[target]}: + continue + if anchor in KNOWN: + listed[anchor] += 1 + if not show_all: + continue + broken.append((f"{p.relative_to(REPO)}:{i}", f"{target}#{anchor}", + KNOWN.get(anchor))) + + real = [b for b in broken if b[2] is None] + print(f"{sum(len(v) for v in pages.values())} headings across {len(pages)} pages") + print(f"{sum(listed.values())} broken anchors carried as KNOWN " + f"({len(listed)} distinct)") + + stale = set(KNOWN) - set(listed) + if real: + print(f"\n{len(real)} anchor link(s) point at a heading that does not exist:\n") + for where, what, _ in real: + print(f" {where}\n {what}") + print("\nEither the heading was renamed -- fix the link -- or the section is gone,\n" + "in which case it needs a destination and belongs in KNOWN with a reason.") + if stale: + print("\nKNOWN entries that matched nothing. The section came back or the link is\n" + "gone -- delete the entry either way:") + for a in sorted(stale): + print(f" {a}") + if show_all and listed: + print("\nKNOWN, for reference:") + for a, n in listed.most_common(): + print(f" {n:3d} {a} -- {KNOWN[a]}") + if not real and not stale: + print("\nEvery anchor link under aigw/ resolves to a real heading.") + return 1 if (real or stale) else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/aigw/scripts/check_prose_endpoints.py b/aigw/scripts/check_prose_endpoints.py new file mode 100755 index 00000000..62b54930 --- /dev/null +++ b/aigw/scripts/check_prose_endpoints.py @@ -0,0 +1,383 @@ +#!/usr/bin/env python3 +"""Assert that every endpoint written in `aigw/` prose exists in the specification. + +The gap this closes. On 2026-09-18 the spec repo dropped the scope segment from the +API key create path: `POST /api-keys/{type}/{sub-type}` became `POST /api-keys/{sub-type}`. +`sync_api_nav.py` caught the navigation and `mint validate` caught the stale navigation +entry, because both compare structure against structure. Neither could see that a +sentence still called the level a path parameter, or that a worked `curl` still posted +to `/api-keys/organisation/service`. Those were found by reading. This finds them by +running. + +The failure mode is what makes it worth automating: a page describing an endpoint that +no longer exists still reads perfectly well. Nothing looks wrong until a reader copies +the request and gets a 404. + +Source of truth is `docs-navigation.json` from the spec repo -- the same document +`sync_api_nav.py` reads, and the reason this script needs no YAML parser. It names every +operation as a `"METHOD /path"` string, which is exactly the set being checked against. + +Matching is deliberately loose, because prose is not a specification: + + - Base prefixes are stripped. The spec's paths are relative to a server URL, so prose + writes `POST /v1/chat/completions` and `POST /ai_gw/v2/api-keys/service` for what the + spec calls `POST /chat/completions` and `POST /api-keys/{sub-type}`. + - A `{placeholder}` segment in the spec matches any single segment, so a worked example + using a real value -- `/api-keys/service`, `/configs/my-config` -- still matches. + - Trailing slashes are ignored. + +Anything genuinely outside the specification goes in ALLOWED below with a reason, never +by loosening the match. An endpoint absent from both is an error, and there are only +three honest resolutions: the prose is stale, the spec is missing something, or it +belongs in ALLOWED. + +Usage: + aigw/scripts/check_prose_endpoints.py non-zero if any endpoint is unknown + aigw/scripts/check_prose_endpoints.py --list print every endpoint found, with its match + aigw/scripts/check_prose_endpoints.py --from PATH read the fragment from a local checkout +""" + +import argparse +import json +import os +import re +import sys +import urllib.request + +FRAGMENT_URL = ("https://raw.githubusercontent.com/PaloAltoNetworks/openapi/" + "refs/heads/main/docs-navigation.json") + +AIGW = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..") + +# Server prefixes the spec factors out into `servers`, so prose carries them and the +# operation paths do not. Longest first -- `/ai_gw/admin/v2` must win over `/ai_gw`. +BASE_PREFIXES = [ + "/ai_gw/admin/v2", + "/ai_gw/v2", + "/admin/v2", + "/ai_gw", + "/v2", + "/v1", +] + +# Endpoints correctly absent from the specification, because the specification was never +# the place for them. Each needs a reason; an entry without one is indistinguishable from +# a bug that someone silenced. +ALLOWED = { + # Not the gateway. Strata Cloud Manager issues Admin API tokens from a Palo Alto + # identity host, documented in admin-api/authentication.mdx. + "POST /oauth2/access_token": "Palo Alto identity service, not the gateway", + "GET /iam/v1/access_policies": "Palo Alto IAM API, not the gateway", + "DELETE /iam/v1/access_policies/{id}": "Palo Alto IAM API, not the gateway", + "GET /oauth/authorize": "generic OAuth authorisation endpoint, not a gateway route", + + # Prompt endpoints are excluded from the specification by design -- they sit in the + # spec repo's _project/drops.yaml and its build fails if one reappears. That prose + # still documents them is a separate question, tracked in TODO.md rather than here. + "POST /prompts/{promptId}/completions": "prompt endpoints dropped upstream by design", + + # Other gateway planes. The MCP Gateway serves /m and the Agent Gateway /agent; this + # specification covers the inference and admin planes only. Real routes, wrong + # document -- see the note in TODO.md about whether they should have one. + "GET /m/v0.1/servers": "MCP Gateway registry, served from /m, not in this spec", + "POST /m/oauth/token": "MCP Gateway OAuth, served from /m, not in this spec", + "POST /agent/{agent-slug}": "Agent Gateway, served from /agent, not in this spec", + "POST /agent/{agent-slug}/.well-known/agent.json": + "Agent Gateway discovery, served from /agent, not in this spec", + + # SCIM is RFC 7644, with its own schema and its own document. + "DELETE /scim/Groups/{id}": "SCIM 2.0, governed by RFC 7644", + + # Provider-native routes the gateway proxies rather than implements. It forwards + # these upstream unchanged, so they are the provider's contract, not ours. + "POST /messages": "Anthropic-native route, proxied not implemented", + "POST /messages/batches": "Anthropic-native route, proxied not implemented", + "GET /messages/batches": "Anthropic-native route, proxied not implemented", + "GET /messages/batches/{batch_id}": "Anthropic-native route, proxied not implemented", + "POST /messages/batches/{batch_id}/cancel": "Anthropic-native route, proxied not implemented", + "POST /messages/count_tokens": "Anthropic-native route, proxied not implemented", + "POST /endpoints/{endpoint_name}/invocations": "SageMaker-native route, proxied", + "PUT /knowledgebases": "Bedrock Knowledge Base route, proxied", + "POST /knowledgebases/{knowledgeBaseId}/retrieve": "Bedrock Knowledge Base route, proxied", + "POST /listen": "Deepgram-native route, proxied", + "POST /predictions": "Replicate-native route, proxied", + "POST /projects/{project}/locations/{location}/cachedContents": "Vertex-native route, proxied", + "POST /v2/vectordb/collections/list": "Milvus-native route, proxied", + "POST /v2/videos": "Together-native route, proxied", +} + +# Endpoints that *should* be in the specification and are not. Unlike ALLOWED, every line +# here is a defect someone owns upstream -- the docs describe a real feature the spec does +# not admit exists, so nothing generates a reference page for it and nothing can verify +# the prose. Kept as a baseline that may only shrink: a new absence fails the check, and +# an entry that starts matching nothing also fails, so these get deleted when they land. +KNOWN_GAPS = { + "POST /logs/exports": "Logs Export API absent from the spec", + "GET /logs/exports/field-restrictions": "Logs Export API absent from the spec", + "GET /logs/exports/{id}": "Logs Export API absent from the spec", + "GET /logs/exports/{id}/download": "Logs Export API absent from the spec", + "POST /logs/exports/{id}/start": "Logs Export API absent from the spec", + "POST /logs/exports/{id}/cancel": "Logs Export API absent from the spec", + "DELETE /logs/exports/{id}": "Logs Export API absent from the spec", + + # Only the input side is written as an endpoint. The output-guardrails path appears in + # the same page as backticked prose, which this check does not read, so an entry for it + # would match nothing and fail. + "PUT /workspace-exclusions/input-guardrails": "raised by Phase 7, still absent", +} + +# A path segment standing in for a value. The corpus uses all three syntaxes, sometimes +# on the same page: `{file_id}` from the spec, `:apiKeyId` from the older admin docs, and +# `` in shell examples. Leaving any of them out truncates the path at the first +# placeholder -- `/v1/files/` became `/files`, which then failed as a nonexistent +# endpoint. Every one of the first run's "findings" in this class was that bug. +# A `*` is a placeholder too: prose writes `GET /v1/responses/*` to mean the read routes +# as a set. +PLACEHOLDER = re.compile(r"^(?:\{[^}]*\}|:[A-Za-z_][A-Za-z0-9_]*|<[^>]*>|\*)$") + +PATH_CHARS = r"[A-Za-z0-9_{}<>:./~*-]" + +# `METHOD /path`, the way an endpoint is written in prose, a heading or a fence. +PROSE = re.compile(r"\b(GET|POST|PUT|DELETE|PATCH)\s+(/" + PATH_CHARS + r"*)") + +# The same thing inside a curl invocation, where the method and the URL are separated by +# flags and the path is buried in a host. +# +# Only these hosts. An earlier version matched any URL and reported 494 failures, nearly +# all of them GitHub links, images and third-party documentation -- which is the shape of +# a check nobody runs twice. The gateway's own hosts are the only ones whose paths this +# repository is entitled to have an opinion about. +API_HOSTS = ( + "aigw.portkey.ai", + "api.apps.paloaltonetworks.com", + "api.portkey.ai", +) +CURL_URL = re.compile( + r"https?://(?:" + "|".join(re.escape(h) for h in API_HOSTS) + r")(/" + PATH_CHARS + r"*)" +) +CURL_METHOD = re.compile(r"(?:-X|--request)\s+(GET|POST|PUT|DELETE|PATCH)\b") +# curl sends POST when given a body, with no -X at all. Most examples in this corpus are +# written that way, so inferring GET from the absence of -X reports every one of them as +# a nonexistent GET endpoint. +CURL_BODY = re.compile(r"(?:^|\s)(?:-d|--data|--data-raw|--data-binary|--json|-F|--form)\b") + + +def load_fragment(path): + if path: + with open(path) as fh: + return json.load(fh) + req = urllib.request.Request(FRAGMENT_URL, headers={"User-Agent": "curl/8"}) + with urllib.request.urlopen(req, timeout=60) as r: + return json.loads(r.read().decode("utf-8")) + + +def spec_operations(fragment): + """Every `"METHOD /path"` string anywhere in the navigation fragment.""" + found = set() + + def walk(node): + if isinstance(node, dict): + for v in node.values(): + walk(v) + elif isinstance(node, list): + for v in node: + walk(v) + elif isinstance(node, str) and PROSE.fullmatch(node): + found.add(node) + + walk(fragment) + return found + + +def normalise(path): + """Strip a server prefix and any trailing slash, so prose and spec are comparable.""" + for prefix in BASE_PREFIXES: + if path == prefix: + return "/" + if path.startswith(prefix + "/"): + path = path[len(prefix):] + break + return path.rstrip("/") or "/" + + +def matches(written, declared): + """True if a written path fits a declared one, with placeholders as wildcards. + + A placeholder on either side matches anything: `/api-keys/service` fits + `/api-keys/{sub-type}` because the spec names the segment, and `/files/` + fits `/files/{file_id}` because the example fills it in. `/api-keys/a/b` fits + neither -- segment counts must agree. + """ + w = written.strip("/").split("/") + d = declared.strip("/").split("/") + if len(w) != len(d): + return False + return all( + PLACEHOLDER.match(ds) or PLACEHOLDER.match(ws) or ds == ws + for ws, ds in zip(w, d) + ) + + +def find_match(method, path, declared_by_method): + for declared in declared_by_method.get(method, ()): + if matches(path, declared): + return f"{method} {declared}" + return None + + +def curl_commands(lines): + """Yield (start_line, command_text) for each curl invocation, joined across wrapping. + + A curl example spans many lines held together by trailing backslashes, and the method, + the body flag and the URL are rarely on the same one. The method has to be decided + from the whole command or not at all. + """ + i = 0 + while i < len(lines): + if re.match(r"\s*curl\b", lines[i]): + start = i + parts = [lines[i]] + while parts[-1].rstrip().endswith("\\") and i + 1 < len(lines): + i += 1 + parts.append(lines[i]) + yield start + 1, " ".join(parts) + i += 1 + + +def scan_file(path): + """Yield (line_number, method, raw_path, method_is_certain) for each endpoint written. + + The method is certain when it is written down -- in prose, or as curl's -X. It is a + guess when curl omits -X and the method has to come from whether a body is present. + Documentation examples are routinely abridged to the header being discussed, so a + POST example with its -d elided looks exactly like a GET. Callers should not fail a + guess that lands on a real path under some other method. + """ + with open(path, encoding="utf-8") as fh: + lines = fh.read().split("\n") + + for i, line in enumerate(lines, 1): + for method, raw in PROSE.findall(line): + yield i, method, raw, True + + for lineno, command in curl_commands(lines): + found = CURL_METHOD.search(command) + if found: + method, certain = found.group(1), True + elif CURL_BODY.search(command): + method, certain = "POST", False + else: + method, certain = "GET", False + for raw in CURL_URL.findall(command): + if raw and raw != "/": + yield lineno, method, raw.split("?")[0], certain + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--from", dest="src", help="read docs-navigation.json from a local path") + ap.add_argument("--list", action="store_true", help="print every endpoint and its match") + args = ap.parse_args() + + declared = spec_operations(load_fragment(args.src)) + by_method = {} + for op in declared: + method, path = op.split(" ", 1) + by_method.setdefault(method, []).append(path) + + unknown = {} + seen = 0 + allowed_hits = set() + + for root, _dirs, files in os.walk(AIGW): + if "_project" in root: + continue + for name in sorted(files): + if not name.endswith(".mdx"): + continue + full = os.path.join(root, name) + rel = os.path.relpath(full, os.path.join(AIGW, "..")) + for lineno, method, raw, certain in scan_file(full): + path = normalise(raw) + if path == "/": + continue + seen += 1 + key = f"{method} {path}" + + hit = find_match(method, path, by_method) + if not hit and not certain: + # The method was inferred. If the path is real under any method, the + # example is fine and the inference was simply wrong. + hit = next( + (find_match(m, path, by_method) for m in by_method + if find_match(m, path, by_method)), + None, + ) + if hit: + if args.list: + print(f" ok {key:58} -> {hit}") + continue + + # Placeholders are wildcards on both sides, so a written `/logs/exports/{id}` + # matches a listed `/logs/exports/field-restrictions` as readily as the + # entry meant for it. Prefer whichever candidate agrees on the most literal + # segments, so the specific entry wins and the general one is not wrongly + # reported as matching nothing. + candidates = [ + a for a in (*ALLOWED, *KNOWN_GAPS) + if a.split(" ", 1)[0] == method and matches(path, a.split(" ", 1)[1]) + ] + listed = max( + candidates, + key=lambda a: sum( + 1 for ws, ds in zip(path.strip("/").split("/"), + a.split(" ", 1)[1].strip("/").split("/")) + if ws == ds + ), + default=None, + ) + if listed: + allowed_hits.add(listed) + if args.list: + why = ALLOWED.get(listed) or KNOWN_GAPS[listed] + label = "allowed" if listed in ALLOWED else "gap" + print(f" {label:8} {key:58} -- {why}") + continue + + unknown.setdefault(key, []).append(f"{rel}:{lineno}") + + print(f"{seen} endpoint mentions checked against {len(declared)} declared operations") + allowed_n = len(allowed_hits & set(ALLOWED)) + gaps_n = len(allowed_hits & set(KNOWN_GAPS)) + if allowed_n: + print(f"{allowed_n} outside the spec by nature (ALLOWED)") + if gaps_n: + print(f"{gaps_n} missing from the spec and tracked upstream (KNOWN_GAPS)") + + stale = sorted((set(ALLOWED) | set(KNOWN_GAPS)) - allowed_hits) + if stale: + print("\nListed endpoints that matched nothing. Either the prose that needed them") + print("is gone or the spec now has them -- delete the entry either way:") + for s in stale: + print(f" {s}") + + if not unknown: + if not stale: + print("\nEvery endpoint written in aigw/ is accounted for.") + return 1 if stale else 0 + + print(f"\n{len(unknown)} endpoint(s) written in aigw/ are not in the specification:\n") + for key in sorted(unknown): + where = unknown[key] + print(f" {key}") + for w in where[:6]: + print(f" {w}") + if len(where) > 6: + print(f" ... and {len(where) - 6} more") + print("\nEither the prose is stale, the spec is missing an operation, or it belongs") + print("in ALLOWED with a reason. Do not loosen the match to make this pass.") + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/aigw/self-hosting/cache-behavior.mdx b/aigw/self-hosting/cache-behavior.mdx index ad729d03..8240f231 100644 --- a/aigw/self-hosting/cache-behavior.mdx +++ b/aigw/self-hosting/cache-behavior.mdx @@ -1,5 +1,5 @@ --- -title: "Cache Behavior" +title: "Cache Behaviour" sidebarTitle: "Cache Behavior" description: "How the Gateway cache works: population, TTL, sync, resync, and cache invalidation and refresh." --- diff --git a/aigw/self-hosting/prometheus-metrics.mdx b/aigw/self-hosting/prometheus-metrics.mdx index e3e67e6f..1743bb19 100644 --- a/aigw/self-hosting/prometheus-metrics.mdx +++ b/aigw/self-hosting/prometheus-metrics.mdx @@ -29,7 +29,7 @@ Ensure your Prometheus server can access the `/metrics` endpoint. In production ### Default Labels -All custom metrics are automatically labeled with: +All custom metrics are automatically labelled with: - `app`: Service identifier (from `SERVICE_NAME` environment variable) - `env`: Deployment environment (from `NODE_ENV` environment variable) @@ -61,7 +61,7 @@ The AI Gateway Enterprise Gateway exposes 15 custom metrics designed to provide ### Universal Label Schema -All custom metrics share a common labeling schema enabling multi-dimensional analysis: +All custom metrics share a common labelling schema enabling multi-dimensional analysis: - `method`: HTTP verb (GET, POST, PUT, DELETE, etc.) - `endpoint`: Normalized API endpoint path (e.g., `/v1/chat/completions`, `/v1/completions`) @@ -181,7 +181,7 @@ Captures the time difference between receiving the first response data and the f **Streaming Analysis Value**: - **Time-to-First-Token**: Indirectly measurable by comparing with total request duration - **Streaming Efficiency**: Identifies providers with consistent vs. bursty streaming patterns -- **Model Behavior Analysis**: Different models exhibit different streaming characteristics +- **Model Behaviour Analysis**: Different models exhibit different streaming characteristics **Use Cases**: - **User Experience Optimization**: Understand perceived responsiveness for streaming applications @@ -508,7 +508,7 @@ ENABLE_PROMETHEUS=false ENABLE_PROMETHEUS=true ``` -**Behavior**: +**Behaviour**: - When `ENABLE_PROMETHEUS=false`: - Metrics middleware is not registered - The `/metrics` endpoint returns 404 @@ -566,7 +566,7 @@ This exports: ### Dynamic Metadata Label System -The AI Gateway Enterprise Gateway supports dynamic metadata labeling through request-specific metadata injection. This powerful feature enables fine-grained observability across custom dimensions specific to your organisation's structure and use cases. +The AI Gateway Enterprise Gateway supports dynamic metadata labelling through request-specific metadata injection. This powerful feature enables fine-grained observability across custom dimensions specific to your organisation's structure and use cases. **Metadata labels are disabled by default** to prevent cardinality issues. Enable them with `PROMETHEUS_INCLUDE_METADATA_LABELS`, then restrict which keys are promoted to labels using `PROMETHEUS_LABELS_METADATA_ALLOWED_KEYS`. diff --git a/docs.json b/docs.json index 6befecca..94ea0a3a 100644 --- a/docs.json +++ b/docs.json @@ -147,7 +147,6 @@ "pages": [ "product/mcp-gateway/authentication", "product/mcp-gateway/authentication/oauth", - "product/mcp-gateway/authentication/cas", "product/mcp-gateway/authentication/external-oauth", "product/mcp-gateway/authentication/identity-forwarding", "product/mcp-gateway/authentication/jwt", @@ -2424,6 +2423,7 @@ "group": "API Reference", "pages": [ "aigw/api-reference/admin-api/introduction", + "aigw/api-reference/admin-api/authentication", "aigw/api-reference/admin-api/error" ] }, @@ -2643,7 +2643,7 @@ "directory": "aigw/api-reference" }, "pages": [ - "POST /api-keys/{type}/{sub-type}", + "POST /api-keys/{sub-type}", "GET /api-keys", "PUT /api-keys/{id}", "GET /api-keys/{id}",