> ## Documentation Index
> Fetch the complete documentation index at: https://docs.gp.scale.com/llms.txt
> Use this file to discover all available pages before exploring further.

# Link Project Data Source

> Link an existing Data Sources connector and create DEX webhook subscription.



## OpenAPI

````yaml https://dex.sgp.scale.com/openapi.json post /v1/projects/{project_id}/data-sources
openapi: 3.1.0
info:
  title: Document Understanding API
  description: API for uploading and processing documents
  version: 0.6.0
servers: []
security:
  - ApiKey: []
    AccountId: []
tags:
  - name: Projects
    description: Operations related to project creation and management
  - name: Files
    description: Operations related to file upload and access
  - name: Parse
    description: Operations related to starting parse jobs and accessing their results
  - name: Vector Stores
    description: Operations related to vector store creation and management
  - name: Extract
    description: Operations related to starting extract jobs and accessing their results
  - name: Research
    description: Dex Research agent kickoff and results.
  - name: Jobs
    description: Operations related to monitoring jobs and their status
  - name: Data sources
    description: Project-scoped integration with the Data Sources service
  - name: Webhooks
    description: Callbacks from external services
paths:
  /v1/projects/{project_id}/data-sources:
    post:
      tags:
        - Data sources
      summary: Link Project Data Source
      description: >-
        Link an existing Data Sources connector and create DEX webhook
        subscription.
      operationId: link_project_data_source_v1_projects__project_id__data_sources_post
      parameters:
        - name: project_id
          in: path
          required: true
          schema:
            type: string
            title: Project Id
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/DataSourceLinkRequest'
      responses:
        '201':
          description: Successful Response
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/DataSourceEntity'
        '422':
          description: Validation Error
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/HTTPValidationError'
components:
  schemas:
    DataSourceLinkRequest:
      properties:
        data_sources_data_source_id:
          anyOf:
            - type: string
              format: uuid
            - type: 'null'
          title: Data Sources Data Source Id
          description: Existing Data Sources service data source id to link.
        connector:
          anyOf:
            - oneOf:
                - $ref: '#/components/schemas/SharePointDataSourceCreate'
                - $ref: '#/components/schemas/GoogleDriveDataSourceCreate'
                - $ref: '#/components/schemas/ConfluenceDataSourceCreate'
              discriminator:
                propertyName: type
                mapping:
                  confluence:
                    $ref: '#/components/schemas/ConfluenceDataSourceCreate'
                  google_drive:
                    $ref: '#/components/schemas/GoogleDriveDataSourceCreate'
                  sharepoint:
                    $ref: '#/components/schemas/SharePointDataSourceCreate'
            - type: 'null'
          title: Connector
          description: Config for a new connector to create and link (project-owned).
        workflow:
          anyOf:
            - $ref: '#/components/schemas/WorkflowDefinition'
            - type: 'null'
          description: Workflow definition to trigger for datasource change events.
      type: object
      title: DataSourceLinkRequest
      description: >-
        Attach a Data Sources connector to a DEX project.


        Provide exactly one of:

        - ``connector``: create a new connector on the Data Sources service and
        link it
          (project-owned), or
        - ``data_sources_data_source_id``: link an existing connector (reuse
        across projects).
    DataSourceEntity:
      properties:
        id:
          type: string
          title: Id
          description: ID of the entity
        project_id:
          type: string
          title: Project Id
          description: ID of the project
        object:
          type: string
          const: data_source
          title: Object
          default: data_source
        name:
          type: string
          title: Name
        provider_type:
          type: string
          title: Provider Type
          description: >-
            Provider discriminator (e.g. sharepoint); aligned with Data Sources
            values
        data_sources_data_source_id:
          type: string
          title: Data Sources Data Source Id
          description: Remote data source id in the Data Sources service
        data_sources_subscription_id:
          type: string
          title: Data Sources Subscription Id
          description: Active subscription id in the Data Sources service
        workflow_id:
          anyOf:
            - type: string
            - type: 'null'
          title: Workflow Id
          description: Workflow triggered by this linked data source.
        workflow_status:
          anyOf:
            - type: string
            - type: 'null'
          title: Workflow Status
          description: Status of the workflow triggered by this linked data source.
        status:
          $ref: '#/components/schemas/DataSourceStatus'
        config_metadata:
          additionalProperties: true
          type: object
          title: Config Metadata
        last_delta_cursor:
          anyOf:
            - type: string
            - type: 'null'
          title: Last Delta Cursor
        last_event_at:
          anyOf:
            - type: string
              format: date-time
            - type: 'null'
          title: Last Event At
        created_at:
          type: string
          format: date-time
          title: Created At
        updated_at:
          type: string
          format: date-time
          title: Updated At
        deleted_at:
          anyOf:
            - type: string
              format: date-time
            - type: 'null'
          title: Deleted At
      type: object
      required:
        - id
        - project_id
        - name
        - provider_type
        - data_sources_data_source_id
        - data_sources_subscription_id
        - status
        - created_at
        - updated_at
      title: DataSourceEntity
    HTTPValidationError:
      properties:
        detail:
          items:
            $ref: '#/components/schemas/ValidationError'
          type: array
          title: Detail
      type: object
      title: HTTPValidationError
    SharePointDataSourceCreate:
      properties:
        name:
          type: string
          title: Name
        type:
          type: string
          const: sharepoint
          title: Type
        credentials:
          oneOf:
            - $ref: '#/components/schemas/SharePointClientCredentials'
            - $ref: '#/components/schemas/SharePointCertificateCredentials'
          title: Credentials
          discriminator:
            propertyName: type
            mapping:
              certificate_credentials:
                $ref: '#/components/schemas/SharePointCertificateCredentials'
              client_credentials:
                $ref: '#/components/schemas/SharePointClientCredentials'
        config:
          oneOf:
            - $ref: '#/components/schemas/SharePointGraphConfig'
            - $ref: '#/components/schemas/SharePointRestConfig'
          title: Config
          discriminator:
            propertyName: api_type
            mapping:
              graph:
                $ref: '#/components/schemas/SharePointGraphConfig'
              sharepoint_rest:
                $ref: '#/components/schemas/SharePointRestConfig'
        sync:
          type: boolean
          title: Sync
          default: true
      additionalProperties: false
      type: object
      required:
        - name
        - type
        - credentials
        - config
      title: SharePointDataSourceCreate
    GoogleDriveDataSourceCreate:
      properties:
        name:
          type: string
          title: Name
        type:
          type: string
          const: google_drive
          title: Type
        credentials:
          oneOf:
            - $ref: '#/components/schemas/GoogleDriveServiceAccountCredentials'
            - $ref: '#/components/schemas/GoogleDriveRefreshTokenCredentials'
          title: Credentials
          discriminator:
            propertyName: type
            mapping:
              refresh_token:
                $ref: '#/components/schemas/GoogleDriveRefreshTokenCredentials'
              service_account:
                $ref: '#/components/schemas/GoogleDriveServiceAccountCredentials'
        config:
          $ref: '#/components/schemas/GoogleDriveConfig'
        sync:
          type: boolean
          title: Sync
          default: true
      additionalProperties: false
      type: object
      required:
        - name
        - type
        - credentials
        - config
      title: GoogleDriveDataSourceCreate
    ConfluenceDataSourceCreate:
      properties:
        name:
          type: string
          title: Name
        type:
          type: string
          const: confluence
          title: Type
        credentials:
          $ref: '#/components/schemas/ConfluenceApiTokenDataSourceCredentials'
        config:
          $ref: '#/components/schemas/ConfluenceConfig'
        sync:
          type: boolean
          title: Sync
          default: true
      additionalProperties: false
      type: object
      required:
        - name
        - type
        - credentials
        - config
      title: ConfluenceDataSourceCreate
    WorkflowDefinition:
      properties:
        name:
          anyOf:
            - type: string
              maxLength: 255
            - type: 'null'
          title: Name
        steps:
          $ref: '#/components/schemas/WorkflowStepsDefinition'
      additionalProperties: false
      type: object
      title: WorkflowDefinition
    DataSourceStatus:
      type: string
      enum:
        - active
        - paused
        - disabled
        - deleted
      title: DataSourceStatus
    ValidationError:
      properties:
        loc:
          items:
            anyOf:
              - type: string
              - type: integer
          type: array
          title: Location
        msg:
          type: string
          title: Message
        type:
          type: string
          title: Error Type
        input:
          title: Input
        ctx:
          type: object
          title: Context
      type: object
      required:
        - loc
        - msg
        - type
      title: ValidationError
    SharePointClientCredentials:
      properties:
        type:
          type: string
          const: client_credentials
          title: Type
        creds:
          $ref: '#/components/schemas/ClientCredentials'
      additionalProperties: false
      type: object
      required:
        - type
        - creds
      title: SharePointClientCredentials
    SharePointCertificateCredentials:
      properties:
        type:
          type: string
          const: certificate_credentials
          title: Type
        creds:
          $ref: '#/components/schemas/CertificateCredentials'
      additionalProperties: false
      type: object
      required:
        - type
        - creds
      title: SharePointCertificateCredentials
    SharePointGraphConfig:
      properties:
        api_type:
          type: string
          const: graph
          title: Api Type
          default: graph
        tenant_id:
          type: string
          title: Tenant Id
        site_url:
          anyOf:
            - type: string
              minLength: 1
              format: uri
            - type: 'null'
          title: Site Url
        site_id:
          anyOf:
            - type: string
            - type: 'null'
          title: Site Id
        drive_id:
          anyOf:
            - type: string
            - type: 'null'
          title: Drive Id
        folder_path:
          anyOf:
            - type: string
            - type: 'null'
          title: Folder Path
        token_exchange:
          $ref: '#/components/schemas/TokenExchangeConfig'
      additionalProperties: false
      type: object
      required:
        - tenant_id
      title: SharePointGraphConfig
    SharePointRestConfig:
      properties:
        api_type:
          type: string
          const: sharepoint_rest
          title: Api Type
          default: sharepoint_rest
        tenant_id:
          type: string
          title: Tenant Id
        site_url:
          type: string
          minLength: 1
          format: uri
          title: Site Url
        list_id:
          type: string
          title: List Id
        folder_server_relative_url:
          anyOf:
            - type: string
            - type: 'null'
          title: Folder Server Relative Url
        token_exchange:
          $ref: '#/components/schemas/TokenExchangeConfig'
      additionalProperties: false
      type: object
      required:
        - tenant_id
        - site_url
        - list_id
      title: SharePointRestConfig
    GoogleDriveServiceAccountCredentials:
      properties:
        type:
          type: string
          const: service_account
          title: Type
        creds:
          $ref: '#/components/schemas/GoogleServiceAccountCredentials'
      additionalProperties: false
      type: object
      required:
        - type
        - creds
      title: GoogleDriveServiceAccountCredentials
    GoogleDriveRefreshTokenCredentials:
      properties:
        type:
          type: string
          const: refresh_token
          title: Type
        creds:
          $ref: '#/components/schemas/GoogleRefreshTokenCredentials'
      additionalProperties: false
      type: object
      required:
        - type
        - creds
      title: GoogleDriveRefreshTokenCredentials
    GoogleDriveConfig:
      properties:
        drive_id:
          anyOf:
            - type: string
            - type: 'null'
          title: Drive Id
        folder_id:
          anyOf:
            - type: string
            - type: 'null'
          title: Folder Id
        include_shared_drives:
          type: boolean
          title: Include Shared Drives
          default: false
      additionalProperties: false
      type: object
      title: GoogleDriveConfig
    ConfluenceApiTokenDataSourceCredentials:
      properties:
        type:
          type: string
          const: api_token
          title: Type
        creds:
          $ref: '#/components/schemas/ConfluenceApiTokenCredentials'
      additionalProperties: false
      type: object
      required:
        - type
        - creds
      title: ConfluenceApiTokenDataSourceCredentials
    ConfluenceConfig:
      properties:
        base_url:
          type: string
          title: Base Url
        space_id:
          anyOf:
            - type: string
            - type: 'null'
          title: Space Id
        cql_scope:
          anyOf:
            - type: string
            - type: 'null'
          title: Cql Scope
      additionalProperties: false
      type: object
      required:
        - base_url
      title: ConfluenceConfig
    WorkflowStepsDefinition:
      properties:
        parse:
          anyOf:
            - oneOf:
                - $ref: '#/components/schemas/ReductoParseJobParams'
                - $ref: '#/components/schemas/IrisParseJobParams'
              discriminator:
                propertyName: engine
                mapping:
                  iris:
                    $ref: '#/components/schemas/IrisParseJobParams'
                  reducto:
                    $ref: '#/components/schemas/ReductoParseJobParams'
            - type: 'null'
          title: Parse
        chunk:
          anyOf:
            - oneOf:
                - $ref: '#/components/schemas/TokenSizeChunkingOptions'
                - $ref: '#/components/schemas/RecursiveChunkingOptions'
                - $ref: '#/components/schemas/PageChunkingOptions'
                - $ref: '#/components/schemas/SectionChunkingOptions'
              discriminator:
                propertyName: strategy
                mapping:
                  by_page:
                    $ref: '#/components/schemas/PageChunkingOptions'
                  by_section:
                    $ref: '#/components/schemas/SectionChunkingOptions'
                  recursive:
                    $ref: '#/components/schemas/RecursiveChunkingOptions'
                  token_size:
                    $ref: '#/components/schemas/TokenSizeChunkingOptions'
            - type: 'null'
          title: Chunk
        vectorize:
          anyOf:
            - $ref: '#/components/schemas/WorkflowSGPVectorStoreStepDefinition'
            - $ref: '#/components/schemas/SGPVectorStoreConfigModelsApi'
            - $ref: '#/components/schemas/WorkflowSGPKnowledgeBaseStepDefinition'
            - $ref: '#/components/schemas/SGPKnowledgeBaseConfigModelsApi'
            - type: 'null'
          title: Vectorize
        index:
          anyOf:
            - $ref: '#/components/schemas/WorkflowFileSystemIndexStepDefinition'
            - type: 'null'
      additionalProperties: false
      type: object
      title: WorkflowStepsDefinition
    ClientCredentials:
      properties:
        client_id:
          type: string
          title: Client Id
        client_secret:
          type: string
          title: Client Secret
        tenant_id:
          anyOf:
            - type: string
            - type: 'null'
          title: Tenant Id
      additionalProperties: false
      type: object
      required:
        - client_id
        - client_secret
      title: ClientCredentials
    CertificateCredentials:
      properties:
        client_id:
          type: string
          title: Client Id
        private_key:
          type: string
          title: Private Key
        certificate_thumbprint:
          type: string
          title: Certificate Thumbprint
        tenant_id:
          anyOf:
            - type: string
            - type: 'null'
          title: Tenant Id
      additionalProperties: false
      type: object
      required:
        - client_id
        - private_key
        - certificate_thumbprint
      title: CertificateCredentials
    TokenExchangeConfig:
      properties:
        user_token_header:
          type: string
          minLength: 1
          title: User Token Header
          default: X-User-Access-Token
        user_assertion_scope:
          anyOf:
            - type: string
              minLength: 1
            - type: 'null'
          title: User Assertion Scope
      additionalProperties: false
      type: object
      title: TokenExchangeConfig
    GoogleServiceAccountCredentials:
      properties:
        client_email:
          type: string
          title: Client Email
        private_key:
          type: string
          title: Private Key
        token_uri:
          type: string
          title: Token Uri
          default: https://oauth2.googleapis.com/token
        scopes:
          items:
            type: string
          type: array
          title: Scopes
      additionalProperties: false
      type: object
      required:
        - client_email
        - private_key
      title: GoogleServiceAccountCredentials
    GoogleRefreshTokenCredentials:
      properties:
        client_id:
          type: string
          title: Client Id
        client_secret:
          type: string
          title: Client Secret
        refresh_token:
          type: string
          title: Refresh Token
      additionalProperties: false
      type: object
      required:
        - client_id
        - client_secret
        - refresh_token
      title: GoogleRefreshTokenCredentials
    ConfluenceApiTokenCredentials:
      properties:
        username:
          type: string
          title: Username
        api_token:
          type: string
          title: Api Token
      additionalProperties: false
      type: object
      required:
        - username
        - api_token
      title: ConfluenceApiTokenCredentials
    ReductoParseJobParams:
      properties:
        engine:
          type: string
          const: reducto
          title: Engine
          description: Engine
          default: reducto
        chunking_options:
          anyOf:
            - oneOf:
                - $ref: '#/components/schemas/TokenSizeChunkingOptions'
                - $ref: '#/components/schemas/RecursiveChunkingOptions'
                - $ref: '#/components/schemas/PageChunkingOptions'
                - $ref: '#/components/schemas/SectionChunkingOptions'
              discriminator:
                propertyName: strategy
                mapping:
                  by_page:
                    $ref: '#/components/schemas/PageChunkingOptions'
                  by_section:
                    $ref: '#/components/schemas/SectionChunkingOptions'
                  recursive:
                    $ref: '#/components/schemas/RecursiveChunkingOptions'
                  token_size:
                    $ref: '#/components/schemas/TokenSizeChunkingOptions'
            - type: 'null'
          title: Chunking Options
          description: Chunking options
        vector_store_metadata:
          anyOf:
            - additionalProperties:
                anyOf:
                  - type: string
                  - type: integer
                  - type: number
                  - type: boolean
              type: object
            - type: 'null'
          title: Vector Store Metadata
          description: >-
            Metadata to populate into the vector store to filter on when
            searching
        options:
          $ref: '#/components/schemas/ReductoParseEngineOptions'
          description: Options
        advanced_options:
          additionalProperties: true
          type: object
          title: Advanced Options
          description: Advanced options
          default: {}
        experimental_options:
          additionalProperties: true
          type: object
          title: Experimental Options
          description: Experimental options
          default: {}
        priority:
          type: boolean
          title: Priority
          description: Priority
          default: false
      type: object
      required:
        - options
      title: ReductoParseJobParams
      description: Parameters for creating a parse job.
    IrisParseJobParams:
      properties:
        engine:
          type: string
          const: iris
          title: Engine
          description: Engine
          default: iris
        chunking_options:
          anyOf:
            - oneOf:
                - $ref: '#/components/schemas/TokenSizeChunkingOptions'
                - $ref: '#/components/schemas/RecursiveChunkingOptions'
                - $ref: '#/components/schemas/PageChunkingOptions'
                - $ref: '#/components/schemas/SectionChunkingOptions'
              discriminator:
                propertyName: strategy
                mapping:
                  by_page:
                    $ref: '#/components/schemas/PageChunkingOptions'
                  by_section:
                    $ref: '#/components/schemas/SectionChunkingOptions'
                  recursive:
                    $ref: '#/components/schemas/RecursiveChunkingOptions'
                  token_size:
                    $ref: '#/components/schemas/TokenSizeChunkingOptions'
            - type: 'null'
          title: Chunking Options
          description: Chunking options
        vector_store_metadata:
          anyOf:
            - additionalProperties:
                anyOf:
                  - type: string
                  - type: integer
                  - type: number
                  - type: boolean
              type: object
            - type: 'null'
          title: Vector Store Metadata
          description: >-
            Metadata to populate into the vector store to filter on when
            searching
        options:
          $ref: '#/components/schemas/IrisParseEngineOptions'
          description: Options
      type: object
      required:
        - options
      title: IrisParseJobParams
    TokenSizeChunkingOptions:
      properties:
        strategy:
          type: string
          const: token_size
          title: Strategy
          description: The chunking strategy
          default: token_size
        chunk_size:
          type: integer
          exclusiveMinimum: 0
          title: Chunk Size
          description: Target size of each chunk in tokens
          default: 512
        chunk_overlap:
          type: integer
          minimum: 0
          title: Chunk Overlap
          description: Number of overlapping tokens between chunks
          default: 50
        encoding_name:
          type: string
          title: Encoding Name
          description: Tiktoken encoding name (e.g., cl100k_base for GPT-4)
          default: cl100k_base
      type: object
      title: TokenSizeChunkingOptions
      description: >-
        Token-based chunking: Splits text into chunks by token count using a
        tokenizer (e.g., tiktoken). Best for LLM APIs with token limits,
        embedding models, and cost optimization. Use when you need precise
        control over token usage.
    RecursiveChunkingOptions:
      properties:
        strategy:
          type: string
          const: recursive
          title: Strategy
          description: The chunking strategy
          default: recursive
        chunk_size:
          type: integer
          exclusiveMinimum: 0
          title: Chunk Size
          description: Target size of each chunk in characters
          default: 1000
        chunk_overlap:
          type: integer
          minimum: 0
          title: Chunk Overlap
          description: Number of overlapping characters between chunks
          default: 200
        separators:
          items:
            type: string
          type: array
          title: Separators
          description: List of separators to try in order
          default:
            - |+


            - |+

            - ' '
            - ''
        keep_separator:
          type: boolean
          title: Keep Separator
          description: Whether to keep the separator in the chunks
          default: true
      type: object
      title: RecursiveChunkingOptions
      description: >-
        Recursive text splitting: Uses a hierarchy of separators (paragraphs →
        sentences → words) to preserve natural text boundaries. Best for
        articles, documentation, and RAG systems where readability and semantic
        coherence matter.
    PageChunkingOptions:
      properties:
        strategy:
          type: string
          const: by_page
          title: Strategy
          description: The chunking strategy
          default: by_page
        pages_per_chunk:
          type: integer
          exclusiveMinimum: 0
          title: Pages Per Chunk
          description: Number of pages to include in each chunk
          default: 1
      type: object
      title: PageChunkingOptions
      description: >-
        Page-based chunking: Splits documents by page boundaries, grouping
        complete pages together. Best for legal documents, forms, and reports
        where page references are important and page structure should be
        preserved.
    SectionChunkingOptions:
      properties:
        strategy:
          type: string
          const: by_section
          title: Strategy
          description: The chunking strategy
          default: by_section
        section_headers:
          anyOf:
            - items:
                type: string
              type: array
            - type: 'null'
          title: Section Headers
          description: Markdown-style headers that denote sections
        include_header_in_chunk:
          type: boolean
          title: Include Header In Chunk
          description: Whether to include the section header in the chunk
          default: true
      type: object
      title: SectionChunkingOptions
      description: >-
        Section-based chunking: Splits text by section headers (e.g., markdown
        #, ##, ###) to keep complete topics together. Best for structured
        documents like technical manuals, wikis, and academic papers where
        semantic coherence within topics is crucial.
    WorkflowSGPVectorStoreStepDefinition:
      properties:
        engine:
          type: string
          const: sgp_vector_store
          title: Engine
          default: sgp_vector_store
        vector_store_metadata_schema:
          anyOf:
            - additionalProperties:
                type: string
                enum:
                  - string
                  - int
                  - double
                  - boolean
              type: object
            - type: 'null'
          title: Vector Store Metadata Schema
          description: >-
            Schema of the vector store metadata. You can set metadata for parsed
            documents and they will be indexed as extra metadata in the vector
            store to filter on when searching.
        embedding_type:
          type: string
          const: base
          title: Embedding Type
          description: Type of embedding configuration for standard models
          default: base
        embedding_model:
          anyOf:
            - type: string
              enum:
                - sentence-transformers/all-MiniLM-L12-v2
                - sentence-transformers/all-mpnet-base-v2
                - sentence-transformers/multi-qa-distilbert-cos-v1
                - sentence-transformers/paraphrase-multilingual-mpnet-base-v2
                - openai/text-embedding-ada-002
                - openai/text-embedding-3-small
                - openai/text-embedding-3-large
                - embed-english-v3.0
                - embed-english-light-v3.0
                - embed-multilingual-v3.0
                - gemini/text-embedding-005
                - gemini/text-multilingual-embedding-002
                - gemini/gemini-embedding-001
            - type: string
          title: Embedding Model
          description: Embedding model to use for the workflow-owned SGP Vector Store.
          default: openai/text-embedding-3-small
      type: object
      title: WorkflowSGPVectorStoreStepDefinition
    SGPVectorStoreConfigModelsApi:
      properties:
        engine:
          type: string
          const: sgp_vector_store
          title: Engine
          default: sgp_vector_store
        vector_store_metadata_schema:
          anyOf:
            - additionalProperties:
                type: string
                enum:
                  - string
                  - int
                  - double
                  - boolean
              type: object
            - type: 'null'
          title: Vector Store Metadata Schema
          description: >-
            Schema of the vector store metadata. You can set metadata for parsed
            documents and they will be indexed as extra metadata in the vector
            store to filter on when searching.
        embedding_type:
          type: string
          const: models_api
          title: Embedding Type
          description: Type of embedding configuration for custom deployed models
          default: models_api
        model_deployment_id:
          type: string
          title: Model Deployment Id
          description: Model deployment ID for 'models_api' type from Models API V4
      type: object
      required:
        - model_deployment_id
      title: SGPVectorStoreConfigModelsApi
      description: Models API embedding configuration using custom deployed models.
    WorkflowSGPKnowledgeBaseStepDefinition:
      properties:
        engine:
          type: string
          const: sgp_knowledge_base
          title: Engine
          default: sgp_knowledge_base
        vector_store_metadata_schema:
          anyOf:
            - additionalProperties:
                type: string
                enum:
                  - string
                  - int
                  - double
                  - boolean
              type: object
            - type: 'null'
          title: Vector Store Metadata Schema
          description: >-
            Schema of the vector store metadata. You can set metadata for parsed
            documents and they will be indexed as extra metadata in the vector
            store to filter on when searching.
        embedding_type:
          type: string
          const: base
          title: Embedding Type
          description: Type of embedding configuration for standard models
          default: base
        embedding_model:
          type: string
          title: Embedding Model
          description: Embedding model to use for the workflow-owned SGP Knowledge Base.
          default: openai/text-embedding-3-small
      type: object
      title: WorkflowSGPKnowledgeBaseStepDefinition
    SGPKnowledgeBaseConfigModelsApi:
      properties:
        engine:
          type: string
          const: sgp_knowledge_base
          title: Engine
          default: sgp_knowledge_base
        vector_store_metadata_schema:
          anyOf:
            - additionalProperties:
                type: string
                enum:
                  - string
                  - int
                  - double
                  - boolean
              type: object
            - type: 'null'
          title: Vector Store Metadata Schema
          description: >-
            Schema of the vector store metadata. You can set metadata for parsed
            documents and they will be indexed as extra metadata in the vector
            store to filter on when searching.
        embedding_type:
          type: string
          const: models_api
          title: Embedding Type
          description: Type of embedding configuration for custom deployed models
          default: models_api
        model_deployment_id:
          type: string
          title: Model Deployment Id
          description: Model deployment ID for 'models_api' type from Models API V4
      type: object
      required:
        - model_deployment_id
      title: SGPKnowledgeBaseConfigModelsApi
      description: Models API embedding configuration using custom deployed models.
    WorkflowFileSystemIndexStepDefinition:
      properties:
        model:
          type: string
          title: Model
          description: Model to use for summarization
          default: openai/gpt-5.2
        file_summarization_prompt:
          anyOf:
            - type: string
            - type: 'null'
          title: File Summarization Prompt
          description: Custom instructions for file summarization
        folder_summarization_prompt:
          anyOf:
            - type: string
            - type: 'null'
          title: Folder Summarization Prompt
          description: Custom instructions for folder summarization
        strategy:
          type: string
          const: file_system
          title: Strategy
          description: Workflow indexing strategy.
          default: file_system
        engine_type:
          type: string
          const: file_system
          title: Engine Type
          description: Type of index engine.
          default: file_system
      additionalProperties: false
      type: object
      title: WorkflowFileSystemIndexStepDefinition
    ReductoParseEngineOptions:
      properties:
        chunking:
          anyOf:
            - $ref: '#/components/schemas/ReductoChunkingOptions'
            - type: 'null'
          description: Chunking options
      additionalProperties: true
      type: object
      title: ReductoParseEngineOptions
      description: Options for the Reducto parse engine.
    IrisParseEngineOptions:
      properties:
        layout:
          anyOf:
            - type: string
              enum:
                - rt_detr_bce
                - pp_doclayout_v3
                - whole_page
            - type: 'null'
          title: Layout
          description: Layout detection model to use
        text_ocr:
          anyOf:
            - type: string
            - type: 'null'
          title: Text Ocr
          description: Text OCR model to use
        table_ocr:
          anyOf:
            - type: string
            - type: 'null'
          title: Table Ocr
          description: Table OCR model to use
        model_parameters:
          anyOf:
            - additionalProperties: true
              type: object
            - type: 'null'
          title: Model Parameters
          description: >-
            Extra parameters passed to LLM inference calls (e.g. max_tokens,
            temperature). Merged into the request payload; user-supplied values
            override defaults.
        bbox_soft_timeout_seconds:
          anyOf:
            - type: number
              exclusiveMinimum: 0
            - type: 'null'
          title: Bbox Soft Timeout Seconds
          description: >-
            Per-region OCR soft-timeout in seconds. If a single region's
            extraction (including retries) exceeds this, that region is degraded
            to empty content with an error rather than failing the page. When
            unset, a layout-aware default applies (higher for whole_page /
            e2e_ocr than region-based layouts).
        text_prompt:
          anyOf:
            - type: string
            - type: 'null'
          title: Text Prompt
          description: Custom prompt for text extraction models (only used by VLMs)
        table_prompt:
          anyOf:
            - type: string
            - type: 'null'
          title: Table Prompt
          description: Custom prompt for table extraction models (only used by VLMs)
        left_to_right:
          anyOf:
            - type: boolean
            - type: 'null'
          title: Left To Right
          description: >-
            Sort regions left-to-right instead of right-to-left when doing
            Markdown assembly (default: False)
        confidence_threshold:
          anyOf:
            - type: number
            - type: 'null'
          title: Confidence Threshold
          description: Minimum confidence threshold for layout detection boxes
        containment_threshold:
          anyOf:
            - type: number
            - type: 'null'
          title: Containment Threshold
          description: >-
            Containment threshold for filtering. If smaller box is X% contained
            in larger box, drop it.
        img_method:
          anyOf:
            - type: string
              enum:
                - base64
                - description
                - skip
            - type: 'null'
          title: Img Method
          description: >-
            Image embedding method: description (LLM-generated), base64
            (self-contained), skip (ignore images)
        text_system_prompt:
          anyOf:
            - type: string
            - type: 'null'
          title: Text System Prompt
          description: Custom system prompt for text extraction models (only used by VLMs)
        table_confidence_threshold:
          anyOf:
            - type: number
            - type: 'null'
          title: Table Confidence Threshold
          description: >-
            Minimum confidence threshold for layout detection boxes that are
            labelled as tables
        image_confidence_threshold:
          anyOf:
            - type: number
            - type: 'null'
          title: Image Confidence Threshold
          description: >-
            Minimum confidence threshold for layout detection boxes that are
            labelled as images
        strict_containment_filter:
          anyOf:
            - type: boolean
            - type: 'null'
          title: Strict Containment Filter
          description: >-
            If True, then filters out all boxes that are contained in a larger
            box, if False, it only filters out boxes of the same type (e.g it
            still extracts text boxes from inside images.)
        img_description_prompt:
          anyOf:
            - type: string
            - type: 'null'
          title: Img Description Prompt
          description: >-
            Custom prompt for image description (only used if img_method is
            'description')
        image_description_model:
          anyOf:
            - type: string
            - type: 'null'
          title: Image Description Model
          description: >-
            LLM model to use for image descriptions (only used if img_method is
            'description'). Examples: 'gpt-4o', 'gemini', 'gpt-4o-mini'
        e2e_ocr:
          anyOf:
            - type: string
            - type: 'null'
          title: E2E Ocr
          description: >-
            End-to-end OCR model that performs layout and content extraction
            simultaneously via the SGP API. When specified, bypasses layout
            detection and sends the full page image to this model. Mutually
            exclusive with layout/text_ocr/table_ocr. Example: 'openai/gpt-4o'
        e2e_response_parser:
          anyOf:
            - type: string
            - type: 'null'
          title: E2E Response Parser
          description: >-
            Response parser for the e2e OCR model. Required when e2e_ocr is set
            so the raw model output can be parsed into per-region bounding
            boxes. Example: 'deepseek_ocr2'
        e2e_prompt:
          anyOf:
            - type: string
            - type: 'null'
          title: E2E Prompt
          description: >-
            Custom prompt for end-to-end OCR model (only used when e2e_ocr is
            set)
        enable_table_substructure_recognition:
          type: boolean
          title: Enable Table Substructure Recognition
          description: >-
            Attach per-row/column bounding boxes to table blocks for table-cell
            citations.
          default: false
      type: object
      title: IrisParseEngineOptions
    ReductoChunkingOptions:
      properties:
        chunk_mode:
          $ref: '#/components/schemas/ReductoChunkingMethod'
          description: Chunking method
          default: variable
        chunk_size:
          anyOf:
            - type: integer
            - type: 'null'
          title: Chunk Size
          description: Chunk size
      type: object
      title: ReductoChunkingOptions
    ReductoChunkingMethod:
      type: string
      enum:
        - disabled
        - block
        - page
        - page_sections
        - section
        - variable
      title: ReductoChunkingMethod
      description: Chunking method used for parsing.
  securitySchemes:
    ApiKey:
      type: apiKey
      in: header
      name: x-api-key
      description: API key for authentication
    AccountId:
      type: apiKey
      in: header
      name: x-selected-account-id
      description: Selected Account ID

````