# AUTO-GENERATED tier-A mirror spec — do not edit.
# Source: openapi/runtime/knowledge.yaml (runtime 0.71.2)
#  + policy: openapi/mirror-policies/knowledge.json
# Regenerate: node scripts/generate-mirror-specs.mjs
openapi: 3.0.3
info:
  title: naturali.ai — Knowledge API
  version: 1.0.0
  description: >-
    Knowledge search: one semantic query across a project's ingested documents, returning the
    matching chunks with their scores and provenance — the read side of the retrieval stack an
    agent's knowledge configuration draws on. A fully runtime-backed module — this spec is generated
    verbatim from the runtime's own, re-rooted under /v1/projects/{project_id}. The project in the
    path is authorized by naturali and enforced upstream by the project's scoped credential.


    This module mirrors the upstream runtime verbatim (tier A, #304): paths are the runtime's own
    re-rooted under /v1/projects/{project_id}, and every field, method, status code and error shape
    passes through unchanged. Errors raised by the runtime arrive in its envelope; errors raised by
    naturali itself (authentication, project resolution, an unreachable runtime) use naturali's.
  contact:
    name: naturali.ai
    url: https://naturali.ai
servers:
  - url: "{baseUrl}"
    description: Host of your naturali.ai deployment; every path carries the /v1 prefix.
    variables:
      baseUrl:
        description: Base host URL.
        default: https://api.naturali.ai
tags:
  - name: Knowledge
    description: Unified search across documents and knowledge sources
security:
  - bearerAuth: []
  - oauth2:
      - mcp:access
paths:
  /v1/projects/{project_id}/knowledge/search:
    post:
      tags:
        - Knowledge
      summary: Search knowledge
      description: Searches across documents and memories using hybrid retrieval — a vector query and a
        full-text query per source, fused by reciprocal rank — or by file paths, document IDs,
        memory store IDs, or tags. At least one of `query`, `tags`, `document_paths`,
        `document_ids`, or `memory_store_ids` must be provided.
      operationId: searchKnowledge
      x-iam-action: knowledge:SearchKnowledge
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              additionalProperties: false
              properties:
                query:
                  type: string
                  description: Search query text. Runs a vector query and a full-text query over each store and fuses
                    the rankings; the full-text channel matches exact tokens and phrases, the vector
                    channel everything else.
                  example: customer communication preferences
                min_similarity:
                  type: number
                  description: "Minimum raw cosine `similarity_score` a **vector** candidate must reach to take part
                    in ranking, applied before fusion. Only applies when `query` is provided.
                    Lexical candidates are deliberately exempt: a result that literally contains the
                    searched token is the evidence, and dropping it for a low cosine is the failure
                    hybrid search exists to prevent. Not a floor on `score`, whose fused value
                    encodes rank position rather than similarity."
                  minimum: 0
                  maximum: 1
                  example: 0.5
                rrf_k:
                  type: integer
                  description: The `k` in the reciprocal rank fusion term `1 / (k + rank)`, which sets how steeply a
                    result's contribution decays with its position in each ranked list. A smaller
                    value weights the very top of each list more heavily. Defaults to the
                    deployment's `KNOWLEDGE_RRF_K`, itself 60 by default. Only applies when `query`
                    is provided.
                  minimum: 1
                  example: 60
                recency_half_life_days:
                  type: number
                  description: "Half-life, in days, of a recency decay applied to **memory store** results after
                    fusion: a result's `score` is multiplied by `2 ^ (-age_in_days /
                    recency_half_life_days)`, where age is measured from its `updated_at`. Document
                    results are never decayed. `0` — the default, and the default of the
                    deployment's `KNOWLEDGE_RECENCY_HALF_LIFE_DAYS` — disables the blend entirely;
                    it is a switch, not a lower bound, and sending `0` turns off a deployment-wide
                    decay for this one request. Accepts fractions (`0.5` is twelve hours). How many
                    ranks a given half-life costs depends on `rrf_k`. Only applies when `query` is
                    provided. **No value is known to be safe in general**: every half-life measured
                    against the reference corpus that lifted recency-sensitive queries also cost
                    relevance-sensitive ones, which is why this ships disabled. Measure against your
                    own corpus before setting it — see the Retrieval Quality guide."
                  minimum: 0
                  example: 30
                limit:
                  type: integer
                  description: Maximum number of results to return (default 10). A value above 100 is clamped to 100 —
                    the ceiling bounds the vector scan this one request performs, so a larger
                    `limit` returns everything there is up to that many rows rather than being
                    refused.
                  minimum: 1
                  example: 10
                include_documents:
                  type: boolean
                  default: true
                  description: Set `false` to leave the document store out of this search. A `query` names no store,
                    so it reaches both; the store-specific filters narrow *within* a store rather
                    than choosing between them. This is the switch, and a filter naming the other
                    store never overrides it. `false` for both stores is `400`.
                  example: false
                include_memories:
                  type: boolean
                  default: true
                  description: Set `false` to leave the memory store out of this search. The mirror of
                    `include_documents`, for a caller that wants documents alone.
                  example: false
                memory_store_ids:
                  x-naturali-ref: memory-stores
                  type: array
                  description: Search memories within these specific memory stores
                  items:
                    type: string
                  example:
                    - mstore_V1StGXR8Z5jdHi6B
                document_paths:
                  type: array
                  description: Filter results to documents whose file path starts with one of these prefixes
                  items:
                    type: string
                  example:
                    - /sales/
                    - /hr/
                document_ids:
                  x-naturali-ref: documents
                  type: array
                  description: Filter results to specific document IDs
                  items:
                    type: string
                  example:
                    - doc_V1StGXR8Z5jdHi6B
                tags:
                  description: "Filter results to documents and memories whose `tags` contain every one of these
                    key-value pairs (exact, case-sensitive match). Scopes both stores, so passing it
                    alone searches both — as `query` does. For memories it matches at memory
                    granularity: a memory is returned when its parent memory store's tags match or
                    its own do."
                  allOf:
                    - $ref: "#/components/schemas/TagBag"
                metadata:
                  description: "Filter document results by their `metadata` bag. A document-store filter: a memory
                    carries no such bag, so passing it alone searches documents, as `document_paths`
                    does."
                  allOf:
                    - $ref: "#/components/schemas/MetadataFilter"
      responses:
        "200":
          description: Search results
          content:
            application/json:
              schema:
                type: object
                required:
                  - results
                properties:
                  results:
                    type: array
                    items:
                      $ref: "#/components/schemas/KnowledgeResult"
        "400":
          description: Bad request — at least one search parameter is required
          content:
            application/json:
              schema:
                $ref: "#/components/schemas/ErrorResponse"
        "401":
          description: Unauthorized
          content:
            application/json:
              schema:
                $ref: "#/components/schemas/ErrorResponse"
        "403":
          description: Forbidden
          content:
            application/json:
              schema:
                $ref: "#/components/schemas/ErrorResponse"
    parameters:
      - $ref: "#/components/parameters/ProjectId"
components:
  schemas:
    KnowledgeResult:
      oneOf:
        - $ref: "#/components/schemas/DocumentKnowledgeResult"
        - $ref: "#/components/schemas/MemoryKnowledgeResult"
      discriminator:
        propertyName: source_type
        mapping:
          document: "#/components/schemas/DocumentKnowledgeResult"
          memory: "#/components/schemas/MemoryKnowledgeResult"
    DocumentKnowledgeResult:
      type: object
      required:
        - source_type
        - document_id
        - content
        - created_at
        - updated_at
      properties:
        source_type:
          type: string
          enum:
            - document
          description: The type of knowledge source this result comes from
          example: document
        document_id:
          x-naturali-ref: documents
          type: string
          description: Public ID of the document
          example: doc_V1StGXR8Z5jdHi6B
        document_version:
          type: integer
          description: The document version the chunk belongs to. Cite it to record which text was read; the
            document's versions keep that text after it changes.
          example: 3
        chunk_id:
          type: string
          description: Public ID of the document chunk that matched the query
          example: dchunk_V1StGXR8Z5jdHi6B
        page:
          type: integer
          nullable: true
          description: Page number within the source PDF (1-indexed). Null for plain-text documents.
          example: 3
        file_id:
          x-naturali-ref: files
          type: string
          description: Public ID of the underlying file
          example: file_V1StGXR8Z5jdHi6B
        project_id:
          x-naturali-ref: projects
          type: string
          description: Public ID of the project the document belongs to
          example: proj_V1StGXR8Z5jdHi6B
        path:
          type: string
          description: Logical path of the file within the project
          example: /sales/policies.txt
        filename:
          type: string
          description: Filename of the underlying file
          example: policies.txt
        size:
          type: integer
          description: File size in bytes
          example: 1024
        title:
          type: string
          description: Document title
          example: Sales Communication Policy
        metadata:
          description: Arbitrary metadata attached to the document, returned verbatim in the casing it was
            written with at create/update time (e.g. a key written as `strapiDocumentId` is returned
            as `strapiDocumentId`, not `strapi_document_id`) — it is not converted between
            snake_case and camelCase like other response fields.
          allOf:
            - $ref: "#/components/schemas/MetadataBag"
        tags:
          $ref: "#/components/schemas/TagBag"
        content:
          type: string
          nullable: true
          description: Full text content of the document
        score:
          type: number
          description: "Reciprocal-rank-fusion relevance ranking — higher is better. The **ordering** it
            produces is the contract; the absolute value is not, is deliberately not rescaled into
            0–1, and the formula behind it may change. Results are sorted by it. Nothing filters on
            it: `min_similarity` filters `similarity_score`. Only present when `query` was provided.
            Use `similarity_score` when you need the raw cosine value."
          example: 0.0328
        signals:
          type: object
          description: "Which retrieval channels ranked this result before fusion, and its 1-based position in
            that channel's own ordering. A channel that did not return the result is absent, so
            presence reads as \"this channel found it\". `score` says where the result landed; this
            says how it got there — `{ \"lexical\": 1 }` is a pure token hit, `{ \"vector\": 2,
            \"lexical\": 1 }` a result both channels found and fusion promoted. Only present when
            `query` was provided; a channel that degraded is absent from every result. Diagnostic:
            nothing filters or sorts on it."
          properties:
            vector:
              type: integer
              minimum: 1
              description: Rank in the vector (cosine) channel, best first.
              example: 2
            lexical:
              type: integer
              minimum: 1
              description: Rank in the lexical (full-text) channel, best first.
              example: 1
          example:
            vector: 2
            lexical: 1
        similarity_score:
          type: number
          description: Raw cosine similarity (0–1) between the query and this result. Pinned to that meaning —
            unlike `score`, it is never redefined — and populated on every result of a `query`
            search, a lexical-only hit included. Absent only when the embedding provider was
            unreachable and the search answered from the lexical channel alone, where there is no
            query vector to measure against.
          minimum: 0
          maximum: 1
          example: 0.82
        created_at:
          type: string
          format: date-time
          description: Creation timestamp
        updated_at:
          type: string
          format: date-time
          description: Last updated timestamp
    MemoryKnowledgeResult:
      type: object
      required:
        - source_type
        - memory_id
        - memory_store_id
        - memory_store_name
        - content
        - created_at
        - updated_at
      properties:
        source_type:
          type: string
          enum:
            - memory
          description: The type of knowledge source this result comes from
          example: memory
        memory_id:
          type: string
          description: Public ID of the memory
          example: mem_V1StGXR8Z5jdHi6B
        memory_store_id:
          x-naturali-ref: memory-stores
          type: string
          description: Public ID of the parent memory store
          example: mstore_V1StGXR8Z5jdHi6B
        memory_store_name:
          type: string
          description: Human-readable name of the parent memory store
          example: Customer Preferences
        content:
          type: string
          description: Text content of the memory
        score:
          type: number
          description: "Reciprocal-rank-fusion relevance ranking — higher is better. The **ordering** it
            produces is the contract; the absolute value is not, is deliberately not rescaled into
            0–1, and the formula behind it may change. Results are sorted by it. Nothing filters on
            it: `min_similarity` filters `similarity_score`. Only present when `query` was provided.
            Use `similarity_score` when you need the raw cosine value."
          example: 0.0161
        signals:
          type: object
          description: "Which retrieval channels ranked this result before fusion, and its 1-based position in
            that channel's own ordering. A channel that did not return the result is absent, so
            presence reads as \"this channel found it\". `score` says where the result landed; this
            says how it got there — `{ \"lexical\": 1 }` is a pure token hit, `{ \"vector\": 2,
            \"lexical\": 1 }` a result both channels found and fusion promoted. Only present when
            `query` was provided; a channel that degraded is absent from every result. Diagnostic:
            nothing filters or sorts on it."
          properties:
            vector:
              type: integer
              minimum: 1
              description: Rank in the vector (cosine) channel, best first.
              example: 2
            lexical:
              type: integer
              minimum: 1
              description: Rank in the lexical (full-text) channel, best first.
              example: 1
          example:
            vector: 2
            lexical: 1
        similarity_score:
          type: number
          description: Raw cosine similarity (0–1) between the query and this result. Pinned to that meaning —
            unlike `score`, it is never redefined — and populated on every result of a `query`
            search, a lexical-only hit included. Absent only when the embedding provider was
            unreachable and the search answered from the lexical channel alone, where there is no
            query vector to measure against.
          minimum: 0
          maximum: 1
          example: 0.79
        created_at:
          type: string
          format: date-time
          description: Creation timestamp
        updated_at:
          type: string
          format: date-time
          description: Last updated timestamp
    TagBag:
      type: object
      additionalProperties:
        type: string
      x-cli-flag-name: tags
      description: >-
        Key-value labels on a resource. A flat object of string values — an array, a nested object
        or a number is rejected with `400 VALIDATION_FAILED`, never coerced. Keys are opaque and
        stored verbatim, so `cost_center` and `costCenter` are two different tags. Matched by JSONB
        containment wherever tags are read: the `?tags=` filter and knowledge search.


        Keys beginning `system.` are reserved: the platform writes them to record which
        conversation, actor, agent and role a row came from, and a write naming one is refused with
        `400 RESERVED_TAG_KEY`. They are read and filtered like any other tag.


        The bag is bounded, because every pair reaches the IAM context of every access check on the
        resource: at most 50 keys, each key at most 128 characters and each value at most 256. A
        write past a bound — including a merge that would grow the stored bag past the key count —
        is `400 VALIDATION_FAILED` with `meta.limit` naming the bound it crossed. `system.*` keys
        are the platform's and do not count against the 50.
      example:
        team: finance
        env: prod
    MetadataFilter:
      type: object
      description: >-
        Narrows results by the `metadata` bag. Each key is a field; its value is either a value to
        match exactly, or an object naming one or more operators.


        Equality and `in` match the stored value exactly, so `3` and `"3"` are different filters.


        `gt`, `gte`, `lt` and `lte` order the field against the operand, and the operand's own JSON
        type says which comparison is meant: a number orders numerically, a string
        lexicographically. They work on any field, in any scope. A document whose field holds
        another type is excluded from the comparison rather than erroring on it, so a range never
        fails on a bag that happens to hold text where another holds a number. An operand that is
        neither a number nor a string — a boolean, `null`, a list, an object — has no ordering and
        is `400 VALIDATION_FAILED`, naming the field in `meta.field`.
      additionalProperties: true
      example:
        quarter: Q1
        revision:
          gte: 3
          lt: 11
        status:
          in:
            - draft
            - final
    ErrorResponse:
      type: object
      required:
        - error
      properties:
        error:
          type: object
          description: Structured error. Every error response uses this shape, so `code` can be read without
            first checking the type of `error`.
          required:
            - code
            - message
          properties:
            code:
              type: string
              description: "Machine-readable error code: lower_snake for an error naturali raises
                (`access_denied`), UPPER_SNAKE for one the runtime reports (`RESOURCE_NOT_FOUND`)."
              example: access_denied
            message:
              type: string
              description: Human-readable explanation.
              example: Your role in this project does not carry this action.
            details:
              type: object
              additionalProperties: true
              description: Structured context for an error naturali raises, such as the `resource` and `limit` of
                a `plan_limit_reached`.
            meta:
              type: object
              description: Structured context for an error the runtime reports.
    MetadataBag:
      type: object
      description: >-
        Caller-owned annotations on a resource, stored as the object they were written as: the types
        a value was written with are the types a read returns, so a filter can ask an ordering
        question about a number. Unlike other body fields, keys are stored and returned verbatim in
        the casing supplied — they are not converted between snake_case and camelCase.


        No key is reserved, and that is the point: every piece of state the platform owns lives in
        its own typed column, so nothing written here reaches platform state. The platform never
        reads the bag — it is not an IAM context, not a policy input and not part of a prompt —
        which is what separates it from a tag bag.
      example:
        author: John
        revision: 2
  parameters:
    ProjectId:
      name: project_id
      in: path
      required: true
      description: Project public ID (proj_ prefix).
      schema:
        type: string
        example: proj_V1StGXR8Z5jdHi6B
  securitySchemes:
    bearerAuth:
      type: http
      scheme: bearer
      description: A naturali API key (nat_sk_…) or a session JWT.
    oauth2:
      type: oauth2
      description: "A connected app's OAuth access token, issued by this API's authorization server
        (discovery: /.well-known/oauth-authorization-server). Its one scope carries every operation,
        confined to the projects the user chose when approving the app."
      flows:
        authorizationCode:
          authorizationUrl: https://api.naturali.ai/authorize
          tokenUrl: https://api.naturali.ai/token
          refreshUrl: https://api.naturali.ai/token
          scopes:
            mcp:access: Every operation this API serves, on the projects the grant covers.
