openapi: 3.0.3
info:
  title: QuantSearch API
  description: |
    AI-powered search for your website. This API allows you to:
    - Search your indexed content
    - Get AI-generated answers
    - Manage sites and crawl jobs
    - Create search groups for federated search
    - Ingest content programmatically
    
    ## Authentication
    
    API access requires a Pro or Enterprise subscription. Generate API keys in your dashboard.
    
    Include your API key in the `Authorization` header:
    ```
    Authorization: Bearer YOUR_API_KEY
    ```
    
    ## Public Endpoints
    
    Public search and chat endpoints don't require authentication, but:
    - Public access must be enabled for your site/group
    - The request origin must match your allowed domains
    - Rate limits apply per IP address
    
    ## Base URLs
    
    - **SaaS API**: `https://www.quantsearch.ai/api` - Authenticated endpoints
    - **Public Search CDN**: `https://cdn.quantsearch.ai/v1` - Public widget/search
  version: 1.2.0
  contact:
    name: QuantSearch Support
    email: support@quantsearch.ai
    url: https://quantsearch.ai/docs
  license:
    name: Proprietary
    url: https://quantsearch.ai/terms

servers:
  - url: https://www.quantsearch.ai/api
    description: SaaS API (authenticated endpoints)
  - url: https://cdn.quantsearch.ai/v1
    description: Public Search CDN (widget, public search/chat)

tags:
  - name: Public Search
    description: Unauthenticated search endpoints (origin-validated)
  - name: Sites
    description: Manage AI Search sites
  - name: Groups
    description: Search groups for federated search (Pro+)
  - name: Content
    description: Ingest and manage indexed content
  - name: Crawler
    description: Manage crawler jobs
  - name: Analytics
    description: Search query, chat, and indexed-content usage at the org and site level

paths:
  # =============================================================================
  # PUBLIC ENDPOINTS (No Auth - Origin Validated)
  # =============================================================================
  
  /public/sites/{siteId}/search:
    get:
      tags: [Public Search]
      summary: Search site content
      description: |
        Search your indexed content. Results are cached by CloudFront for performance.
        
        Public access must be enabled for your site, and the request origin must be allowed.
      operationId: publicSearchGet
      servers:
        - url: https://cdn.quantsearch.ai/v1
      parameters:
        - name: siteId
          in: path
          required: true
          schema:
            type: string
          description: Your site ID
        - name: q
          in: query
          required: true
          schema:
            type: string
          description: Search query
          example: how to reset password
        - name: limit
          in: query
          schema:
            type: integer
            default: 10
            maximum: 100
          description: Maximum results to return
        - name: minScore
          in: query
          schema:
            type: number
            default: 0.3
          description: Minimum relevance score (0-1)
      responses:
        '200':
          description: Search results
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/SearchResponse'
        '400':
          $ref: '#/components/responses/BadRequest'
        '403':
          $ref: '#/components/responses/Forbidden'
        '429':
          $ref: '#/components/responses/RateLimited'
    
    post:
      tags: [Public Search]
      summary: Search site content (POST)
      description: Same as GET but accepts JSON body. Useful for complex queries.
      operationId: publicSearchPost
      servers:
        - url: https://cdn.quantsearch.ai/v1
      parameters:
        - name: siteId
          in: path
          required: true
          schema:
            type: string
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [query]
              properties:
                query:
                  type: string
                  description: Search query
                limit:
                  type: integer
                  default: 10
                  maximum: 100
                minScore:
                  type: number
                  default: 0.3
      responses:
        '200':
          description: Search results
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/SearchResponse'

  /public/sites/{siteId}/chat:
    post:
      tags: [Public Search]
      summary: Get AI-generated answer
      description: |
        Ask a question and get an AI-generated answer based on your indexed content.
        
        Optionally include a `sessionId` for conversation continuity.
      operationId: publicChat
      servers:
        - url: https://cdn.quantsearch.ai/v1
      parameters:
        - name: siteId
          in: path
          required: true
          schema:
            type: string
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [message]
              properties:
                message:
                  type: string
                  description: User question
                  example: How do I reset my password?
                sessionId:
                  type: string
                  description: Optional session ID for conversation history
      responses:
        '200':
          description: AI-generated answer
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ChatResponse'
        '403':
          $ref: '#/components/responses/Forbidden'
        '429':
          $ref: '#/components/responses/RateLimited'

  /public/groups/{groupId}/search:
    get:
      tags: [Public Search, Groups]
      summary: Federated search across group
      description: |
        Search across all sites in a search group. Returns unified results ranked by relevance.
        
        Requires the group to have public access enabled.
      operationId: publicGroupSearchGet
      servers:
        - url: https://cdn.quantsearch.ai/v1
      parameters:
        - name: groupId
          in: path
          required: true
          schema:
            type: string
          description: Search group ID
        - name: q
          in: query
          required: true
          schema:
            type: string
          description: Search query
        - name: limit
          in: query
          schema:
            type: integer
            default: 10
            maximum: 50
        - name: minScore
          in: query
          schema:
            type: number
            default: 0.3
      responses:
        '200':
          description: Federated search results
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/GroupSearchResponse'
        '403':
          $ref: '#/components/responses/Forbidden'
        '404':
          $ref: '#/components/responses/NotFound'
        '429':
          $ref: '#/components/responses/RateLimited'

    post:
      tags: [Public Search, Groups]
      summary: Federated search across group (POST)
      operationId: publicGroupSearchPost
      servers:
        - url: https://cdn.quantsearch.ai/v1
      parameters:
        - name: groupId
          in: path
          required: true
          schema:
            type: string
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [query]
              properties:
                query:
                  type: string
                limit:
                  type: integer
                  default: 10
                minScore:
                  type: number
                  default: 0.3
      responses:
        '200':
          description: Federated search results
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/GroupSearchResponse'

  # =============================================================================
  # SITES (Authenticated)
  # =============================================================================

  /sites:
    get:
      tags: [Sites]
      summary: List sites
      description: Get all AI Search sites for your organization.
      operationId: listSites
      security:
        - BearerAuth: []
      responses:
        '200':
          description: List of sites
          content:
            application/json:
              schema:
                type: object
                properties:
                  sites:
                    type: array
                    items:
                      $ref: '#/components/schemas/Site'
        '401':
          $ref: '#/components/responses/Unauthorized'

    post:
      tags: [Sites]
      summary: Create site
      description: Create a new AI Search site.
      operationId: createSite
      security:
        - BearerAuth: []
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/CreateSiteRequest'
      responses:
        '201':
          description: Site created
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/Site'
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'

  /sites/{siteId}:
    get:
      tags: [Sites]
      summary: Get site
      description: Get details of a specific site.
      operationId: getSite
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '200':
          description: Site details
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/Site'
        '404':
          $ref: '#/components/responses/NotFound'

    put:
      tags: [Sites]
      summary: Update site
      description: |
        Update site configuration. Pass only the fields you want to change;
        others are left as-is. Returns `{success: true}` on success — call
        `GET /sites/{siteId}` afterwards if you need the updated record.
      operationId: updateSite
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/UpdateSiteRequest'
      responses:
        '200':
          description: Site updated
          content:
            application/json:
              schema:
                type: object
                properties:
                  success:
                    type: boolean

    delete:
      tags: [Sites]
      summary: Delete site
      description: Delete a site and all its indexed content.
      operationId: deleteSite
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '200':
          description: Site deleted
          content:
            application/json:
              schema:
                type: object
                properties:
                  success:
                    type: boolean

  /sites/{siteId}/crawl:
    post:
      tags: [Sites, Crawler]
      summary: Start crawl
      description: Start a new crawl job for the site.
      operationId: startCrawl
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '201':
          description: Crawl started
          content:
            application/json:
              schema:
                type: object
                properties:
                  success:
                    type: boolean
                  jobId:
                    type: string
                    format: uuid
                  message:
                    type: string
        '409':
          description: Crawl already in progress

  /sites/{siteId}/crawl-urls:
    get:
      tags: [Sites, Crawler]
      summary: Get saved crawl scope
      description: >-
        The site's saved crawl scope — a persistent list of the only URLs the
        site should crawl. Empty means the whole site is crawled.
      operationId: getCrawlUrls
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '200':
          description: The saved scope
          content:
            application/json:
              schema:
                type: object
                properties:
                  urls:
                    type: array
                    items:
                      type: string
                  count:
                    type: integer
    put:
      tags: [Sites, Crawler]
      summary: Replace saved crawl scope
      description: >-
        Replace the saved crawl scope. URLs are filtered to the site's own
        origin; off-origin URLs are dropped and reported in `rejected`. There is
        no cap on scope size beyond a generous safety limit.
      operationId: setCrawlUrls
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [urls]
              properties:
                urls:
                  type: array
                  items:
                    type: string
      responses:
        '200':
          description: Scope saved
          content:
            application/json:
              schema:
                type: object
                properties:
                  urls:
                    type: array
                    items:
                      type: string
                  count:
                    type: integer
                  rejected:
                    type: integer
                  warning:
                    type: string
        '400':
          description: Body is not an array, exceeds the size limit, or the site base URL is not crawlable
        '503':
          description: The site's own address could not be verified just now (transient); nothing was changed. Retry.
    delete:
      tags: [Sites, Crawler]
      summary: Clear saved crawl scope
      operationId: deleteCrawlUrls
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '200':
          description: Scope cleared

  /sites/{siteId}/schedules:
    get:
      tags: [Sites, Crawler]
      summary: List crawl schedules
      operationId: listSchedules
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '200':
          description: The site's schedules (at most one)
          content:
            application/json:
              schema:
                type: object
                properties:
                  schedules:
                    type: array
                    items:
                      type: object
    post:
      tags: [Sites, Crawler]
      summary: Create or replace the crawl schedule
      description: >-
        Set a recurring crawl. Frequency is plan-gated (free: none; pro: daily
        or weekly; enterprise: hourly, daily or weekly). The first run is one
        interval after creation, never immediately.
      operationId: setSchedule
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [frequency]
              properties:
                frequency:
                  type: string
                  enum: [hourly, daily, weekly]
      responses:
        '201':
          description: Schedule created
          content:
            application/json:
              schema:
                type: object
                properties:
                  id:
                    type: string
                    format: uuid
                  frequency:
                    type: string
                  enabled:
                    type: boolean
        '400':
          description: Invalid or missing frequency
        '403':
          description: The requested frequency is not available on the org's plan
    delete:
      tags: [Sites, Crawler]
      summary: Delete the crawl schedule
      operationId: deleteSchedule
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '200':
          description: Schedule deleted

  /sites/{siteId}/purge:
    post:
      tags: [Sites, Content]
      summary: Purge site index
      description: Delete all indexed content for a site (the site itself is preserved — only the index is wiped).
      operationId: purgeSite
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '200':
          description: Index purged
          content:
            application/json:
              schema:
                type: object
                properties:
                  success:
                    type: boolean
                  deletedCount:
                    type: integer

  /sites/{siteId}/pages:
    get:
      tags: [Sites, Content]
      summary: List indexed pages
      description: List pages indexed for this site with pagination.
      operationId: listPages
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
        - name: limit
          in: query
          schema:
            type: integer
            default: 50
            maximum: 200
        - name: cursor
          in: query
          schema:
            type: string
          description: Pagination cursor (URL of last item)
        - name: search
          in: query
          schema:
            type: string
          description: Filter by title or URL
      responses:
        '200':
          description: List of indexed pages
          content:
            application/json:
              schema:
                type: object
                properties:
                  items:
                    type: array
                    items:
                      $ref: '#/components/schemas/IndexedPage'
                  nextCursor:
                    type: string
                  hasMore:
                    type: boolean

    post:
      tags: [Content]
      summary: Ingest content
      description: |
        Ingest content directly without crawling. Useful for:
        - CMS integrations
        - Dynamic content
        - Bulk imports
        
        Content is processed with AI to extract metadata and generate embeddings.
      operationId: ingestPages
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [pages]
              properties:
                pages:
                  type: array
                  maxItems: 100
                  items:
                    $ref: '#/components/schemas/PageContent'
      responses:
        '200':
          description: Content ingested
          content:
            application/json:
              schema:
                type: object
                properties:
                  processed:
                    type: integer
                  skipped:
                    type: integer
                  results:
                    type: array
                    items:
                      type: object
                      properties:
                        url:
                          type: string
                        status:
                          type: string
                          enum: [success, skipped, error]
                        error:
                          type: string

    delete:
      tags: [Content]
      summary: Delete pages
      description: Delete specific pages from the index by URL or pattern.
      operationId: deletePages
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              properties:
                urls:
                  type: array
                  items:
                    type: string
                  description: Exact URLs to delete
                patterns:
                  type: array
                  items:
                    type: string
                  description: URL patterns with wildcards (*)
                keys:
                  type: array
                  items:
                    type: string
                  description: Document keys to delete
      responses:
        '200':
          description: |
            Pages deleted. `deletedPages` is what was actually removed and can be
            lower than `requestedUrls` -- deleting a URL that is not in the index
            succeeds and removes nothing. Check `deletedPages`, not the status
            code, to know whether anything changed.
          content:
            application/json:
              schema:
                type: object
                properties:
                  requestedUrls:
                    type: integer
                  deletedPages:
                    type: integer
                  deletedChunks:
                    type: integer

  /sites/{siteId}/index-pages:
    get:
      tags: [Sites, Content]
      summary: List indexed documents
      description: |
        List documents currently in the site's vector index, with SQL-backed
        pagination. Use to verify an ingest or delete landed.
      operationId: listIndexPages
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
        - name: limit
          in: query
          schema:
            type: integer
            default: 50
        - name: cursor
          in: query
          schema:
            type: string
      responses:
        '200':
          description: Indexed documents
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'
    delete:
      tags: [Content]
      summary: Delete indexed documents by URL
      description: |
        Remove documents from the index by exact URL. Takes an array, so one
        request removes many. `deleted` is what was actually removed and can be
        lower than `requested`, because deleting a URL that is not indexed
        succeeds and removes nothing.
      operationId: deleteIndexPages
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [urls]
              properties:
                urls:
                  type: array
                  items:
                    type: string
      responses:
        '200':
          description: Deletion result
          content:
            application/json:
              schema:
                type: object
                properties:
                  success:
                    type: boolean
                  requested:
                    type: integer
                  deleted:
                    type: integer
                  deletedChunks:
                    type: integer
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'

  /sites/{siteId}/reviews:
    get:
      tags: [Sites, Content]
      summary: List held content removals
      description: |
        Stale-content removals held for approval. When a crawl finds that a large
        share of a site's pages have disappeared, the removal is held rather than
        applied, and surfaced here for a human to approve or reject.
      operationId: listContentRemovals
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
      responses:
        '200':
          description: Held removals, newest first
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'

  /sites/{siteId}/reviews/{reviewId}:
    get:
      tags: [Sites, Content]
      summary: Get one held removal
      description: |
        Includes `urls` (the pages that would be removed) and `urlsAvailable`.
        A null `urls` with `urlsAvailable: false` means the list could not be
        read -- approval is refused in that state, because approving without
        seeing the pages means deleting them unseen.
      operationId: getContentRemoval
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
        - name: reviewId
          in: path
          required: true
          schema:
            type: string
      responses:
        '200':
          description: The held removal
        '404':
          $ref: '#/components/responses/NotFound'

  /sites/{siteId}/reviews/{reviewId}/approve:
    post:
      tags: [Sites, Content]
      summary: Approve a held removal
      description: |
        Launches a scoped sweep that RE-PROBES every listed URL and deletes only
        those still confirmed gone; pages that have come back are retained.
        Requires the `admin` org role. A `409` means the review changed while you
        were looking at it (or a sweep is already running) -- reload and re-check
        rather than retrying blindly.
      operationId: approveContentRemoval
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
        - name: reviewId
          in: path
          required: true
          schema:
            type: string
      responses:
        '200':
          description: Approved; the sweep has been launched
        '400':
          description: Refused (URL list unreadable, address mismatch, or not pending)
        '403':
          description: Requires the admin org role
        '409':
          description: Conflict -- reload and re-check

  /sites/{siteId}/reviews/{reviewId}/reject:
    post:
      tags: [Sites, Content]
      summary: Reject a held removal
      description: Nothing is deleted. Requires the `admin` org role.
      operationId: rejectContentRemoval
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/SiteId'
        - name: reviewId
          in: path
          required: true
          schema:
            type: string
      responses:
        '200':
          description: Rejected; nothing was removed
        '403':
          description: Requires the admin org role
        '409':
          description: Conflict -- reload and re-check

  # =============================================================================
  # SEARCH GROUPS (Pro+ Feature)
  # =============================================================================

  /groups:
    get:
      tags: [Groups]
      summary: List search groups
      description: List all search groups for your organization.
      operationId: listGroups
      security:
        - BearerAuth: []
      responses:
        '200':
          description: List of groups
          content:
            application/json:
              schema:
                type: object
                properties:
                  groups:
                    type: array
                    items:
                      $ref: '#/components/schemas/SearchGroup'

    post:
      tags: [Groups]
      summary: Create search group
      description: |
        Create a new search group for federated search across multiple sites.
        Requires Pro or Enterprise subscription.
      operationId: createGroup
      security:
        - BearerAuth: []
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/CreateGroupRequest'
      responses:
        '201':
          description: Group created
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/SearchGroup'
        '402':
          description: Pro subscription required

  /groups/{groupId}:
    get:
      tags: [Groups]
      summary: Get search group
      operationId: getGroup
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/GroupId'
      responses:
        '200':
          description: Group details
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/SearchGroup'
        '404':
          $ref: '#/components/responses/NotFound'

    put:
      tags: [Groups]
      summary: Update search group
      operationId: updateGroup
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/GroupId'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/UpdateGroupRequest'
      responses:
        '200':
          description: Group updated
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/SearchGroup'

    delete:
      tags: [Groups]
      summary: Delete search group
      operationId: deleteGroup
      security:
        - BearerAuth: []
      parameters:
        - $ref: '#/components/parameters/GroupId'
      responses:
        '204':
          description: Group deleted

  # =============================================================================
  # JOBS
  # =============================================================================

  /jobs/{jobId}:
    get:
      tags: [Crawler]
      summary: Get crawl job status
      operationId: getJob
      security:
        - BearerAuth: []
      parameters:
        - name: jobId
          in: path
          required: true
          schema:
            type: string
      responses:
        '200':
          description: Job details
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/CrawlJob'
        '404':
          $ref: '#/components/responses/NotFound'

    delete:
      tags: [Crawler]
      summary: Cancel crawl job
      operationId: cancelJob
      security:
        - BearerAuth: []
      parameters:
        - name: jobId
          in: path
          required: true
          schema:
            type: string
      responses:
        '200':
          description: Job cancelled
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/CrawlJob'

  # =============================================================================
  # ANALYTICS (Authenticated)
  # =============================================================================

  /analytics:
    get:
      tags: [Analytics]
      summary: Org-wide search analytics
      description: |
        Aggregated analytics across the entire organization (or one site if
        `siteId` is supplied). Includes total search/chat counts, top queries,
        per-site breakdown, and a daily time-series.

        Backed by upstream telemetry; figures may lag real-time activity by a
        few minutes.
      operationId: getAnalytics
      security:
        - BearerAuth: []
      parameters:
        - name: range
          in: query
          required: false
          description: Time window for the analytics (e.g. `7d`, `30d`, `90d`).
          schema:
            type: string
            default: "30d"
        - name: siteId
          in: query
          required: false
          description: Optional. If supplied, scope all metrics to one site.
          schema:
            type: string
            format: uuid
      responses:
        '200':
          description: Analytics rollup
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/AnalyticsResponse'
        '401':
          $ref: '#/components/responses/Unauthorized'

  /usage:
    get:
      tags: [Analytics]
      summary: Org-level monthly usage and quota
      description: |
        Monthly usage figures used for billing and quota tracking. Returns AI
        token consumption (split between ingest and chat), vector-DB stats,
        and locally-tracked search/chat query counts.

        Use this for billing dashboards. Use `/analytics` for product analytics.
      operationId: getUsage
      security:
        - BearerAuth: []
      parameters:
        - name: month
          in: query
          required: false
          description: Month in `YYYY-MM` format. Defaults to the current month (UTC).
          schema:
            type: string
            pattern: '^\d{4}-\d{2}$'
            example: "2026-05"
      responses:
        '200':
          description: Usage rollup for the month
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/UsageResponse'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '500':
          description: Failed to fetch usage from upstream
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/Error'

  /sites/{siteId}/usage:
    get:
      tags: [Analytics]
      summary: Per-site search analytics
      description: |
        Search and chat counts for a single site over the requested range,
        plus a daily time-series. Always returns a stable shape; if the
        upstream telemetry source is unavailable, all counts return as 0
        rather than erroring.
      operationId: getSiteUsage
      security:
        - BearerAuth: []
      parameters:
        - name: siteId
          in: path
          required: true
          schema:
            type: string
            format: uuid
        - name: range
          in: query
          required: false
          schema:
            type: string
            default: "30d"
      responses:
        '200':
          description: Per-site usage
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/SiteUsageResponse'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'

  /sites/{siteId}/top-queries:
    get:
      tags: [Analytics]
      summary: Top search queries for a site
      description: |
        The most-searched queries for one site over the requested range,
        sorted descending by count. Useful for understanding user intent
        and content gaps. The org-level `/analytics` endpoint also
        aggregates this data across all sites if you need a single view.
      operationId: getSiteTopQueries
      security:
        - BearerAuth: []
      parameters:
        - name: siteId
          in: path
          required: true
          schema:
            type: string
            format: uuid
        - name: range
          in: query
          required: false
          schema:
            type: string
            default: "30d"
        - name: limit
          in: query
          required: false
          description: Max number of queries to return. Capped at 100.
          schema:
            type: integer
            default: 20
            minimum: 1
            maximum: 100
      responses:
        '200':
          description: Top queries
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/SiteTopQueriesResponse'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '404':
          $ref: '#/components/responses/NotFound'

components:
  securitySchemes:
    BearerAuth:
      type: http
      scheme: bearer
      description: API key from your dashboard

  parameters:
    SiteId:
      name: siteId
      in: path
      required: true
      schema:
        type: string
      description: Site ID
    
    GroupId:
      name: groupId
      in: path
      required: true
      schema:
        type: string
      description: Search group ID

  schemas:
    SearchResponse:
      type: object
      properties:
        query:
          type: string
        results:
          type: array
          items:
            $ref: '#/components/schemas/SearchResult'
        totalResults:
          type: integer
        timingMs:
          type: integer

    SearchResult:
      type: object
      properties:
        url:
          type: string
        title:
          type: string
        snippet:
          type: string
        score:
          type: number
        metadata:
          type: object
          properties:
            summary:
              type: string
            tags:
              type: array
              items:
                type: string
            publishedAt:
              type: string
              format: date
              description: Content publication date (if detected)
            crawledAt:
              type: string
              format: date
              description: When content was last indexed

    GroupSearchResponse:
      type: object
      properties:
        query:
          type: string
        groupId:
          type: string
        results:
          type: array
          items:
            allOf:
              - $ref: '#/components/schemas/SearchResult'
              - type: object
                properties:
                  siteId:
                    type: string
                  siteName:
                    type: string
        totalResults:
          type: integer
        timingMs:
          type: integer

    ChatResponse:
      type: object
      properties:
        response:
          type: string
        sources:
          type: array
          items:
            type: object
            properties:
              url:
                type: string
              title:
                type: string
              relevance:
                type: number
        sessionId:
          type: string
        tokensUsed:
          type: integer

    Site:
      type: object
      properties:
        siteId:
          type: string
        name:
          type: string
        baseUrl:
          type: string
        domain:
          type: string
          description: Domain for URL prefixing in search results (useful for multi-site)
        status:
          type: string
          enum: [pending, crawling, ready, error]
        pageCount:
          type: integer
        lastCrawledAt:
          type: integer
        createdAt:
          type: integer
        publicAccess:
          $ref: '#/components/schemas/PublicAccess'
        rateLimits:
          $ref: '#/components/schemas/RateLimits'
        metadataSchema:
          type: array
          description: Custom metadata fields for filtering/faceting.
          items:
            $ref: '#/components/schemas/MetadataField'
        tagBoosts:
          type: array
          description: Per-tag score multipliers applied at query time. No reindex needed.
          items:
            $ref: '#/components/schemas/TagBoost'
        urlBoosts:
          type: array
          description: Per-URL-prefix score multipliers applied at query time. No reindex needed.
          items:
            $ref: '#/components/schemas/UrlBoost'

    CreateSiteRequest:
      type: object
      required: [name, baseUrl]
      properties:
        name:
          type: string
        baseUrl:
          type: string
        domain:
          type: string
          description: Optional domain for URL prefixing
        crawlerConfig:
          $ref: '#/components/schemas/CrawlerConfig'

    UpdateSiteRequest:
      type: object
      properties:
        name:
          type: string
        domain:
          type: string
        publicAccess:
          $ref: '#/components/schemas/PublicAccess'
        rateLimits:
          $ref: '#/components/schemas/RateLimits'
        crawlerConfig:
          $ref: '#/components/schemas/CrawlerConfig'
        metadataSchema:
          type: array
          items:
            $ref: '#/components/schemas/MetadataField'
        tagBoosts:
          type: array
          items:
            $ref: '#/components/schemas/TagBoost'
        urlBoosts:
          type: array
          items:
            $ref: '#/components/schemas/UrlBoost'

    MetadataField:
      type: object
      required: [name, type]
      properties:
        name:
          type: string
          description: Field name (alphanumeric + underscore).
        type:
          type: string
          enum: [string, number, date]
        label:
          type: string
          description: Human-readable display label for facets.
        facet:
          type: boolean
          description: When true, the field is exposed via the /facets endpoint.

    TagBoost:
      type: object
      required: [tag, boost]
      description: |
        Multiply each result's score by `boost` if the result has `tag` in its
        `metadata.tags` array. Values >1 boost, <1 deboost. Applied at query
        time before URL deduplication; affects both the rendered results list
        and the AI summary's source selection. Negative or zero boosts are
        silently skipped.
      properties:
        tag:
          type: string
          example: type:product
        boost:
          type: number
          format: float
          minimum: 0
          example: 2.0

    UrlBoost:
      type: object
      required: [urlPrefix, boost]
      description: |
        Multiply each result's score by `boost` when the result URL's path
        starts with `urlPrefix`. Same semantics as TagBoost otherwise.
      properties:
        urlPrefix:
          type: string
          description: Path prefix (e.g. `/products`) or full URL prefix.
          example: /products
        boost:
          type: number
          format: float
          minimum: 0
          example: 2.0

    SearchGroup:
      type: object
      properties:
        groupId:
          type: string
        name:
          type: string
        description:
          type: string
        siteIds:
          type: array
          items:
            type: string
          description: Sites included in this group
        publicAccess:
          $ref: '#/components/schemas/PublicAccess'
        rateLimits:
          $ref: '#/components/schemas/RateLimits'
        createdAt:
          type: integer

    CreateGroupRequest:
      type: object
      required: [name, siteIds]
      properties:
        name:
          type: string
        description:
          type: string
        siteIds:
          type: array
          items:
            type: string
          minItems: 1
        publicAccess:
          $ref: '#/components/schemas/PublicAccess'

    UpdateGroupRequest:
      type: object
      properties:
        name:
          type: string
        description:
          type: string
        siteIds:
          type: array
          items:
            type: string
        publicAccess:
          $ref: '#/components/schemas/PublicAccess'
        rateLimits:
          $ref: '#/components/schemas/RateLimits'

    IndexedPage:
      type: object
      properties:
        url:
          type: string
        title:
          type: string
        summary:
          type: string
        tags:
          type: array
          items:
            type: string
        publishedAt:
          type: string
          format: date
        crawledAt:
          type: string
          format: date

    PublicAccess:
      type: object
      properties:
        enabled:
          type: boolean
        searchEnabled:
          type: boolean
        chatEnabled:
          type: boolean
        allowedDomains:
          type: array
          items:
            type: string
          description: Domains allowed to embed (use * for all)
        anonymousSessions:
          type: boolean
        sessionTTLMinutes:
          type: integer

    RateLimits:
      type: object
      properties:
        searchPerMinute:
          type: integer
        chatPerMinute:
          type: integer
        perIpPerMinute:
          type: integer

    CrawlerConfig:
      type: object
      properties:
        maxPages:
          type: integer
        maxDepth:
          type: integer
        respectRobotsTxt:
          type: boolean
        excludePatterns:
          type: array
          items:
            type: string
        includePatterns:
          type: array
          items:
            type: string
        javascriptEnabled:
          type: boolean
        delayMs:
          type: integer
          description: Delay between requests in milliseconds
        singleUrls:
          type: array
          items:
            type: string
          description: One-shot list of exact URLs to crawl (no link-following)
        sitemaps:
          type: array
          description: >-
            Sitemap URLs to seed the crawl from. Relative paths resolve against
            the site base URL; sitemap index files are followed automatically.
          items:
            type: object
            properties:
              url:
                type: string
              recursive:
                type: boolean

    PageContent:
      type: object
      required: [url, content]
      properties:
        url:
          type: string
        title:
          type: string
        content:
          type: string
          description: HTML or plain text content
        contentType:
          type: string
          enum: [html, text]
          default: html
        summary:
          type: string
        tags:
          type: array
          items:
            type: string
        preProcessed:
          type: boolean
          default: false
          description: Skip AI processing if content is already cleaned

    CrawlJob:
      type: object
      properties:
        jobId:
          type: string
        siteId:
          type: string
        status:
          type: string
          enum: [pending, running, completed, failed, cancelled]
        pagesDiscovered:
          type: integer
        pagesCrawled:
          type: integer
        pagesProcessed:
          type: integer
        pagesErrored:
          type: integer
        startedAt:
          type: integer
        completedAt:
          type: integer
        taskId:
          type: string
          description: ECS task ID (for cancellation)

    Error:
      type: object
      properties:
        error:
          type: string
        code:
          type: string

    AnalyticsResponse:
      type: object
      required: [totalSearches, totalChats, totalPagesIndexed, topQueries, siteBreakdown, dailyUsage]
      properties:
        totalSearches:
          type: integer
          description: Total search queries in the range.
        totalChats:
          type: integer
          description: Total chat queries in the range.
        totalPagesIndexed:
          type: integer
          description: Sum of indexed pages across all sites returned in siteBreakdown.
        topQueries:
          type: array
          description: Top 10 search queries across the org (or single site if filtered), sorted by count desc.
          items:
            type: object
            required: [query, count]
            properties:
              query:
                type: string
              count:
                type: integer
        siteBreakdown:
          type: array
          description: Per-site row including pages indexed and search/chat counts.
          items:
            type: object
            required: [id, name, pagesIndexed, searches, chats]
            properties:
              id:
                type: string
                format: uuid
              name:
                type: string
              pagesIndexed:
                type: integer
              searches:
                type: integer
              chats:
                type: integer
        dailyUsage:
          type: array
          description: Daily time-series, oldest first.
          items:
            type: object
            required: [date, searches, chats]
            properties:
              date:
                type: string
                format: date
                example: "2026-05-08"
              searches:
                type: integer
              chats:
                type: integer

    UsageResponse:
      type: object
      required: [month, ai, vectorDb, searches]
      properties:
        month:
          type: string
          pattern: '^\d{4}-\d{2}$'
          example: "2026-05"
        ai:
          type: object
          required: [totalRequests, totalInputTokens, totalOutputTokens, totalTokens, estimatedCost, ingest, chat]
          description: LLM token-level usage from upstream Quant AI API.
          properties:
            totalRequests:
              type: integer
            totalInputTokens:
              type: integer
            totalOutputTokens:
              type: integer
            totalTokens:
              type: integer
            estimatedCost:
              type: number
              description: Estimated cost in USD for the AI calls in this month.
            ingest:
              type: object
              required: [requests, tokens]
              properties:
                requests:
                  type: integer
                tokens:
                  type: integer
            chat:
              type: object
              required: [requests, tokens]
              properties:
                requests:
                  type: integer
                tokens:
                  type: integer
        vectorDb:
          type: object
          required: [totalCollections, totalDocuments]
          properties:
            totalCollections:
              type: integer
            totalDocuments:
              type: integer
        searches:
          type: object
          required: [total, chat]
          description: Search/chat query counts from local tracking (usage_monthly).
          properties:
            total:
              type: integer
            chat:
              type: integer

    SiteUsageResponse:
      type: object
      required: [siteId, range, searches, chats, totalTokens, daily]
      properties:
        siteId:
          type: string
          format: uuid
        range:
          type: string
          example: "30d"
        searches:
          type: integer
        chats:
          type: integer
        totalTokens:
          type: integer
          description: Sum of AI tokens consumed by this site in the range. May be 0 if upstream telemetry is unavailable.
        daily:
          type: array
          description: Daily time-series for this site, oldest first.
          items:
            type: object
            required: [date, searches, chats]
            properties:
              date:
                type: string
                format: date
              searches:
                type: integer
              chats:
                type: integer

    SiteTopQueriesResponse:
      type: object
      required: [siteId, range, topQueries]
      properties:
        siteId:
          type: string
          format: uuid
        range:
          type: string
        topQueries:
          type: array
          items:
            type: object
            required: [query, count]
            properties:
              query:
                type: string
              count:
                type: integer

  responses:
    BadRequest:
      description: Bad request
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
    
    Unauthorized:
      description: Unauthorized - invalid or missing API key
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
    
    Forbidden:
      description: Forbidden - origin not allowed or feature disabled
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
    
    NotFound:
      description: Resource not found
      content:
        application/json:
          schema:
            $ref: '#/components/schemas/Error'
    
    RateLimited:
      description: Rate limit exceeded
      headers:
        X-RateLimit-Limit:
          schema:
            type: integer
        X-RateLimit-Remaining:
          schema:
            type: integer
        X-RateLimit-Reset:
          schema:
            type: integer
      content:
        application/json:
          schema:
            type: object
            properties:
              error:
                type: string
              retryAfter:
                type: integer
