diff --git a/AGENTS.md b/AGENTS.md index c57a3ea..495fe87 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -22,7 +22,7 @@ lighton/ workspace.py # Workspace, active-record, lives at root apikey.py # ApiKey / ApiKeyScope, active-record, lives at root tag.py # Tag, active-record (list/create/delete only; no single GET) - content_type.py # ContentType/Facet/Attribute/FacetAction/FacetResult, taxonomy + file facets + content_type.py # ContentType(Ref)/Facet/Attribute/FacetAction/FacetResult/FacetScope, taxonomy + file facets + scope file.py # File, active-record + wait_all(); upload = ingestion batch.py # ingest_many() batch upload behavior: BatchIngestJob (threads/poll) job.py # ParseJob/ExtractJob, client-bound async handles you poll() @@ -71,7 +71,7 @@ a field whose full domain is known, `workspace_type`/`document_upload_method` st - **Sync only.** `httpx.Client`. No async client until a real event-loop caller needs one, `_request` is the only logic to mirror. - **One `_request`** does auth header, error mapping (→ raises), and JSON parse. All calls route through it. A 2xx body that isn't JSON → `MalformedResponseError`, **unless `raw=True`**, which returns the 2xx body as `bytes` instead of parsing it. Two endpoints serve files rather than JSON (`GET /files//download` and `GET /files//thumbnail`) and this is the one decision on the list that got *edited* rather than extended. A flag on the existing chokepoint, not a `_download` sibling: auth, error mapping, the 429 cooldown and the rate gate are exactly what a sibling would have had to duplicate, and errors stay JSON even on a binary endpoint, so they must keep flowing through the same mapping (a test pins each: bytes back, a 404 still raising, a 429 still retrying). `raw` is keyword-only so it can't collide with an httpx kwarg, and an empty raw body is `b""`, not the `None` the JSON path returns. `_ActiveRecord._api` widened its `**kwargs` from `object` to `Any` to forward it. -- **Primary verbs** live one-per-file in `verbs/` as mixins (`AskMixin`/`SearchMixin`/`ParseMixin`/`ExtractMixin`) composed onto `LightOn`. Each references `self._request`; the stub on `_VerbClient` (their shared base) makes them type-check in isolation, and `LightOn._request` overrides it at runtime. Keeps `_client.py` to just the transport core. They take explicit typed params and return the generated response models via `model_validate`. `ask`/`search` take `workspaces`/`tags`/`files` (lists of `Workspace`/`Tag`/`File` objects or bare ids; `_ids()` in `utils.py` coerces via duck-typed `.id` → the API's `workspace_id`/`tag_id`/`file_id`; server-side, `file_id` can't combine with `workspace_id`/`tag_id`, and `tag_id` is OR-matched). `tags` additionally accepts **name strings**, resolved through `tag.resolve_ids` (same helper as `File.tag`), so the verb `cast`s `self` to `LightOn` (the mixin `self` is typed `_VerbClient`) to call `Tag.list`; resolution only lists when a name is present. `ask`/`search` also take the facet filters `content_type` (`ContentType` objects or path strings, coerced by `_paths()` in `utils.py`, the `.path` sibling of `_ids()`; OR-matched, exact-or-subtree, wildcards) and `attribute` (plain `list[str]` passthrough, entries ANDed, `|` ORs within an entry), both straight to the same-named API fields. `parse` takes keyword-only `path` XOR `url` (multipart vs JSON body; raises `ValueError` unless exactly one). `extract` takes keyword-only `path` XOR `url` XOR `file` (raises `ValueError` unless exactly one; `path` → multipart, the other two → JSON body) plus a `schema` that is **either a pydantic model class** or a **raw JSON-Schema dict**; returns `ExtractJobResponse`. `file` is an **already-ingested** file (`File` or bare id → the API's `file_id`, coerced by `_id()` in `utils.py`, the scalar sibling of `_ids()`), no re-upload; `File` is imported under `TYPE_CHECKING` only, so the annotation costs no import cycle. The multipart `file` **part** still isn't in the OpenAPI schema (`ExtractRequest` models `document`/`file_id`/`schema`/`options`) but the endpoint accepts it, verified by curl; on multipart, `schema`/`options` ride as JSON-encoded form fields alongside it. Note the name collision: the `file=` **param** means file_id, the multipart part is what `path=` sends. `ask` takes the same `schema` (pydantic class or dict) and sends it as the API's `response_format` for structured output; the answer then comes back as JSON **text** in `AskResponse.answer` (the response model is unchanged), and the SDK deliberately does **not** parse it back, callers do `Model.model_validate_json(resp.answer)`, since a model class is only one of the two accepted inputs and re-validating would make the return type depend on which one was passed. Schema handling (in `utils.py`, `as_json_schema()` is the shared entry point for both verbs): **both inputs end up normalized by `normalize_response_format_json`**: `$defs`/`$ref` inlined (`_inline_refs`), nullable `anyOf` collapsed to `type: [X, "null"]` (`_collapse_nullable`), draft-2020-12 `$schema` marker added (an existing one is kept). A pydantic model goes `model_json_schema()` → normalize (`convert_pydantic_to_response_format_json`); a dict is first validated against the draft-2020-12 meta-schema via `jsonschema` (`validate_response_format_json`, raises `SchemaError`) and then normalized too, **not** passed through as it used to be, because the endpoint 422s on `$ref` and a dict is usually just someone's own `model_json_schema()` call, which carries them (the original bug: nested models only worked via the model-class path). A `#/$defs/` ref with no target raises `SchemaError` rather than a bare `KeyError` from inside the recursion. `jsonschema` is a runtime dep (meta-schema validation is its job; hand-rolling would be flimsy). Ceiling: `_inline_refs` recurses through refs, so a self-referential model would overflow, fine, guided-gen grammars can't express unbounded recursion anyway. +- **Primary verbs** live one-per-file in `verbs/` as mixins (`AskMixin`/`SearchMixin`/`ParseMixin`/`ExtractMixin`) composed onto `LightOn`. Each references `self._request`; the stub on `_VerbClient` (their shared base) makes them type-check in isolation, and `LightOn._request` overrides it at runtime. Keeps `_client.py` to just the transport core. They take explicit typed params and return the generated response models via `model_validate`. `ask`/`search` take `workspaces`/`tags`/`files` (lists of `Workspace`/`Tag`/`File` objects or bare ids; `_ids()` in `utils.py` coerces via duck-typed `.id` → the API's `workspace_id`/`tag_id`/`file_id`; server-side, `file_id` can't combine with `workspace_id`/`tag_id`, and `tag_id` is OR-matched). `tags` additionally accepts **name strings**, resolved through `tag.resolve_ids` (same helper as `File.tag`), so the verb `cast`s `self` to `LightOn` (the mixin `self` is typed `_VerbClient`) to call `Tag.list`; resolution only lists when a name is present. `ask`/`search` also take the facet filters `content_type` (a `list[ContentTypeRef]`: nodes, scored `scope()` hits, or path strings, coerced by `_paths()` in `utils.py`, the `.path` sibling of `_ids()`; OR-matched, exact-or-subtree, wildcards) and `attribute` (plain `list[str]` passthrough, entries ANDed, `|` ORs within an entry), both straight to the same-named API fields, plus **`scope`** (`FacetScope | bool`), which derives that same pair from the query instead. Both verbs route the pair through `content_type.resolve_scope()` (the `resolve_ids` of scoping, and the only place that knows what `scope=` means, so the two can't drift), which also owns the `ValueError` refusing `scope=` alongside an explicit filter. It resolves eagerly, so on streaming `ask` the scope request goes out before the generator is returned, exactly like tag-name resolution. `parse` takes keyword-only `path` XOR `url` (multipart vs JSON body; raises `ValueError` unless exactly one). `extract` takes keyword-only `path` XOR `url` XOR `file` (raises `ValueError` unless exactly one; `path` → multipart, the other two → JSON body) plus a `schema` that is **either a pydantic model class** or a **raw JSON-Schema dict**; returns `ExtractJobResponse`. `file` is an **already-ingested** file (`File` or bare id → the API's `file_id`, coerced by `_id()` in `utils.py`, the scalar sibling of `_ids()`), no re-upload; `File` is imported under `TYPE_CHECKING` only, so the annotation costs no import cycle. The multipart `file` **part** still isn't in the OpenAPI schema (`ExtractRequest` models `document`/`file_id`/`schema`/`options`) but the endpoint accepts it, verified by curl; on multipart, `schema`/`options` ride as JSON-encoded form fields alongside it. Note the name collision: the `file=` **param** means file_id, the multipart part is what `path=` sends. `ask` takes the same `schema` (pydantic class or dict) and sends it as the API's `response_format` for structured output; the answer then comes back as JSON **text** in `AskResponse.answer` (the response model is unchanged), and the SDK deliberately does **not** parse it back, callers do `Model.model_validate_json(resp.answer)`, since a model class is only one of the two accepted inputs and re-validating would make the return type depend on which one was passed. Schema handling (in `utils.py`, `as_json_schema()` is the shared entry point for both verbs): **both inputs end up normalized by `normalize_response_format_json`**: `$defs`/`$ref` inlined (`_inline_refs`), nullable `anyOf` collapsed to `type: [X, "null"]` (`_collapse_nullable`), draft-2020-12 `$schema` marker added (an existing one is kept). A pydantic model goes `model_json_schema()` → normalize (`convert_pydantic_to_response_format_json`); a dict is first validated against the draft-2020-12 meta-schema via `jsonschema` (`validate_response_format_json`, raises `SchemaError`) and then normalized too, **not** passed through as it used to be, because the endpoint 422s on `$ref` and a dict is usually just someone's own `model_json_schema()` call, which carries them (the original bug: nested models only worked via the model-class path). A `#/$defs/` ref with no target raises `SchemaError` rather than a bare `KeyError` from inside the recursion. `jsonschema` is a runtime dep (meta-schema validation is its job; hand-rolling would be flimsy). Ceiling: `_inline_refs` recurses through refs, so a self-referential model would overflow, fine, guided-gen grammars can't express unbounded recursion anyway. - **Async jobs.** `parse`/`extract` take `mode: ExecMode` (default `ExecMode.SYNC`); `ExecMode.ASYNC` (uppercase members, value `"async"`, and lowercase `async` can't be a member name) sends `options={"async": true}`. `ExecMode` lives in `enums.py` (StrEnum, exported). Async returns a **pollable job handle** (`job.py`): `parse(mode=ASYNC)` → `ParseJob`, `extract(mode=ASYNC)` → `ExtractJob`; sync returns the full response model as before. Each verb has two `@overload`s keyed on `mode: Literal[ExecMode.SYNC|ASYNC]` so callers get the exact return type (`ParseResponse` vs `ParseJob`) instead of the union, the impl signature keeps the `ExecMode` default and the `... | ...Job` return. `Job.poll(page=None)` GETs `/`, absorbs the response onto itself in place (mirrors `_ActiveRecord._absorb`), returns self; `.done` (terminal, `completed_at` set) and `.succeeded` (`status == completed`) read state. `_Job` is a hand-written curated model (`extra="ignore"`) holding the shared plumbing + fields; `ParseJob`/`ExtractJob` subclass it ONLY because `result` differs (`ParseResult.pages` vs `ExtractResult.data`, whose optional fields make a union ambiguous), parse also has `error`. The job binds to the client via the `_VerbClient` transport surface (all it needs is `_request`), not a full `LightOn` (keeps the mixin's `self` assignable without a cast). `JobStatus` (enums.py) has only the documented `pending`/`completed`, the API doesn't publish the failure vocab, so it's for call-site comparison (StrEnum, unknown server values compare unequal, never validated onto the field), and the "poll until `.succeeded`, raise once `.done`" pattern keys off `completed_at`, not a failure string. `_Job.wait(timeout=300, poll=2)` is the auto-wait: a `File.wait`-style poll loop (no webhook exists) that returns self once terminal, raises `TimeoutError` past the deadline and `LightOnError` if `not .succeeded` (detail from `error` when the subclass has one, `getattr`, since only `ParseJob` does). The verbs expose it as `wait=False`/`timeout=300.0` (**same pair as `Workspace.ingest`**), declared **only on the ASYNC `@overload`** so `wait=True` without `mode=ASYNC` is a static error *and* a `ValueError` (sync already blocks); the two negative tests carry a `# ty: ignore[no-matching-overload]`. `wait=True` still returns the job (not the sync response model), so the return-type overloads stay two. No `poll` knob on the verbs, callers who need one use `job.wait(poll=...)`. - **Taxonomy writes** live on `ContentType` as **classmethods** (it isn't an `_ActiveRecord`: the endpoint returns a nested tree, not a paginated flat list, so @@ -114,7 +114,7 @@ a field whose full domain is known, `workspace_type`/`document_upload_method` st so a new server event can't break existing callers. Being a generator, the request goes out on first iteration, so errors surface there, which the docstring and README both state. -- Deferred: an async client, add it when a real event-loop caller needs one. **`POST /api/v3/preview`** (any supported document to PDF, sync, 20 MB cap) is deliberately **not** a verb: it's internal tooling, not SDK surface. `_request("POST", "/api/v3/preview", files=..., raw=True)` reaches it in one line if that ever changes, which is part of what the `raw` flag buys. Also unwrapped from the current OpenAPI schema: `POST /api/v3/content-types/scope` (`FacetScopeRequest`/`FacetScopeResponse`, LLM scope inference) and `scoped_api_keys` on the workspace responses. +- Deferred: an async client, add it when a real event-loop caller needs one. **`POST /api/v3/preview`** (any supported document to PDF, sync, 20 MB cap) is deliberately **not** a verb: it's internal tooling, not SDK surface. `_request("POST", "/api/v3/preview", files=..., raw=True)` reaches it in one line if that ever changes, which is part of what the `raw` flag buys. Still unwrapped from the current OpenAPI schema: `scoped_api_keys` on the workspace responses. - **Config object.** Non-essential knobs (`base_url`, `timeout`, `retries`, `transport`) live in `LightOnConfiguration` (pydantic, `arbitrary_types_allowed`). `api_key` stays a direct `LightOn()` arg; falls back to `LIGHTON_API_KEY` env. - **Retries / rate limiting.** Two layers: `httpx.HTTPTransport(retries=)` handles *connection* errors (exp. backoff); `_request` itself handles **HTTP 429**, retries up to @@ -433,6 +433,65 @@ models no facet fields locally. so the *position* is the payload, and it rides on the exception (`LightOnAPIError.index`). +- **Scope is a `ContentType` classmethod, not a verb.** `ContentType.scope(client, + query=, max_results=, threshold=, model=, relevance_scoring=)` POSTs + `/content-types/scope` and returns a `FacetScope`. It sits with `list`/`templates`/ + `batch` because the URL, the OpenAPI tag and the domain are all content types, and + because `verbs/` means the *primary* verbs (ask/search/parse/extract); scope serves + search rather than standing beside it. It can't reuse `_action()`, which injects an + `action` field and posts to `_BASE`. The endpoint is stateless: no ids, no CRUD, no + pagination, so there was never an active-record question. +- **Curated `FacetScope`/`ScopeGroup`/`ScopedContentType`/`ScopeCompletion`**, not the + generated `FacetScopeResponse`, for one concrete reason on top of the file's existing + preference: the schema types `ScoredContentType.attributes` as free-form `object[]` + (`additionalProperties: true`), so codegen yields `list[dict]` where the wire actually + sends exactly the existing **`Attribute`** shape. Curating reuses `Attribute` and gets + `.choices` typed. `ScopedContentType` carries `.path`, so `_path()`/`_paths()` accept + a scored hit wherever a `ContentType` goes. The curated `ScopeCompletion` shares its + name with the generated one; they never meet, since `types/api` isn't re-exported from + `lighton`. `FacetScope.completion` is the **one aliased field** in the SDK + (`alias="scope_completion"`, with `populate_by_name=True`): the wire name stutters on + a `FacetScope`. Don't spread the pattern without a reason that good. +- **`filters()` precedence**: a `completion` that inferred anything wins outright, then + `has_signal` false means `{}`, then the scored paths. A completion outranks + `has_signal` deliberately: `has_signal` gates *retrieval* scores, and vetoing an + explicit answer from the model with it would throw away the LLM call the caller paid + for. Keys are absent rather than empty, so "no filter" and "empty filter" can't be + confused. It returns a plain dict and the docs pass it **by keyword, not splatted**: + ty checks `**dict[str, list[str]]` against every parameter of `search`, so the splat + fails on `workspaces`. Not worth widening a signature or a `# ty: ignore`, and + `scope=` is the ergonomic path anyway. +- **`scope=True` is retrieval-only narrowing**, resolving with no `model` and filtering + by the content types it scores. It is the same code path as a passed-in `FacetScope`, + just resolved for you, because `filters()` already falls back to the scored paths when + there is no completion. No string form (`scope="some-model"`) on purpose: `ask` + already has a `model=` for answer generation, and two differently-meaning model names + in one call is a trap. Paying for scope inference stays an explicit + `ContentType.scope(..., model=...)`. Known ceiling, documented rather than fixed: + `True` resolves at the endpoint's defaults, so it keeps **every** scored type, up to + 20 — a loose net that on a small taxonomy narrows almost nothing. It can't be + tightened client-side (the response echoes no per-hit threshold, and inventing a cut + would be the SDK making retrieval policy), so `max_results=` at resolve time is the + lever, and the README/docstrings say so instead of implying `True` is precise. Give + it a `scope_max_results`-style knob only if callers actually ask: that is a second + way to spell `ContentType.scope()`. +- **`ContentTypeRef = ContentType | ScopedContentType | str`** (content_type.py, + exported) is the type of every parameter that names a content type: `ask`/`search`'s + `content_type`, `File.classify`/`unclassify`/`set_attribute`/`clear_attribute`, the + `FacetAction` constructors, and `ContentType`'s own `parent`/`content_type` args. One + alias, not a union spelled out fifteen times, because it is one rule — *we accept + anything with a `.path`*, the annotation counterpart of `_path()`/`_paths()`. It + earned itself when `ScopedContentType` arrived: duck-typing made a scored hit work at + runtime while every signature still said `ContentType | str`, so `ty` rejected the + call the README was recommending. Widening one alias fixed all fifteen. The SDK's + other coercions (`File | int`, `Tag | int | str`) stay spelled out, they have two + members and no third is coming; alias a union once a *third* accepted type shows up, + not before. +- **`relevance_scoring` on `scope()` is `Literal[RelevanceScoring.none] | None`**, since + the endpoint accepts only `"none"` or nothing while the shared enum has three members. + Static typing catches it, and a `ValueError` catches it at runtime too, the same trade + as `define_attribute`'s missing-`choices` check. + If adding new resources, subclass `_ActiveRecord`: set `_base`/`_resource`, declare the field schema (narrow `id`), and add `create()`/`save()`. Everything else is inherited. Only push behavior down into the base when a new resource actually shares it, don't diff --git a/README.md b/README.md index 9e067d1..759db0d 100644 --- a/README.md +++ b/README.md @@ -253,6 +253,14 @@ resp = client.search( are ANDed, `|` ORs within one entry; also `name` (has any value), `name:>value`, `name:prefix*`, `name:*text*`. +Don't know the paths, or the query came from a user? `scope=` derives both filters +from the question itself (also on `ask`), see [resolving the scope from a +question](#resolving-the-scope-from-a-question): + +```python +resp = client.search("termination clause", scope=True) +``` + `relevance_scoring` tunes the scoring step (applies to `ask` too): - `.scoring_and_filtering` (default): score, drop chunks below the quality threshold @@ -812,6 +820,10 @@ questions: a document filed under `legal:contract:nda` also comes back for labels with no values. Reach for tags for cross-cutting marks like `confidential`, and content types for what a document *is*. +Step 3 assumes you know your taxonomy by heart. When you don't, or when the query +comes from a user rather than from you, [let the API pick the +filters](#resolving-the-scope-from-a-question). + ### Browsing the taxonomy `ContentType.list()` returns the tree (each node has `children`, and `attributes` @@ -826,6 +838,85 @@ for ct in ContentType.list(client, include_attributes=True): print(" ", attr.name, attr.type, attr.choices) ``` +### Resolving the scope from a question + +`ContentType.scope()` reads a question and ranks the taxonomy against it, so you get +`content_type=`/`attribute=` filters without hard-coding paths. Pass it a `model` and +the API runs that LLM for you, returning the filters it inferred: + +```python +from lighton import ContentType, LightOn + +with LightOn() as client: + scope = ContentType.scope( + client, + "rejected electronics patents filed in Q1 2023", + model="mistral-large-latest", + ) + + print(scope.filters()) + # {'content_type': ['patent:electricity'], + # 'attribute': ['decision:Rejected', 'filing_date:>=2023-01-01', + # 'filing_date:<=2023-03-31']} + + answer = client.ask("why were they rejected", scope=scope) +``` + +`scope=` is accepted by both [`ask`](#ask-single-turn-rag) and +[`search`](#search-retrieval-only-no-generation), and applies the filters for you. It +also takes `True`, which resolves a scope from that same query with no LLM at all: +cheaper (one extra request, no generation), and by content type only, since +inferring attribute filters is what the model is for. + +```python +# no model, no attribute filters, one extra round trip +client.search("termination clause", scope=True) +``` + +`True` is a loose net, not a precise one: with no `model` to pick a winner, it keeps +*every* content type the endpoint scored — up to its default of 20, weak matches +included — so on a small taxonomy it may not narrow much at all. Resolve the scope +yourself when you want a tight filter, and pass that: + +```python +top = ContentType.scope(client, "termination clause", max_results=3) +client.search("termination clause", scope=top) +``` + +Nothing is stored: each call is one stateless inference. What comes back: + +| | | +|---|---| +| `has_signal` | Whether anything scored above `threshold`. False means narrow nothing. | +| `content_types` | Every scored content type, best first, with typed `attributes`. | +| `groups` | The same hits kept in the server's per-tree grouping. | +| `completion` | The model's inferred scope. `None` unless you passed `model`. | +| `prompt_context` | An LLM-ready description of the matched schema (see below). | +| `filters()` | The filters this scope narrows to, `{}` when it narrows nothing. | + +Without a `model` you get `prompt_context` instead of a `completion`: a ready-made +prompt describing the matched content types and their attributes, for running against +whatever LLM you already have. Omit the query too and you get the whole taxonomy, +which is what you want baked into a system prompt: + +```python +catalog = ContentType.scope(client) # no query: the full schema +system_prompt = f"Available document types:\n{catalog.prompt_context}" +``` + +A scored hit carries `.path`, so it goes anywhere a `ContentType` does — every +parameter that names a content type takes the `ContentTypeRef` union (node, scored +hit, or path string), so this type-checks as well as it runs: + +```python +best = ContentType.scope(client, "mutual NDA expiring 2025").content_types[0] +print(best.path, best.score, best.doc_count) +client.search("termination clause", content_type=[best]) +``` + +`scope=` refuses to combine with an explicit `content_type=`/`attribute=`: pass the +scope or the filters, not both, so it is always clear which narrowed the search. + ### Classifying a file Assign a content type and record its attribute values. `classify`, `unclassify`, diff --git a/lighton/__init__.py b/lighton/__init__.py index 80dd402..4d2af63 100644 --- a/lighton/__init__.py +++ b/lighton/__init__.py @@ -9,10 +9,15 @@ Attribute, ContentType, ContentTypeAction, + ContentTypeRef, ContentTypeResult, Facet, FacetAction, FacetResult, + FacetScope, + ScopeCompletion, + ScopedContentType, + ScopeGroup, Template, ) from lighton.enums import ( @@ -63,6 +68,7 @@ "ContentType", "ContentTypeAction", "ContentTypeActionType", + "ContentTypeRef", "ContentTypeResult", "DoneEvent", "DownloadPurpose", @@ -73,6 +79,7 @@ "FacetAction", "FacetActionType", "FacetResult", + "FacetScope", "FailedIngest", "File", "FileStatus", @@ -86,6 +93,9 @@ "ReprocessLevel", "Role", "RootContentType", + "ScopeCompletion", + "ScopeGroup", + "ScopedContentType", "SearchMode", "SourcesEvent", "Tag", diff --git a/lighton/content_type.py b/lighton/content_type.py index 13738eb..78bc73d 100644 --- a/lighton/content_type.py +++ b/lighton/content_type.py @@ -10,6 +10,14 @@ name/type/value shape used by both. `ContentTypeAction`/`ContentTypeResult` and `FacetAction`/`FacetResult` are the write and result shapes of the two batch endpoints, `ContentType.batch()` and `File.batch_facets()`. + +`FacetScope` is what `ContentType.scope()` resolves: the taxonomy ranked against a +natural-language query, plus the `content_type`/`attribute` filters to narrow a +search with. `ScopeGroup`/`ScopedContentType`/`ScopeCompletion` are its parts, and +`resolve_scope()` is the coercion helper `ask`/`search` call for their `scope=`. + +`ContentTypeRef` is what every "name a content type" parameter takes across the +SDK, here and on `File`/`ask`/`search`. """ from __future__ import annotations @@ -17,12 +25,17 @@ # The list() classmethod shadows builtin list in annotations (class scope). from builtins import list as _list from collections.abc import Sequence -from typing import TYPE_CHECKING, Any +from typing import TYPE_CHECKING, Any, Literal from pydantic import BaseModel, ConfigDict, Field, model_validator -from lighton.enums import AttributeType, ContentTypeActionType, FacetActionType -from lighton.utils import _compact, _path +from lighton.enums import ( + AttributeType, + ContentTypeActionType, + FacetActionType, + RelevanceScoring, +) +from lighton.utils import _compact, _path, _paths if TYPE_CHECKING: from lighton._client import LightOn @@ -108,6 +121,71 @@ def list( data = client._request("GET", _BASE, params=params) return [cls.model_validate(n) for n in data["content_types"]] + @classmethod + def scope( + cls, + client: LightOn, + query: str | None = None, + *, + max_results: int | None = None, + threshold: float | None = None, + model: str | None = None, + relevance_scoring: Literal[RelevanceScoring.none] | None = None, + ) -> FacetScope: + """Rank the taxonomy against a question (POST /content-types/scope). + + Turns free text into the `content_type`/`attribute` filters `ask` and + `search` take, so a caller doesn't have to know the taxonomy by heart. + Three modes, by argument: + + - **prompt** (no `model`): scored content types plus `prompt_context`, a + ready-made prompt for your own LLM. No `completion`. + - **completion** (`model=`): the API runs that LLM itself and returns the + inferred scope in `completion`, attribute filters included. + - **catalog** (`relevance_scoring=RelevanceScoring.none`): every content + type at score 0, `max_results`/`threshold` ignored. Pair it with `model` + to infer over the whole taxonomy rather than the query-relevant slice. + + Nothing is stored: this is one stateless inference call. + + Args: + client: The client to query with. + query: The natural-language question (max 2000 chars). Omit it for the + full schema catalog, which is what you want for a system prompt. + max_results: Content types to return (1-100; server default 20). + threshold: Score above which `has_signal` goes true (server default + 1.8). Pass 0 to disable the gate. + model: Technical name of the LLM that infers the scope. Omit to get + `prompt_context` and run your own. On `ask`, note this is a + different model from its answer-generation `model`. + relevance_scoring: Only `RelevanceScoring.none` is accepted here (skip + scoring, return the whole catalog); omit for the default scoring. + + Returns: + The resolved scope. Feed it to `ask`/`search` as `scope=`, or read + `filters()` yourself. + + Raises: + ValueError: If `relevance_scoring` is anything but + `RelevanceScoring.none`, which the API rejects anyway, caught here + to save the round trip. + """ + if relevance_scoring is not None and relevance_scoring != RelevanceScoring.none: + raise ValueError( + f"scope() only accepts relevance_scoring=RelevanceScoring.none, " + f"got {relevance_scoring}" + ) + body = _compact( + query=query, + max_results=max_results, + threshold=threshold, + model=model, + relevance_scoring=relevance_scoring, + ) + return FacetScope.model_validate( + client._request("POST", f"{_BASE}/scope", json=body) + ) + # --- taxonomy writes --------------------------------------------------- # Mirrors File._facet: one helper posts the action, the named methods just # build it. Every action is idempotent server-side. @@ -153,7 +231,7 @@ def define( code: str, label: str, *, - parent: ContentType | str | None = None, + parent: ContentTypeRef | None = None, description: str | None = None, inherit_attributes: bool | None = None, ) -> ContentType: @@ -189,7 +267,7 @@ def define( ) @classmethod - def undefine(cls, client: LightOn, content_type: ContentType | str) -> None: + def undefine(cls, client: LightOn, content_type: ContentTypeRef) -> None: """Delete a node **and cascade its whole subtree**. Args: @@ -205,7 +283,7 @@ def undefine(cls, client: LightOn, content_type: ContentType | str) -> None: def define_attribute( cls, client: LightOn, - content_type: ContentType | str, + content_type: ContentTypeRef, name: str, attribute_type: AttributeType | str, *, @@ -253,7 +331,7 @@ def define_attribute( @classmethod def undefine_attribute( - cls, client: LightOn, content_type: ContentType | str, name: str + cls, client: LightOn, content_type: ContentTypeRef, name: str ) -> None: """Remove an attribute column from a node. @@ -481,7 +559,7 @@ def define( code: str, label: str, *, - parent: ContentType | str | None = None, + parent: ContentTypeRef | None = None, description: str | None = None, inherit_attributes: bool | None = None, ) -> ContentTypeAction: @@ -508,7 +586,7 @@ def define( ) @classmethod - def undefine(cls, content_type: ContentType | str) -> ContentTypeAction: + def undefine(cls, content_type: ContentTypeRef) -> ContentTypeAction: """Delete a node and its subtree, the batch form of `ContentType.undefine()`. Args: @@ -525,7 +603,7 @@ def undefine(cls, content_type: ContentType | str) -> ContentTypeAction: @classmethod def define_attribute( cls, - content_type: ContentType | str, + content_type: ContentTypeRef, name: str, attribute_type: AttributeType | str, *, @@ -566,7 +644,7 @@ def define_attribute( @classmethod def undefine_attribute( - cls, content_type: ContentType | str, name: str + cls, content_type: ContentTypeRef, name: str ) -> ContentTypeAction: """Remove an attribute, the batch form of `ContentType.undefine_attribute()`. @@ -620,6 +698,165 @@ class ContentTypeResult(BaseModel): ) +class ScopedContentType(BaseModel): + """One content type scored against a query by `ContentType.scope()`. + + It carries `.path`, so it is a `ContentTypeRef`: `search(content_type=[hit])` + and `doc.classify(hit)` both work, and type-check. + """ + + model_config = ConfigDict(extra="ignore") + + path: str = Field(description="Full taxonomy path, e.g. legal:contract:nda.") + label: str = Field(description="Human-readable label.") + root: str = Field(description="Path of the tree this node belongs to.") + score: float = Field( + description="Relevance to the query; 0 when scoring was skipped." + ) + chunk_count: int = Field(description="Chunks matching the query under this type.") + doc_count: int = Field(description="Documents classified under this type.") + attributes: _list[Attribute] = Field( + default_factory=list, + description="Attribute definitions on this type; `value` is always None here.", + ) + + +ContentTypeRef = ContentType | ScopedContentType | str +"""Anything that names a content type: a taxonomy node, a scored hit, or a path. + +Exactly what `_path()`/`_paths()` coerce, so every parameter that takes a content +type takes all three. Declared once here rather than spelled out per signature: +it is one rule (*we accept a `.path`*), and `ScopedContentType` was added to it +without touching the fifteen signatures that state it. +""" + + +class ScopeGroup(BaseModel): + """The scored content types of one taxonomy tree, as `scope()` groups them.""" + + model_config = ConfigDict(extra="ignore") + + root: str = Field(description="Path of the tree's root node.") + root_label: str = Field(description="Human-readable label of the root.") + max_score: float = Field(description="Best score among this group's types.") + content_types: _list[ScopedContentType] = Field( + default_factory=list, description="The tree's scored content types." + ) + + +class ScopeCompletion(BaseModel): + """The LLM's inferred scope, present only when `scope()` was given a `model`. + + Its `content_type`/`attribute` are already normalized to the paths and filter + syntax `ask`/`search` expect, which is what `FacetScope.filters()` hands back. + """ + + model_config = ConfigDict(extra="ignore") + + content_type: str | None = Field( + None, description="Inferred content-type path, None when it inferred none." + ) + attribute: _list[str] = Field( + default_factory=list, + description='Inferred attribute filters, e.g. ["filing_date:>=2023-01-01"].', + ) + raw_output: str = Field("", description="The model's unparsed output.") + normalized: bool = Field( + False, description="Whether a label had to be mapped onto its attribute name." + ) + warnings: _list[str] = Field( + default_factory=list, + description=( + "Why inference degraded, empty in the happy path. A failed model call " + "lands here rather than failing the request." + ), + ) + + +class FacetScope(BaseModel): + """A search scope resolved from a question, what `ContentType.scope()` returns.""" + + # populate_by_name so `completion` works as a field name too, not just the + # `scope_completion` the wire sends. + model_config = ConfigDict(extra="ignore", populate_by_name=True) + + has_signal: bool = Field( + description="Whether any content type scored above `threshold`." + ) + groups: _list[ScopeGroup] = Field( + default_factory=list, description="Scored content types, grouped by tree." + ) + prompt_context: str | None = Field( + None, + description=( + "LLM-ready description of the matched schema. Feed it to your own model " + "when you resolve the scope yourself (no `model=`)." + ), + ) + prompt_version: str | None = Field( + None, + description=( + "Hash of the prompt template and the data behind it, for reproducing a " + "run. Same value means the same prompt was built." + ), + ) + completion: ScopeCompletion | None = Field( + None, + alias="scope_completion", + description="The LLM's inferred scope; None unless `scope()` got a `model`.", + ) + + @property + def content_types(self) -> _list[ScopedContentType]: + """Every scored content type across the groups, best score first. + + Returns: + The flattened, score-ordered hits. `groups` keeps the server's grouping + if you want it by tree. + """ + hits = [ct for group in self.groups for ct in group.content_types] + return sorted(hits, key=lambda ct: ct.score, reverse=True) + + def filters(self) -> dict[str, _list[str]]: + """The `content_type`/`attribute` filters this scope narrows to. + + Passing the scope itself, `client.search(query, scope=scope)`, applies + these for you; read them here to see (or log) what a scope would narrow + to before spending the search on it. Pass them by keyword rather than + splatting the dict, which a type checker can't match to the signature: + + client.search(query, content_type=scope.filters().get("content_type")) + + Precedence: + + 1. A `completion` that inferred anything wins. That is an explicit answer + from the model, and `has_signal` (a retrieval-score gate) must not veto it. + 2. Otherwise, no signal means no narrowing, so `{}`. + 3. Otherwise the scored paths, which is retrieval-only narrowing. **Every** + scored path, not just the best ones: `threshold` gates `has_signal` for + the whole scope and isn't echoed per hit, so there is nothing to cut on + here. Narrowing to the top few is `scope(..., max_results=3)` at resolve + time. No attribute filters either: inferring those needs a `model`. + + Returns: + A dict holding `content_type` and/or `attribute`, empty when the scope + says to narrow nothing. A key is absent rather than empty, so a + missing filter reads the same as one never asked for. + """ + done = self.completion + if done is not None and (done.content_type or done.attribute): + narrowed: dict[str, _list[str]] = {} + if done.content_type: + narrowed["content_type"] = [done.content_type] + if done.attribute: + narrowed["attribute"] = _list(done.attribute) + return narrowed + if not self.has_signal: + return {} + paths = [ct.path for ct in self.content_types] + return {"content_type": paths} if paths else {} + + class Facet(BaseModel): """A content type assigned to a file, with the file's attribute values on it.""" @@ -678,7 +915,7 @@ def _value_actions_need_an_attribute(self) -> FacetAction: return self @classmethod - def classify(cls, content_type: ContentType | str) -> FacetAction: + def classify(cls, content_type: ContentTypeRef) -> FacetAction: """Assign a content type, the batch form of `File.classify()`. Args: @@ -692,7 +929,7 @@ def classify(cls, content_type: ContentType | str) -> FacetAction: ) @classmethod - def unclassify(cls, content_type: ContentType | str) -> FacetAction: + def unclassify(cls, content_type: ContentTypeRef) -> FacetAction: """Remove a content-type assignment, the batch form of `File.unclassify()`. Args: @@ -707,7 +944,7 @@ def unclassify(cls, content_type: ContentType | str) -> FacetAction: @classmethod def set_attribute( - cls, content_type: ContentType | str, name: str, value: Any + cls, content_type: ContentTypeRef, name: str, value: Any ) -> FacetAction: """Set an attribute value, the batch form of `File.set_attribute()`. @@ -728,7 +965,7 @@ def set_attribute( ) @classmethod - def clear_attribute(cls, content_type: ContentType | str, name: str) -> FacetAction: + def clear_attribute(cls, content_type: ContentTypeRef, name: str) -> FacetAction: """Clear an attribute value, the batch form of `File.clear_attribute()`. Args: @@ -786,5 +1023,52 @@ class FacetResult(BaseModel): ) +def resolve_scope( + client: LightOn, + query: str, + scope: FacetScope | bool, + content_type: _list[ContentTypeRef] | None, + attribute: _list[str] | None, +) -> tuple[_list[str] | None, _list[str] | None]: + """The `content_type`/`attribute` a verb should send, applying `scope=` if asked. + + The one place that knows what `scope=` means, so `ask` and `search` can't + drift (same reasoning as `FacetAction._body()`). `True` resolves the scope from + the verb's own query with no `model`, which is retrieval-only narrowing and + costs one extra request but no LLM call. It takes the endpoint's defaults, so + it narrows to as many as 20 content types, weak matches included: a loose net, + not a precise one. Tighter narrowing is `ContentType.scope(client, query, + max_results=3)` passed in, and a `FacetScope` you resolved yourself is applied + as-is, attribute filters included. + + Args: + client: The client to resolve with, used only when `scope` is True. + query: The verb's query, what an unresolved scope is inferred from. + scope: False for no scoping, True to resolve one, or an already-resolved + `FacetScope`. + content_type: The verb's explicit `content_type` argument. + attribute: The verb's explicit `attribute` argument. + + Returns: + The `(content_type_paths, attribute)` pair to put in the request body. + + Raises: + ValueError: If `scope` is combined with an explicit `content_type` or + `attribute`. Merging them silently would hide which filters ran. + """ + if not scope: # False, and a FacetScope is always truthy + return _paths(content_type), attribute + if content_type is not None or attribute is not None: + raise ValueError( + "scope= replaces content_type=/attribute=; pass the scope or the " + "explicit filters, not both" + ) + resolved = ( + scope if isinstance(scope, FacetScope) else ContentType.scope(client, query) + ) + narrowed = resolved.filters() + return narrowed.get("content_type"), narrowed.get("attribute") + + ContentType.model_rebuild() # resolve the self-referential `children` forward ref Template.model_rebuild() diff --git a/lighton/file.py b/lighton/file.py index 59eb26c..ac3aa62 100644 --- a/lighton/file.py +++ b/lighton/file.py @@ -26,6 +26,7 @@ from lighton._active_record import _ActiveRecord from lighton.content_type import ( MAX_FACET_ACTIONS, + ContentTypeRef, Facet, FacetAction, FacetResult, @@ -39,7 +40,6 @@ if TYPE_CHECKING: from lighton._client import LightOn - from lighton.content_type import ContentType from lighton.tag import Tag from lighton.workspace import Workspace @@ -463,8 +463,8 @@ def _facet(self, action: FacetAction): # so the single-action and batch bodies cannot drift apart. return self._api("POST", f"{_BASE}/{self.id}/facets", json=action._body()) - def classify(self, content_type: ContentType | str) -> File: - """Assign a content type to this file (ContentType object or path string). + def classify(self, content_type: ContentTypeRef) -> File: + """Assign a content type to this file (node, scored hit, or path string). Args: content_type: The content type to assign, e.g. "legal:contract:nda". @@ -478,7 +478,7 @@ def classify(self, content_type: ContentType | str) -> File: self._facet(FacetAction.classify(content_type)) return self - def unclassify(self, content_type: ContentType | str) -> File: + def unclassify(self, content_type: ContentTypeRef) -> File: """Remove a content-type assignment from this file. Args: @@ -491,7 +491,7 @@ def unclassify(self, content_type: ContentType | str) -> File: return self def set_attribute( - self, content_type: ContentType | str, name: str, value: object + self, content_type: ContentTypeRef, name: str, value: object ) -> File: """Set an attribute value under an assigned content type. @@ -507,7 +507,7 @@ def set_attribute( self._facet(FacetAction.set_attribute(content_type, name, value)) return self - def clear_attribute(self, content_type: ContentType | str, name: str) -> File: + def clear_attribute(self, content_type: ContentTypeRef, name: str) -> File: """Clear an attribute value under an assigned content type. Args: diff --git a/lighton/verbs/ask.py b/lighton/verbs/ask.py index 7b3edb1..4a57134 100644 --- a/lighton/verbs/ask.py +++ b/lighton/verbs/ask.py @@ -8,17 +8,17 @@ from pydantic import BaseModel +from lighton.content_type import ContentTypeRef, FacetScope, resolve_scope from lighton.enums import RelevanceScoring from lighton.exceptions import StreamError from lighton.tag import resolve_ids from lighton.types.api import AskResponse from lighton.types.events import AskEvent, DoneEvent, SourcesEvent, TokenEvent -from lighton.utils import _compact, _ids, _paths, as_json_schema +from lighton.utils import _compact, _ids, as_json_schema from lighton.verbs._base import _VerbClient if TYPE_CHECKING: from lighton._client import LightOn - from lighton.content_type import ContentType from lighton.file import File from lighton.tag import Tag from lighton.workspace import Workspace @@ -73,8 +73,9 @@ def ask( workspaces: list[Workspace | int] | None = ..., tags: list[Tag | int | str] | None = ..., files: list[File | int] | None = ..., - content_type: list[ContentType | str] | None = ..., + content_type: list[ContentTypeRef] | None = ..., attribute: list[str] | None = ..., + scope: FacetScope | bool = ..., max_results: int | None = ..., relevance_scoring: RelevanceScoring | None = ..., model: str | None = ..., @@ -89,8 +90,9 @@ def ask( workspaces: list[Workspace | int] | None = ..., tags: list[Tag | int | str] | None = ..., files: list[File | int] | None = ..., - content_type: list[ContentType | str] | None = ..., + content_type: list[ContentTypeRef] | None = ..., attribute: list[str] | None = ..., + scope: FacetScope | bool = ..., max_results: int | None = ..., relevance_scoring: RelevanceScoring | None = ..., model: str | None = ..., @@ -104,8 +106,9 @@ def ask( workspaces: list[Workspace | int] | None = None, tags: list[Tag | int | str] | None = None, files: list[File | int] | None = None, - content_type: list[ContentType | str] | None = None, + content_type: list[ContentTypeRef] | None = None, attribute: list[str] | None = None, + scope: FacetScope | bool = False, max_results: int | None = None, relevance_scoring: RelevanceScoring | None = None, model: str | None = None, @@ -123,13 +126,23 @@ def ask( must exist. Excludes files. files: Restrict to these files (File objects or ids). Excludes workspaces and tags. - content_type: Restrict to these content-type paths, ContentType objects - or path strings (OR-matched, exact-or-subtree, e.g. "legal" also - matches "legal:contract"; wildcards `legal:contract*`, `*nda*`). + content_type: Restrict to these content types (nodes, scored hits + from `scope()`, or path strings; OR-matched, exact-or-subtree, + e.g. "legal" also matches "legal:contract"; wildcards + `legal:contract*`, `*nda*`). attribute: Restrict by attribute value, e.g. `["fiscal_year:2024|2025", "status:active"]`. Entries are ANDed, `|` ORs within one entry. Also `name` (has any value), `name:>value`, `name:prefix*`, `name:*text*`. + scope: Derive `content_type`/`attribute` from the query instead of + naming them. True resolves a scope from this query (one extra + request, no LLM) and narrows by every content type it scores, up + to the endpoint's default 20, so it is a loose net; pass a + `ContentType.scope(..., max_results=3)` to narrow harder. A + `FacetScope` from `ContentType.scope(..., model=...)` is applied + as-is, attribute filters included. That `model` infers the filters + and is not this call's `model`, which generates the answer. + Refuses to combine with an explicit `content_type`/`attribute`. max_results: Chunks to retrieve for context (1–50; server default 10). relevance_scoring: RelevanceScoring, .scoring_and_filtering (default), .scoring_only, or .none. @@ -150,18 +163,24 @@ class or a JSON-Schema dict (same inputs as `extract`, sent as and `DoneEvent` (last). Being a generator, nothing is sent until you start iterating, so request errors surface on the first step, not here. Iterate it fully or close it, so the connection is released. + `tags` and `scope` are the exception: they resolve eagerly, here, since + the body can't be built without them. Raises: StreamError: If the server reports a failure mid-stream; the answer is incomplete at that point. + ValueError: If `scope` is combined with `content_type`/`attribute`. """ tag_ids = resolve_ids(cast("LightOn", self), tags) if tags else None + content_type_paths, attribute = resolve_scope( + cast("LightOn", self), query, scope, content_type, attribute + ) body = _compact( query=query, workspace_id=_ids(workspaces), tag_id=tag_ids, file_id=_ids(files), - content_type=_paths(content_type), + content_type=content_type_paths, attribute=attribute, max_results=max_results, relevance_scoring=relevance_scoring, diff --git a/lighton/verbs/search.py b/lighton/verbs/search.py index c70112e..867a4d5 100644 --- a/lighton/verbs/search.py +++ b/lighton/verbs/search.py @@ -4,15 +4,15 @@ from typing import TYPE_CHECKING, cast +from lighton.content_type import ContentTypeRef, FacetScope, resolve_scope from lighton.enums import RelevanceScoring, SearchMode from lighton.tag import resolve_ids from lighton.types.api import SearchResponse -from lighton.utils import _compact, _ids, _paths +from lighton.utils import _compact, _ids from lighton.verbs._base import _VerbClient if TYPE_CHECKING: from lighton._client import LightOn - from lighton.content_type import ContentType from lighton.file import File from lighton.tag import Tag from lighton.workspace import Workspace @@ -26,8 +26,9 @@ def search( workspaces: list[Workspace | int] | None = None, tags: list[Tag | int | str] | None = None, files: list[File | int] | None = None, - content_type: list[ContentType | str] | None = None, + content_type: list[ContentTypeRef] | None = None, attribute: list[str] | None = None, + scope: FacetScope | bool = False, max_results: int | None = None, mode: SearchMode | None = None, relevance_scoring: RelevanceScoring | None = None, @@ -45,13 +46,22 @@ def search( must exist. Excludes files. files: Restrict to these files (File objects or ids). Excludes workspaces and tags. - content_type: Restrict to these content-type paths, ContentType objects - or path strings (OR-matched, exact-or-subtree, e.g. "legal" also - matches "legal:contract"; wildcards `legal:contract*`, `*nda*`). + content_type: Restrict to these content types (nodes, scored hits + from `scope()`, or path strings; OR-matched, exact-or-subtree, + e.g. "legal" also matches "legal:contract"; wildcards + `legal:contract*`, `*nda*`). attribute: Restrict by attribute value, e.g. `["fiscal_year:2024|2025", "status:active"]`. Entries are ANDed, `|` ORs within one entry. Also `name` (has any value), `name:>value`, `name:prefix*`, `name:*text*`. + scope: Derive `content_type`/`attribute` from the query instead of + naming them. True resolves a scope from this query (one extra + request, no LLM) and narrows by every content type it scores, up + to the endpoint's default 20, so it is a loose net; pass a + `ContentType.scope(..., max_results=3)` to narrow harder. A + `FacetScope` from `ContentType.scope(..., model=...)` is applied + as-is, attribute filters included. Refuses to combine with an + explicit `content_type`/`attribute`. max_results: Chunks to return after reranking (1–100; server default 10). mode: SearchMode.text (hybrid keyword+vector) or .vision (page-image). relevance_scoring: RelevanceScoring, .scoring_and_filtering (default), @@ -61,14 +71,20 @@ def search( Returns: The ranked search results. + + Raises: + ValueError: If `scope` is combined with `content_type`/`attribute`. """ tag_ids = resolve_ids(cast("LightOn", self), tags) if tags else None + content_type_paths, attribute = resolve_scope( + cast("LightOn", self), query, scope, content_type, attribute + ) body = _compact( query=query, workspace_id=_ids(workspaces), tag_id=tag_ids, file_id=_ids(files), - content_type=_paths(content_type), + content_type=content_type_paths, attribute=attribute, max_results=max_results, mode=mode, diff --git a/tests/e2e/cli.py b/tests/e2e/cli.py index ca57115..a19e071 100644 --- a/tests/e2e/cli.py +++ b/tests/e2e/cli.py @@ -42,6 +42,7 @@ ExecMode, ExternalMetadata, FacetAction, + FacetScope, File, LightOn, MAX_CONTENT_TYPE_ACTIONS, @@ -104,12 +105,15 @@ class Ctx: stamp: str ask_query: str | None search_query: str | None + scope_model: str | None = None # names the LLM the `scope` step infers with ws: Workspace | None = None file: File | None = None tag: Tag | None = None content_type: ContentType | None = None # the file stays classified as this other_content_type: ContentType | None = None # a leaf it is NOT classified as attribute_filter: str | None = None # an `attribute=` entry that should match + seeded_roots: list[str] = field(default_factory=list) # roots _seed_taxonomy made + resolved_scope: FacetScope | None = None # what the `scope` step resolved topic: str | None = None # derived once by _topic(), cached here cleanup: list[Callable[[], object]] = field(default_factory=list) @@ -323,14 +327,38 @@ def _seed_taxonomy(c: Ctx) -> list[ContentType]: cascades the subtree, so one per root is enough). """ code = f"e2e-{c.stamp}" # codes are lowercase alphanumeric + hyphens, server-side - for suffix in ("", "-other"): - root = ContentType.define( - c.client, code + suffix, f"E2E {c.stamp}{suffix}", description="SDK e2e run" - ) + # The labels and descriptions carry real meaning on purpose: `scope()` *ranks + # the taxonomy against a question*, so a tree labelled "E2E " gives the + # scorer nothing to rank and every scope assertion would be flaky — on exactly + # the empty tenants that need seeding. The two roots are deliberately about + # different subjects, which is what makes `other_content_type` a real negative. + # Codes stay stamped: teardown keys on them, and undefine cascades, so it must + # only ever reach a root this run created. + for suffix, label, about in ( + ( + "", + "Technical Documentation", + "Maintenance manuals, technical orders, service bulletins and " + "incident reports.", + ), + ( + "-other", + "Financial Statements", + "Invoices, balance sheets, quarterly earnings and audit reports.", + ), + ): + root = ContentType.define(c.client, code + suffix, label, description=about) # undefine cascades, so the root takes its children and attributes with it. c.cleanup.append(lambda path=root.path: ContentType.undefine(c.client, path)) + c.seeded_roots.append(root.path) - child = ContentType.define(c.client, "child", "Child", parent=code) + child = ContentType.define( + c.client, + "child", + "Incident Reports", + parent=code, + description="Fault and incident reports filed against equipment.", + ) attr = ContentType.define_attribute( c.client, code, "e2e_marker", AttributeType.text ) @@ -348,7 +376,13 @@ def _seed_taxonomy(c: Ctx) -> list[ContentType]: results = ContentType.batch( c.client, [ - ContentTypeAction.define("batched", "Batched", parent=code), + ContentTypeAction.define( + "batched", + "Service Bulletins", + parent=code, + description="Manufacturer service bulletins and airworthiness " + "directives.", + ), # a raw dict stays a valid entry, the escape hatch both batches share { "action": "define_attribute", @@ -459,6 +493,261 @@ def facet_batch(c: Ctx) -> None: raise AssertionError(f"{MAX_FACET_ACTIONS + 1} actions should be refused") +@step +def scope(c: Ctx) -> None: + """content-types/scope: the three modes, the wire shapes, the knobs. + + Scores whatever taxonomy the account already has; run it after the + `content_types` step if you want something guaranteed to be there. The offline + suite already pins the client side (what gets sent, what `scope=` means), so + everything here is a *server* contract: shapes the curated models bet on, and + knobs only the live endpoint can honour. + """ + query = c.search_query or _topic(c) + + # Prompt mode: no model, so no completion, just the prompt to run yourself. + resolved = ContentType.scope(c.client, query) + c.resolved_scope = resolved + hits = resolved.content_types + _say( + f"scope({query!r}) → has_signal={resolved.has_signal}, " + f"{len(hits)} content type(s)" + ) + assert resolved.completion is None, ( + "no model= was passed, so there is no completion" + ) + if not hits: + _say( + "nothing scored — run with --only content_types --only scope", + typer.colors.YELLOW, + ) + return + # prompt_context describes the types that scored, so it's only guaranteed once + # something did — an empty taxonomy is a skipped step, not a bug. + assert resolved.prompt_context, "prompt mode must return a prompt_context" + for hit in hits[:3]: + _say(f" {hit.score:.2f} {hit.path} ({hit.doc_count} doc(s))") + + # prompt_version identifies the prompt that was built, as a `t: