{
 "version": 1,
 "note": "Terms a newcomer meets on this site. Every definition is written from the technique page named in `page`, which must say the same thing this entry does. `see` entries cross-reference other terms in this file by their `term` string.",
 "terms": [
  {
   "term": "agent",
   "page": "single-agent",
   "definition": "A model put in charge of a loop rather than one choice: it acts, reads the result, and decides what to do next and when to stop, instead of following steps your code chose in advance.",
   "see": [
    "agent loop",
    "cap",
    "stop condition"
   ]
  },
  {
   "term": "agent graph",
   "page": "agent-graphs",
   "definition": "A team of agents described as a graph: nodes are agents rather than fixed steps, and the output of one node (usually called the supervisor) picks which agent runs next from a fixed list of names.",
   "see": [
    "handoff",
    "checkpoint",
    "agent loop"
   ]
  },
  {
   "term": "agent loop",
   "page": "computer-use",
   "definition": "The repeating cycle behind every agent: the model proposes one action from what it currently sees, your code carries it out, and the result goes back to the model, until the model itself decides to stop."
  },
  {
   "term": "agentic AI",
   "page": "single-agent",
   "definition": "A loose umbrella for systems where the model, not your code, chooses each step and decides when the task is finished. On this site that begins at level 5; below it a person or a program picks the steps and the model only fills them in.",
   "see": [
    "agent",
    "agent loop",
    "level"
   ]
  },
  {
   "term": "agentic RAG",
   "aka": [
    "deep research"
   ],
   "page": "agentic-rag",
   "definition": "Retrieval where the model, not your code, decides how many times to search, what to search for next, and when it has read enough to answer, instead of searching once and answering once.",
   "see": [
    "RAG",
    "agent loop",
    "retrieval"
   ]
  },
  {
   "term": "AI gateway",
   "page": "ai-gateways",
   "definition": "One entry point every call to a model provider goes through instead of calling each provider directly, so provider keys, routing and fallback, per-caller budgets, caching, logging and policy checks sit in one place rather than in every application that calls a model.",
   "see": [
    "latency",
    "local model"
   ]
  },
  {
   "term": "always-on agent",
   "aka": [
    "always-on assistant",
    "agent teammate"
   ],
   "page": "agent-teammates",
   "definition": "An agent built around a computer of its own that keeps running between the moments a person talks to it, deciding on each scheduled check whether anything needs doing and acting under a fixed approval policy.",
   "see": [
    "approval policy",
    "audit trail",
    "long-running task"
   ]
  },
  {
   "term": "approval policy",
   "page": "agent-teammates",
   "definition": "A fixed, code-side table that sorts every action a model proposes into one of a few classes (run it automatically, queue it for a person, or never run it), regardless of what the model asked for."
  },
  {
   "term": "audit trail",
   "page": "safety",
   "definition": "A record of what an agent did and why, kept independent of the agent itself, that stands in for a person who was not there to catch a problem as it happened."
  },
  {
   "term": "automated prompt optimization",
   "page": "prompt-optimization",
   "definition": "Searching for a better prompt against a measured score instead of a person hand-editing the wording, tuning instructions, examples or weights the way a training run tunes a model."
  },
  {
   "term": "autonomy",
   "page": "who-approves-what",
   "definition": "How much of a task a system settles for itself. On this site it is not one quality but a question asked level by level: who decides the next step, and what a person is still holding at that level.",
   "see": [
    "human approval",
    "approval policy",
    "level"
   ]
  },
  {
   "term": "batch API",
   "page": "ops",
   "definition": "A way to submit many model requests at once for processing that finishes within a day rather than instantly, usually billed at roughly half the price of an ordinary synchronous call."
  },
  {
   "term": "best-of-N",
   "page": "inference-time-reasoning",
   "definition": "Generating several candidate answers to the same question and keeping the one a checker judges best, rather than trusting whatever the first attempt produces."
  },
  {
   "term": "briefing",
   "page": "briefing",
   "definition": "Writing down everything a model needs in order to act without guessing: the goal, context it lacks, constraints, what a finished result looks like, what to do when unsure, and the output format."
  },
  {
   "term": "calibrating trust",
   "aka": [
    "trust calibration"
   ],
   "page": "trust",
   "definition": "Keeping how much you rely on a model without checking in line with how often it has actually been right on tasks like the one in front of you, tracked per task type, not as one overall impression.",
   "see": [
    "reviewing",
    "eval"
   ]
  },
  {
   "term": "cap",
   "page": "single-agent",
   "definition": "A hard limit your code enforces on a loop, such as a maximum number of steps or tokens, that forces a stop the model cannot override or even see."
  },
  {
   "term": "checkpoint",
   "page": "workflow-graphs",
   "definition": "A saved snapshot of a workflow's shared state, written after a step, so a crashed or interrupted run can resume from that point instead of starting over from the beginning."
  },
  {
   "term": "chunk",
   "page": "rag",
   "definition": "A passage a document is cut into before indexing, small enough to embed and retrieve on its own, ideally holding one complete idea rather than splitting a sentence or table row in half."
  },
  {
   "term": "citation",
   "page": "rag",
   "definition": "A pointer from part of an answer back to the specific source passage it came from, so a reader can check whether the source actually supports what the answer claims."
  },
  {
   "term": "citation hit rate",
   "page": "rag",
   "definition": "The share of questions where every source a grading rule expects was actually cited in the model's answer, used to score how well a retrieval step is working."
  },
  {
   "term": "code execution",
   "page": "code-execution",
   "definition": "Letting the model write a small program instead of choosing among named tools, then running that program in a sandbox your code controls rather than trusting or interpreting it directly."
  },
  {
   "term": "coding agent",
   "page": "coding-agents",
   "definition": "A single agent whose tools read files, edit them, run commands and run tests, looping on a propose-edit-run-test cycle until its own tests pass or a cap ends the run.",
   "see": [
    "sandbox",
    "cap",
    "agent loop"
   ]
  },
  {
   "term": "compaction",
   "page": "long-horizon",
   "definition": "Replacing an aging context window with a short written summary once it nears its limit, so a session can carry forward what mattered without keeping the full history."
  },
  {
   "term": "computer use",
   "page": "computer-use",
   "definition": "Letting the model operate a real screen: it looks at a screenshot, picks one action such as a click or a keystroke, your code carries it out, and a new screenshot goes back."
  },
  {
   "term": "content credentials",
   "aka": [
    "C2PA"
   ],
   "page": "multimodal",
   "definition": "A record of where a piece of media came from, carried with the file as signed assertions, specified by C2PA rather than by one maker. It says whether that history validates and is free from tampering, not whether the history is good or bad."
  },
  {
   "term": "context engineering",
   "page": "context-engineering",
   "definition": "Deciding what goes into a model's request (which instructions, examples, documents and history) and in what order, since the model only knows what it was trained on and what the request contains.",
   "see": [
    "context window",
    "token",
    "prompt caching"
   ]
  },
  {
   "term": "context window",
   "page": "context-engineering",
   "definition": "The amount of text a model can read in one request; material that does not comfortably fit has to be trimmed, retrieved, or summarized before the model ever sees the question."
  },
  {
   "term": "cosine similarity",
   "page": "embeddings-search",
   "definition": "How closely two vectors point in the same direction, used to measure how related two pieces of text are once both have been turned into embeddings."
  },
  {
   "term": "credential vault",
   "page": "agent-teammates",
   "definition": "Secure storage that holds a password, key or payment method on an agent's behalf, so the agent can use it without ever seeing the raw value itself."
  },
  {
   "term": "delegating",
   "page": "delegating",
   "definition": "Deciding which parts of a task to hand to a model and which to keep, based on what a wrong answer would cost, how checkable the result is, and how reversible the action is.",
   "see": [
    "human approval",
    "approval policy",
    "least privilege"
   ]
  },
  {
   "term": "distillation",
   "page": "distillation",
   "definition": "Training a smaller model to imitate a larger model's outputs on a given task, so the smaller one can stand in for the larger one on that same narrow job."
  },
  {
   "term": "embedding",
   "page": "embeddings-search",
   "definition": "A list of floating-point numbers standing in for a piece of text's meaning, positioned so texts with similar meaning get vectors that point in similar directions.",
   "see": [
    "vector",
    "cosine similarity",
    "index"
   ]
  },
  {
   "term": "eval",
   "aka": [
    "evaluation"
   ],
   "page": "evals",
   "definition": "Running the same fixed set of questions against a system before and after a change, graded the same way both times, so a claim that the change helped can be checked instead of assumed.",
   "see": [
    "golden set",
    "grader",
    "rubric"
   ]
  },
  {
   "term": "evaluation framework",
   "page": "eval-frameworks",
   "definition": "A tool that holds a dataset of examples, runs a program or a prompt against every one of them, grades each result and lets two runs be compared, instead of a team building that machinery from scratch. Some also record a trace of what happened inside each run.",
   "see": [
    "eval",
    "golden set",
    "grader"
   ]
  },
  {
   "term": "few-shot example",
   "aka": [
    "few-shot learning"
   ],
   "page": "prompt-engineering",
   "definition": "One or more worked examples included in a prompt to show the model a format or pattern rather than only describing it in words."
  },
  {
   "term": "fine-tuning",
   "page": "adaptation",
   "definition": "Training a model further on your own examples so its behavior on that kind of task becomes more consistent, without repeating the same instructions in every request."
  },
  {
   "term": "function calling",
   "aka": [
    "tool call",
    "tool use"
   ],
   "page": "function-calling",
   "definition": "Giving the model a fixed list of actions your code defined, each with a name, a description and an argument schema, and letting it choose whether to use one, which one, and what arguments to send.",
   "see": [
    "schema",
    "sandbox"
   ]
  },
  {
   "term": "golden set",
   "page": "evals",
   "definition": "A fixed list of questions with a known right answer, or a rubric for judging one, run the same way before and after a change so two runs can be fairly compared."
  },
  {
   "term": "grader",
   "page": "evals",
   "definition": "The thing that scores an answer against a golden set, either by matching it exactly against a pattern or by having another model read it against a rubric, which is itself a judgment call worth checking by hand.",
   "see": [
    "rubric",
    "golden set"
   ]
  },
  {
   "term": "graph engineering",
   "page": "graph-engineering",
   "definition": "One phrase for two different techniques that happen to share a data structure. A knowledge graph connects information: entities and the relationships between them. A workflow graph or an agent graph connects work: steps, and who or what picks the next one.",
   "see": [
    "knowledge graph",
    "agent graph",
    "checkpoint"
   ]
  },
  {
   "term": "grounding",
   "aka": [
    "grounded"
   ],
   "page": "rag",
   "definition": "Tying a claim in an answer to a specific source that can actually be checked, the way RAG grounds an answer in retrieved documents instead of whatever the model remembers from training.",
   "see": [
    "citation",
    "RAG"
   ]
  },
  {
   "term": "guardrail",
   "page": "guardrails",
   "definition": "A check on what goes into a model or what comes out: an input filter, an output validator, a separate classifier trained to judge safety, a schema check. It is probabilistic and can be wrong in both directions, so it is never the control that holds; a code check that tests a specific fact is.",
   "see": [
    "prompt injection",
    "sandbox",
    "red teaming"
   ]
  },
  {
   "term": "hallucination",
   "page": "reviewing",
   "definition": "A fluent, confident answer that is not actually true or not supported by any real source; research argues this happens because training and grading reward a plausible guess over admitting uncertainty.",
   "see": [
    "grounding",
    "reviewing"
   ]
  },
  {
   "term": "handoff",
   "page": "agent-graphs",
   "definition": "The edge in an agent graph where one agent's output hands control to another named agent, chosen by a model call rather than a rule your code wrote in advance."
  },
  {
   "term": "harness",
   "aka": [
    "agent harness"
   ],
   "page": "agent-harness",
   "definition": "Everything around the model in an agent: the loop that calls it, the tool definitions it is shown and the code that runs them, what goes into its next request, whether an action needs approval, the caps on steps and tokens, the sandbox, and what gets logged. None of it is the model.",
   "see": [
    "agent loop",
    "cap",
    "stop condition",
    "loop engineering"
   ]
  },
  {
   "term": "human approval",
   "aka": [
    "human-in-the-loop"
   ],
   "page": "human-in-the-loop",
   "definition": "A pause your code inserts before something costly, irreversible or too uncertain to ship, handing the decision to a person instead of letting the run continue on its own."
  },
  {
   "term": "hybrid search",
   "page": "embeddings-search",
   "definition": "Running a keyword search and a meaning-based search over the same documents and merging the two result lists, so an exact identifier and a paraphrased question can both be found."
  },
  {
   "term": "index",
   "aka": [
    "vector index"
   ],
   "page": "rag",
   "definition": "The set of embeddings, or the keyword structure, built from a document set in advance so a later question can be compared against it and ranked, without re-reading every document."
  },
  {
   "term": "knowledge cutoff",
   "page": "chat",
   "definition": "The date after which a model's training data stops, so anything that changed after that date is not something the model actually knows, however confidently it answers."
  },
  {
   "term": "knowledge graph",
   "page": "knowledge-graphs",
   "definition": "Facts stored as entities and the relationships between them, so a chain of hops can join facts across documents instead of needing one passage to state the whole answer.",
   "see": [
    "retrieval",
    "RAG"
   ]
  },
  {
   "term": "latency",
   "page": "ops",
   "definition": "How long a request takes to complete, tracked separately from cost; caching, batching and routing to a smaller model are among the main ways to bring it down at volume."
  },
  {
   "term": "least privilege",
   "page": "safety",
   "definition": "Giving a system only the access its task actually needs, so an action it was never granted is one no instruction, however cleverly worded, can talk it into taking."
  },
  {
   "term": "level",
   "page": "single-agent",
   "definition": "One of the eight levels this site sorts a technique into. A new level starts where the answer to “who decides the next step” changes: nobody, you, your code, the model for one action, the model for every step, several models, the models including when to start."
  },
  {
   "term": "local model",
   "page": "ops",
   "definition": "A model run on your own hardware instead of called over an API, trading a per-call bill for hardware you own and buy, and for models small enough to fit on it."
  },
  {
   "term": "long-running task",
   "aka": [
    "long-horizon task"
   ],
   "page": "long-horizon",
   "definition": "Work that starts on a schedule or an event and continues across many separate sessions until its queue or goal is finished or a person steps in, with no single session seeing the one before it directly.",
   "see": [
    "queue",
    "checkpoint",
    "compaction"
   ]
  },
  {
   "term": "loop engineering",
   "page": "agent-harness",
   "definition": "Designing an agent's loop on purpose (what starts it, what it repeats, what stops it) rather than only writing its prompt. Anthropic's Claude Code team defines loops as agents repeating cycles of work until a stop condition is met.",
   "see": [
    "harness",
    "agent loop",
    "stop condition"
   ]
  },
  {
   "term": "LoRA",
   "aka": [
    "low-rank adaptation",
    "adapter"
   ],
   "page": "adaptation",
   "definition": "A small, trainable add-on layered onto a model's frozen weights instead of retraining the whole model, cutting the parameters and memory a fine-tuning run needs by orders of magnitude."
  },
  {
   "term": "MCP",
   "aka": [
    "Model Context Protocol"
   ],
   "page": "mcp",
   "definition": "A standard way for an application to connect to servers that expose tools, resources and prompts to a model, instead of a developer wiring each integration by hand.",
   "see": [
    "function calling",
    "sandbox"
   ]
  },
  {
   "term": "memory",
   "page": "memory",
   "definition": "Keeping information from one conversation to the next by writing facts down somewhere and reading them back into a later, otherwise unrelated conversation, since a chat has no memory of its own."
  },
  {
   "term": "memory layer",
   "page": "memory",
   "definition": "The part of a product that writes facts down after one conversation and reads them back into a later one. Two different things get called this: a summary, cheap to reread but lossy, and a record kept on its own and searched on demand.",
   "see": [
    "memory",
    "recall",
    "context window"
   ]
  },
  {
   "term": "model-decided step",
   "page": "single-agent",
   "definition": "A step in a recorded run where the model's own output chose what happened next (which tool, which query, whether to stop), as opposed to a step your code decided regardless of what the model said.",
   "see": [
    "trace",
    "cap"
   ]
  },
  {
   "term": "multi-agent system",
   "page": "orchestrator-workers",
   "definition": "Several agents working on a task, using the same underlying model or different models: a lead agent splitting the work among others, agents handing work to each other across a graph, or two agents checking each other's output. That is level 6 on this site.",
   "see": [
    "orchestrator",
    "agent graph",
    "review and debate"
   ]
  },
  {
   "term": "multimodal",
   "page": "multimodal",
   "definition": "A model that takes or produces more than text (images, audio, video and documents) inside the same one-call request and response shape as an ordinary chat message."
  },
  {
   "term": "observability",
   "page": "observability",
   "definition": "Recording what each run did in enough detail that a bad result can be traced back to the step that caused it: which passages a retrieval step picked, which tool the model called and with what arguments, which branch a workflow took, and what each step spent.",
   "see": [
    "trace",
    "latency",
    "audit trail"
   ]
  },
  {
   "term": "orchestrator",
   "aka": [
    "lead agent"
   ],
   "page": "orchestrator-workers",
   "definition": "A lead model that reads a task, decides how to split it, hands each piece to a worker, and combines what comes back, rather than doing the whole task itself."
  },
  {
   "term": "organization of agents",
   "aka": [
    "agent swarm"
   ],
   "page": "organizations-swarms",
   "definition": "Several standing agents with distinct roles and a shared goal, where the roster itself keeps running and changing over time rather than being assembled fresh for one job and torn down after."
  },
  {
   "term": "parallel calls",
   "aka": [
    "parallelization",
    "sectioning",
    "voting"
   ],
   "page": "parallelization",
   "definition": "Running more than one model call at the same time instead of one after another, then combining the results in code, either splitting one task into independent parts or running the same task several times to vote."
  },
  {
   "term": "progressive disclosure",
   "page": "skills",
   "definition": "Loading only a skill's short name and description into context up front, and its full instructions only once the model actually decides to use it, so unused skills cost almost nothing."
  },
  {
   "term": "prompt",
   "page": "prompt-engineering",
   "definition": "The request text sent to a model in one call: instructions, examples and the question, everything the model sees that is not already baked into its training.",
   "see": [
    "system prompt",
    "few-shot example"
   ]
  },
  {
   "term": "prompt caching",
   "page": "context-engineering",
   "definition": "Reusing a matching, unchanged prefix of a request across calls at a reduced billing rate, which is why makers recommend putting content that never changes first and content that changes every call last."
  },
  {
   "term": "prompt chaining",
   "page": "prompt-chaining",
   "definition": "Splitting one task into a fixed sequence of steps and handing each step's output to the next, with your code deciding how many steps there are and what gate sits between them."
  },
  {
   "term": "prompt engineering",
   "page": "prompt-engineering",
   "definition": "Writing the request itself well (instructions, examples, an assigned role, a required format, sometimes asking the model to reason first) to get a more consistent result from a single call."
  },
  {
   "term": "prompt injection",
   "page": "safety",
   "definition": "Text written to look like an instruction, hidden in the user's message or in retrieved or tool-returned content, that tries to redirect what the model does instead of answering the actual question.",
   "see": [
    "guardrail",
    "sandbox"
   ]
  },
  {
   "term": "quantization",
   "page": "local-inference",
   "definition": "Storing a model's weights at lower precision so a large model fits smaller hardware and may run faster. llama.cpp's own documentation says it shrinks the model and can speed up inference, and that it may cost some accuracy. Name the exact level you ran, not just \"4-bit\".",
   "see": [
    "local model",
    "eval"
   ]
  },
  {
   "term": "queue",
   "page": "long-horizon",
   "definition": "A persisted list of pending work items a long-running task drains one at a time across separate sessions, surviving a crash because the queue is checkpointed after every item, not held only in memory."
  },
  {
   "term": "RAG",
   "aka": [
    "retrieval-augmented generation"
   ],
   "page": "rag",
   "definition": "Searching your own documents for the passages closest to a question, putting those passages in the prompt, and asking the model to answer once using only what it was given.",
   "see": [
    "retrieval",
    "embedding",
    "chunk",
    "agentic RAG",
    "citation"
   ]
  },
  {
   "term": "RAG 2.0",
   "page": "rag",
   "definition": "A label rather than a defined technique: different people use it for different bundles of retrieval improvements: better chunking, reranking, hybrid search, retrieval the model itself drives. This site has no page under that name; the pages on RAG, embeddings and search, and agentic RAG cover the substance.",
   "see": [
    "RAG",
    "reranking",
    "hybrid search",
    "agentic RAG"
   ]
  },
  {
   "term": "reasoning effort",
   "aka": [
    "extended thinking"
   ],
   "page": "inference-time-reasoning",
   "definition": "How much a model is allowed to work through a problem in its own words before answering; makers recommend more of it for math, debugging and planning and less for simple lookups."
  },
  {
   "term": "reasoning model",
   "aka": [
    "thinking model"
   ],
   "page": "inference-time-reasoning",
   "definition": "A model trained, usually with reinforcement learning, to write out a long chain of thought before it answers, and to notice and fix its own mistakes along the way. OpenAI's o1 (September 2024) was the first sold as one.",
   "see": [
    "reasoning effort",
    "test-time compute"
   ]
  },
  {
   "term": "recall",
   "page": "memory",
   "definition": "The operation that ranks stored facts or documents against a new question and returns the closest matches, the same mechanism whether the store is a document corpus or a small set of personal facts."
  },
  {
   "term": "red teaming",
   "page": "red-teaming",
   "definition": "Attacking your own system on purpose, before someone else does it without permission, across the system around the model and not only the model itself. What matters is not the report but the change it forces, most durably a test that keeps failing until the finding is fixed.",
   "see": [
    "guardrail",
    "prompt injection",
    "eval"
   ]
  },
  {
   "term": "reranking",
   "page": "embeddings-search",
   "definition": "A second pass over a first, larger cut of retrieved candidates, scored by a model trained specifically to sort results by relevance to the original query."
  },
  {
   "term": "retrieval",
   "page": "rag",
   "definition": "Searching an index of document chunks for the ones closest to a question, and keeping a fixed number of the best matches to hand to the model."
  },
  {
   "term": "reviewing",
   "page": "reviewing",
   "definition": "Checking work you did not do yourself before it goes anywhere: pulling out specific claims and checking them against their source, then checking what is missing, then judging the result.",
   "see": [
    "hallucination",
    "citation"
   ]
  },
  {
   "term": "review and debate",
   "aka": [
    "debate and review",
    "multi-agent debate"
   ],
   "page": "debate-review",
   "definition": "A separate agent, with its own context and often its own retrieval, checks or argues with another agent's work and decides whether to accept it, instead of one fixed, code-owned test.",
   "see": [
    "write and check"
   ]
  },
  {
   "term": "routing",
   "page": "routing",
   "definition": "Looking at an input, deciding which of several fixed kinds it is, and sending it down the handler built for that kind, with a real fallback for whatever fits none of them.",
   "see": [
    "parallel calls",
    "structured output"
   ]
  },
  {
   "term": "rubric",
   "page": "evals",
   "definition": "A written checklist a grader model reads an answer against when the answer is too open-ended to match against a fixed pattern, used to turn a judgment call into a repeatable score."
  },
  {
   "term": "sandbox",
   "page": "code-execution",
   "definition": "An isolated environment with no network access and fixed resource limits, where code the model wrote is actually run, so what the model produces is data your code hands to an interpreter, never code it trusts directly."
  },
  {
   "term": "schema",
   "page": "structured-output",
   "definition": "A fixed shape for a reply (named fields with defined types) that a model's output is constrained to match, so downstream code can parse it without guessing at its structure."
  },
  {
   "term": "self-consistency",
   "page": "inference-time-reasoning",
   "definition": "Asking the same question several times as independent calls and returning whichever answer the largest share of the samples agree on, instead of trusting a single attempt."
  },
  {
   "term": "skill",
   "page": "skills",
   "definition": "Instructions an agent keeps on the shelf until it decides it needs them, loaded into context only when triggered, instead of text repeated into every single turn like a system prompt.",
   "see": [
    "progressive disclosure",
    "system prompt"
   ]
  },
  {
   "term": "stateless MCP",
   "aka": [
    "sessionless MCP"
   ],
   "page": "mcp",
   "definition": "The current MCP specification, revision 2026-07-28, removed the initialize handshake and protocol-level sessions. Every request now carries its own protocol version and capabilities, and a server that needs state across calls returns an explicit, server-minted handle the client passes back as an ordinary tool argument.",
   "see": [
    "MCP",
    "function calling"
   ]
  },
  {
   "term": "stop condition",
   "page": "single-agent",
   "definition": "The test a model-driven loop uses to decide it is finished and should give a final answer rather than take another action; a cap can force a stop the model never actually reaches."
  },
  {
   "term": "structured output",
   "page": "structured-output",
   "definition": "A reply constrained to come back in a fixed shape, such as a JSON object with named fields, instead of a paragraph your code has to parse by guessing."
  },
  {
   "term": "synthetic data",
   "page": "synthetic-data",
   "definition": "Training examples generated with a model rather than collected from real use, for a later fine-tuning or evaluation run."
  },
  {
   "term": "System One model",
   "page": "structured-output",
   "definition": "TypeSafe AI's name for a model that generates no text and returns a typed decision with a probability in one fast pass, for use inside software. Jev (September 2026) is the first. The name borrows psychology's fast System 1, as against slow, deliberate System 2.",
   "see": [
    "structured output",
    "reasoning model"
   ]
  },
  {
   "term": "system prompt",
   "page": "chat",
   "definition": "The instructions sent to a model at the start of a request, separate from what the user or the retrieved content says, setting how it should behave for that call."
  },
  {
   "term": "test-time compute",
   "aka": [
    "inference-time compute"
   ],
   "page": "inference-time-reasoning",
   "definition": "Computation spent while answering, as opposed to while training. Reasoning models made it a dial: more thinking time, better answers on hard problems, at the cost of seconds and output tokens.",
   "see": [
    "reasoning model",
    "reasoning effort"
   ]
  },
  {
   "term": "token",
   "page": "context-engineering",
   "definition": "The unit a model's input and output are measured and billed in; every cost strip on this site counts tokens in and tokens out for the run it illustrates."
  },
  {
   "term": "trace",
   "page": "single-agent",
   "definition": "A recorded, step-by-step account of a run (every model call, every tool call, and whether each step was decided by code or by the model) used to explain how an answer was actually produced.",
   "see": [
    "model-decided step",
    "eval"
   ]
  },
  {
   "term": "trajectory",
   "page": "eval-frameworks",
   "definition": "The path a run took to reach its answer: which tools were called, in what order, with what arguments. Scoring it is a separate measurement from scoring the final answer, and it gives partial credit for the steps a run got right.",
   "see": [
    "trace",
    "eval",
    "agent loop"
   ]
  },
  {
   "term": "vector",
   "page": "embeddings-search",
   "definition": "A fixed-length list of numbers representing a piece of text, an embedding, positioned in space so that texts with related meaning end up close together."
  },
  {
   "term": "vector database",
   "aka": [
    "vector store"
   ],
   "page": "embeddings-search",
   "definition": "A database built to hold embeddings and rank them by similarity to a query vector: pgvector, Pinecone, Weaviate and Qdrant are four. It stores the index a retrieval step searches; the embeddings it stores come from an embedding model, not from the database.",
   "see": [
    "embedding",
    "index",
    "vector"
   ]
  },
  {
   "term": "vision-language-action model",
   "aka": [
    "VLA"
   ],
   "page": "embodied",
   "definition": "A model that converts vision and language input directly into motor control, letting a robot take a physical action rather than only produce text."
  },
  {
   "term": "voice agent",
   "page": "voice-agents",
   "definition": "A single agent you talk to instead of type to, in real time, that additionally has to decide when a caller has stopped talking and what to do if they interrupt."
  },
  {
   "term": "write and check",
   "aka": [
    "evaluator-optimizer"
   ],
   "page": "evaluator-optimizer",
   "definition": "A loop where one prompt writes a draft and a separate prompt checks it against one specific, testable criterion, revising and rechecking until it passes or a fixed revision cap is reached.",
   "see": [
    "review and debate",
    "rubric"
   ]
  }
 ]
}