From ce3f03cb2d854dd4649c7b228e6d3affdf41bd8a Mon Sep 17 00:00:00 2001 From: RapidPoseidon Date: Fri, 25 Sep 2026 13:32:03 +0000 Subject: [PATCH 1/7] feat(agents): hint once per session and flag stale installed skills Co-Authored-By: Claude Opus 5.5 Co-Authored-By: lino@rapidata.ai <68745352+LinoGiger@users.noreply.github.com> --- .github/workflows/release_and_publish.yml | 5 +- docs/ai_agents.md | 8 + src/rapidata/__main__.py | 32 +- src/rapidata/_agent_hint.py | 238 +++++- src/rapidata/_skill/SKILL.md | 926 ++++++++++++++++++++++ tests/conftest.py | 33 + tests/test_agent_hint.py | 195 ++++- tests/test_main.py | 64 +- 8 files changed, 1430 insertions(+), 71 deletions(-) create mode 100644 src/rapidata/_skill/SKILL.md diff --git a/.github/workflows/release_and_publish.yml b/.github/workflows/release_and_publish.yml index 9514284eb0..fe869d0992 100644 --- a/.github/workflows/release_and_publish.yml +++ b/.github/workflows/release_and_publish.yml @@ -142,9 +142,12 @@ jobs: f.write(f'__version__ = "{version}"\n') EOF + - name: Refresh bundled agent skill + run: curl -fsSL https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/SKILL.md -o src/rapidata/_skill/SKILL.md + - name: Commit and push changes run: | - git add pyproject.toml src/rapidata/__init__.py + git add pyproject.toml src/rapidata/__init__.py src/rapidata/_skill/SKILL.md git commit -m "Bump version from ${{ steps.update_version.outputs.old_version }} to ${{ steps.update_version.outputs.new_version }}" git push origin ${{ github.event.inputs.branch }} diff --git a/docs/ai_agents.md b/docs/ai_agents.md index 9226cf0555..2e346c1856 100644 --- a/docs/ai_agents.md +++ b/docs/ai_agents.md @@ -126,3 +126,11 @@ Or update every skill you've installed at once: ```bash npx skills update ``` + +Copies written by `python -m rapidata skill --install` carry a stamp of the version they were made from. When a coding agent imports the SDK, it checks that stamp against the live skill at most once a day and, if it is out of date, tells the agent to run `python -m rapidata skill --install` again. + +## The import-time hint + +When the SDK is imported by a coding agent (Claude Code, Codex, Cursor, Gemini CLI) that has no copy of the skill installed, it prints a short pointer to `python -m rapidata skill` on stderr. It shows once per agent session and stops as soon as that session reads the guide. Humans running the SDK directly never see it. + +Processes an agent merely started — a dev server, a script whose stderr is parsed — inherit its environment and would show the hint too. Set `RAPIDATA_AGENT_HINT=0` for those. diff --git a/src/rapidata/__main__.py b/src/rapidata/__main__.py index 5496d29706..af48ff1676 100644 --- a/src/rapidata/__main__.py +++ b/src/rapidata/__main__.py @@ -14,6 +14,7 @@ import argparse import os import sys +from importlib import resources from pathlib import Path import requests @@ -25,6 +26,8 @@ SKILL_INSTALL_PATHS, SKILL_RAW_URL, mark_skill_read, + record_live_skill, + stamp_skill, ) @@ -34,10 +37,22 @@ def fetch_skill(timeout: float = 10) -> str: return response.text +def bundled_skill() -> str | None: + """The copy of the skill refreshed into the wheel at release time, for when GitHub is unreachable.""" + try: + return ( + resources.files("rapidata") + .joinpath("_skill/SKILL.md") + .read_text(encoding="utf-8") + ) + except (OSError, ModuleNotFoundError): + return None + + def install_skill(root: Path, agent: str, content: str) -> Path: target = root / SKILL_INSTALL_PATHS[agent] target.parent.mkdir(parents=True, exist_ok=True) - target.write_text(content, encoding="utf-8") + target.write_text(stamp_skill(content), encoding="utf-8") return target @@ -151,13 +166,22 @@ def main(argv: list[str] | None = None) -> int: try: content = fetch_skill() + record_live_skill(content) except requests.RequestException as e: - print(f"Could not fetch the skill ({e}).", file=sys.stderr) + bundled = bundled_skill() + if bundled is None: + print(f"Could not fetch the skill ({e}).", file=sys.stderr) + print( + f"Read it online instead: {SKILL_RAW_URL} or {LLMS_FULL_URL}", + file=sys.stderr, + ) + return 1 print( - f"Read it online instead: {SKILL_RAW_URL} or {LLMS_FULL_URL}", + f"Could not fetch the latest skill ({e}); using the copy bundled with " + f"rapidata {__version__}. Latest: {SKILL_RAW_URL}", file=sys.stderr, ) - return 1 + content = bundled mark_skill_read() if args.install: diff --git a/src/rapidata/_agent_hint.py b/src/rapidata/_agent_hint.py index 963a239f81..ce169916de 100644 --- a/src/rapidata/_agent_hint.py +++ b/src/rapidata/_agent_hint.py @@ -1,23 +1,42 @@ -"""Point a coding agent at the maintained SDK guide the first time it imports ``rapidata``. +"""Point a coding agent at the maintained SDK guide when it imports ``rapidata``. Agents explore a freshly installed SDK with ``import rapidata`` / ``dir()`` / ``inspect`` before they write a script, and they read stderr but not -docstrings, so the pointer is printed at import time. It stays silent once the -guide has been read (``python -m rapidata skill`` writes :data:`SKILL_READ_MARKER`) -or installed into the project, and outside a known agent runtime. +docstrings, so the pointer is printed at import time: -Kept free of SDK imports: ``rapidata/__init__.py`` calls it before loading the client. +- Once per agent session until that session reads the guide with + ``python -m rapidata skill``. Sessions are told apart by the id the runtime + exports (:data:`_SESSION_ENV_VARS`); runtimes without one get a read that + expires after :data:`ANON_READ_TTL`. +- Never while a copy of the skill the agent loads on its own is installed + (project, user level or the Claude Code plugin), unless a copy written by + ``--install`` no longer matches the live skill. That is checked at most once + per :data:`FRESHNESS_TTL` with a :data:`FRESHNESS_TIMEOUT` request, and then + the pointer asks for a reinstall instead. + +``RAPIDATA_AGENT_HINT=0`` switches it off for processes an agent merely started. +State lives in :data:`STATE_FILE`, falling back to the temp dir when a sandbox +makes the home directory read-only. Kept free of SDK imports: +``rapidata/__init__.py`` calls it before loading the client. """ from __future__ import annotations +import hashlib +import json import os +import re import sys +import tempfile +import time +import urllib.request +from datetime import datetime, timezone from pathlib import Path SKILL_RAW_URL = "https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/SKILL.md" LLMS_FULL_URL = "https://docs.rapidata.ai/llms-full.txt" AGENT_DOCS_URL = "https://docs.rapidata.ai/ai_agents/" +PLUGIN_NAME = "rapidata-sdk-plugin" # Where each agent picks up a project-local skill file, relative to the project root. SKILL_INSTALL_PATHS: dict[str, str] = { @@ -27,7 +46,19 @@ "generic": "AGENTS.md", } -SKILL_READ_MARKER = Path.home() / ".config" / "rapidata" / "skill-read" +# User-level skill files the runtimes load in every project, relative to the home directory. +USER_SKILL_PATHS: dict[str, str] = { + "claude": ".claude/skills/rapidata/SKILL.md", + "codex": ".codex/skills/rapidata/SKILL.md", +} + +STATE_FILE = Path.home() / ".config" / "rapidata" / "agent-state.json" +FALLBACK_STATE_FILE = Path(tempfile.gettempdir()) / "rapidata-agent-state.json" + +ANON_READ_TTL = 12 * 3600 +FRESHNESS_TTL = 24 * 3600 +FRESHNESS_TIMEOUT = 1.0 +_PRUNE_AFTER = 7 * 24 * 3600 # Env var each agent runtime exports -> the name reported in traces. Specific # vars come before the generic AI_AGENT so the first match names the runtime. @@ -40,15 +71,18 @@ "AI_AGENT": "unknown", } +_SESSION_ENV_VARS = ("CLAUDE_CODE_SESSION_ID", "CODEX_THREAD_ID", "CODEX_SESSION_ID") + +_STAMP_RE = re.compile(r"\n") + AGENT_HINT = ( - "rapidata: coding agent detected. Before exploring the installed source, " - "read the maintained SDK guide (job types, audiences, result fields, common mistakes):\n" + "rapidata: coding agent detected. Read the maintained SDK guide before exploring " + "the installed source (skip if this session already read it):\n" " python -m rapidata skill # print the guide\n" " python -m rapidata skill --install # keep it in this project\n" f" {LLMS_FULL_URL}\n" "Before the first RapidataClient(), run `python -m rapidata status`. If it reports not logged in,\n" - "run `python -m rapidata login` and show the user the URL it prints (it waits up to 5 minutes).\n" - "This message stops once the guide has been read. RAPIDATA_AGENT_HINT=0 silences it." + "run `python -m rapidata login` and show the user the URL it prints (it waits up to 5 minutes)." ) @@ -63,33 +97,161 @@ def detected_coding_agent() -> str | None: def running_under_coding_agent() -> bool: - """Return True when a known coding-agent runtime drives this process. - - ``RAPIDATA_AGENT_HINT`` overrides detection: ``0``/``false``/``no`` never - hints, ``1``/``true``/``yes`` always hints. - """ - override = os.environ.get("RAPIDATA_AGENT_HINT", "").lower() - if override in ("0", "false", "no"): + """Return True when a known coding-agent runtime drives this process and the hint is not switched off.""" + if os.environ.get("RAPIDATA_AGENT_HINT", "").lower() in ("0", "false", "no"): return False - if override in ("1", "true", "yes"): - return True return detected_coding_agent() is not None -def skill_seen(root: Path | None = None) -> bool: - """Return True when the guide was read on this machine or installed under ``root`` (default: cwd).""" - if SKILL_READ_MARKER.is_file(): - return True - root = root or Path.cwd() - return any((root / rel).is_file() for rel in SKILL_INSTALL_PATHS.values()) +def skill_digest(content: str) -> str: + return hashlib.sha256(content.encode("utf-8")).hexdigest() + + +def stamp_skill(content: str, now: datetime | None = None) -> str: + """Return ``content`` with a provenance line after its front matter, so staleness can be checked later.""" + fetched = (now or datetime.now(timezone.utc)).strftime("%Y-%m-%dT%H:%M:%SZ") + stamp = ( + f"\n" + ) + if content.startswith("---\n"): + end = content.find("\n---\n", 4) + if end != -1: + cut = end + len("\n---\n") + return content[:cut] + stamp + content[cut:] + return stamp + content + + +def _installed_digest(text: str) -> str | None: + """Digest of the live skill this installed copy was made from, or None when it is not the Rapidata skill.""" + match = _STAMP_RE.search(text) + if match: + return match.group(1) + # Copies written by --install before stamping existed are verbatim. + if text.startswith("---\nname: rapidata\n"): + return skill_digest(text) + return None + + +def _load_state() -> dict: + state: dict = {"reads": {}} + for path in (FALLBACK_STATE_FILE, STATE_FILE): + try: + loaded = json.loads(path.read_text(encoding="utf-8")) + except (OSError, ValueError): + continue + if isinstance(loaded, dict): + reads = {**state["reads"], **loaded.get("reads", {})} + state.update(loaded) + state["reads"] = reads + return state + + +def _save_state(state: dict) -> None: + now = time.time() + state["reads"] = { + k: t for k, t in state.get("reads", {}).items() if now - t < _PRUNE_AFTER + } + payload = json.dumps(state) + for path in (STATE_FILE, FALLBACK_STATE_FILE): + try: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(payload, encoding="utf-8") + return + except OSError: + continue + + +def _session_key() -> str | None: + return next( + ( + f"{var}:{os.environ[var]}" + for var in _SESSION_ENV_VARS + if os.environ.get(var) + ), + None, + ) + + +def _read_this_session(state: dict) -> bool: + reads = state.get("reads", {}) + key = _session_key() + if key: + return key in reads + read_at = reads.get("anonymous") + return read_at is not None and time.time() - read_at < ANON_READ_TTL def mark_skill_read() -> None: + state = _load_state() + state["reads"][_session_key() or "anonymous"] = time.time() + _save_state(state) + + +def record_live_skill(content: str) -> None: + """Remember the live skill's digest so the next freshness check can skip the network.""" + state = _load_state() + state["live_sha"] = skill_digest(content) + state["live_checked_at"] = time.time() + _save_state(state) + + +def _plugin_installed() -> bool: + config_dir = Path(os.environ.get("CLAUDE_CONFIG_DIR") or Path.home() / ".claude") + try: + plugins = json.loads( + (config_dir / "plugins" / "installed_plugins.json").read_text( + encoding="utf-8" + ) + ).get("plugins", {}) + except (OSError, ValueError, AttributeError): + return False + return any(name.split("@", 1)[0] == PLUGIN_NAME for name in plugins) + + +def installed_copies(root: Path | None = None) -> list[tuple[str, Path, Path, str]]: + """Return ``(agent, install_root, path, digest)`` for each Rapidata skill file in the project ``root`` or the home directory.""" + root = root or Path.cwd() + candidates = [(a, root, rel) for a, rel in SKILL_INSTALL_PATHS.items()] + candidates += [(a, Path.home(), rel) for a, rel in USER_SKILL_PATHS.items()] + copies = [] + for agent, base, rel in candidates: + path = base / rel + try: + digest = _installed_digest(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError): + continue + if digest: + copies.append((agent, base, path, digest)) + return copies + + +def _fetch_live_digest() -> str | None: try: - SKILL_READ_MARKER.parent.mkdir(parents=True, exist_ok=True) - SKILL_READ_MARKER.touch() - except OSError: - pass + with urllib.request.urlopen(SKILL_RAW_URL, timeout=FRESHNESS_TIMEOUT) as resp: + return skill_digest(resp.read().decode("utf-8")) + except Exception: + return None + + +def _live_digest(state: dict) -> str | None: + if time.time() - state.get("live_checked_at", 0) < FRESHNESS_TTL: + return state.get("live_sha") + live = _fetch_live_digest() + # Recorded on failure too, so an offline machine pays the timeout once a day, not per import. + state["live_checked_at"] = time.time() + if live: + state["live_sha"] = live + _save_state(state) + return state.get("live_sha") + + +def _stale_hint(agent: str, base: Path, path: Path) -> str: + cmd = "python -m rapidata skill --install" + if agent != "claude": + cmd += f" --agent {agent}" + if base != Path.cwd(): + cmd += f" --dir {base}" + return f"rapidata: the installed Rapidata skill at {path} is outdated. Update it with: {cmd}" def _running_the_cli() -> bool: @@ -101,13 +263,23 @@ def _running_the_cli() -> bool: def agent_hint() -> str | None: - """Return the hint when an agent imports the SDK without having read the guide, else None.""" + """Return the pointer an agent should see on import, or None when it has what it needs.""" try: - if not running_under_coding_agent() or _running_the_cli() or skill_seen(): + if not running_under_coding_agent() or _running_the_cli(): + return None + state = _load_state() + copies = installed_copies() + if copies: + live = _live_digest(state) + for agent, base, path, digest in copies: + if live and digest != live: + return _stale_hint(agent, base, path) + return None + if _plugin_installed() or _read_this_session(state): return None - except OSError: + return AGENT_HINT + except Exception: return None - return AGENT_HINT def print_agent_hint() -> None: diff --git a/src/rapidata/_skill/SKILL.md b/src/rapidata/_skill/SKILL.md new file mode 100644 index 0000000000..3b4cb22a4e --- /dev/null +++ b/src/rapidata/_skill/SKILL.md @@ -0,0 +1,926 @@ +--- +name: rapidata +description: Explains how to use the Rapidata API to get real and fast human annotations for your data. Use when writing code that creates labeling tasks, compares models, collects human feedback, or integrates with the Rapidata Python SDK. +--- + +# Rapidata Python SDK + +Rapidata connects you with distributed human labelers worldwide for fast, high-quality data annotation. The SDK lets you create labeling tasks, manage annotator audiences, and retrieve results programmatically. + +## Before you start: check the skill is up to date + +This skill is pinned to **Rapidata SDK v3.25.7**. Run this check **once at the start of a Rapidata task** (not on every call) to confirm the user's runtime matches the skill: + +```bash +python -c "import rapidata; print(rapidata.__version__)" 2>/dev/null \ + || pip show rapidata 2>/dev/null | awk -F': ' '/^Version:/{print $2}' +``` + +Compare the output to the pinned version above: + +- **Installed > pinned** — this skill is **outdated**. The SDK may have new features, renamed methods, or changed signatures that this skill does not document. + 1. Suggest updating the skill: in Claude Code, `/plugin marketplace update rapidata-sdk-marketplace` (or `/plugin` → `rapidata-sdk-plugin` → update); outside the plugin, `python -m rapidata skill --install` rewrites the project-local copy. + 2. Tell the user clearly: + > ⚠️ The Rapidata skill is pinned to v3.25.7 but v{installed} is installed — the skill docs may be out of date. I've suggested updating the plugin; if the update isn't available yet, I'll proceed with the documented API and flag any surprises. + 3. Proceed using the documented API. If you hit an unexpected error (missing attribute, changed signature), stop and tell the user the skill is likely the cause — don't guess at the new API. + +- **Installed < pinned** — the user's runtime is older than this skill. Suggest `pip install -U rapidata` so the runtime matches. + +- **Match, or rapidata not installed** — proceed normally. (If not installed, the installation section below is the first step anyway.) + +## Installation & Authentication + +```bash +pip install -U rapidata +``` + +```python +from rapidata import RapidataClient +client = RapidataClient() # Credentials resolved: env vars → ~/.config/rapidata/credentials.json → browser login + +# Or pass credentials directly +client = RapidataClient(client_id="...", client_secret="...") +# Credentials are saved to ~/.config/rapidata/credentials.json + +# Environment variables (useful for headless/container deployments): +# RAPIDATA_CLIENT_ID, RAPIDATA_CLIENT_SECRET — authenticate without a browser +# RAPIDATA_ENVIRONMENT — override the API endpoint (default: rapidata.ai) +# RAPIDATA_TOKEN_FILE — read a shared access token from this file (see below) +# Empty values are treated as unset and fall through to the next resolution layer. +``` + +### Sharing a token across many workers (distributed training) + +When Rapidata is queried from a large distributed job (e.g. a ranking flow hit from hundreds or thousands of GPU workers), don't let every worker authenticate on its own — all their tokens expire at the same instant and the simultaneous re-auth looks like a coordinated burst that gets rate-limited. Instead, authenticate **once** and share the token via a file: + +```python +from rapidata import RapidataClient + +# Coordinator (holds the client credentials): keep a shared token file fresh +coordinator = RapidataClient(leeway=300) # renew the token 5 min before expiry +coordinator.maintain_token_file("/shared/rapidata_token.json").join() +# maintain_token_file() writes the file immediately, then keeps rewriting it +# atomically from a background thread (every 60s by default), creating the +# directory if needed. .join() blocks forever — drop it if the coordinator +# also does other work (e.g. rank 0 both trains and refreshes the token). + +# Workers (never see the client secret): point the SDK at that file +client = RapidataClient(token_file="/shared/rapidata_token.json") +# or set RAPIDATA_TOKEN_FILE and construct with no arguments. +# The SDK reads the token at startup and re-reads the file whenever the +# in-memory token is within 60s of expiry (configurable via leeway). +``` + +The token file contains a bearer token — write it only to storage your job alone can access. To roll your own file writer, export the current token with `coordinator.get_token()` (cheap to call at any frequency; only contacts the auth server once the token is within `leeway` of expiry), write it atomically, and keep the absolute `expires_at` field so workers know when to re-read. + +The file is just one transport. To move the token over any transport (key-value store, RPC, secret manager, message queue), pair `coordinator.get_token()` with `worker.set_token(fresh_token)`, which injects a fresh token into a running worker client (effective from its next request) without reconstructing it. This supports both a **push** system (the coordinator distributes a fresh token to every worker before the old one expires) and a **pull** system (each worker periodically fetches the current token from your own endpoint). Whatever the transport, pass the complete token object around and keep its absolute `expires_at` field. On older SDKs without `token_file`, pass the token dict directly with `RapidataClient(token=json.load(f))` — but the SDK never re-reads it, so each worker must call `set_token()` (or construct a new client) when the token expires. + +## Core Concepts + +- **Job Definition**: A reusable configuration template for a labeling task (task type, instruction, datapoints, answer options) +- **Audience**: A group of annotators selected and qualified for **one specific task** — via qualification examples and/or recruitment filters chosen to match that task. The whole point is to put the *right* people on *that* task. An audience trained for one task is **not** meant to be reused on a different, unrelated task: its qualification examples define what "good" means for the original task only, so reusing it elsewhere silently loses the quality it was built for. Reusing the same audience for repeated or scheduled runs of the **same** task is exactly right; for a different task, create a new audience. Three kinds: + - **global** — the generic baseline pool for tasks that need no special qualification; instant, no setup. Use `client.audience.get_audience_by_id("global")`. + - **curated** — pre-trained on a domain (e.g. alignment via `aud_MU1GZYoESyO`). + - **custom** — trained with your own task-specific qualification examples (`client.audience.create_audience(...)` + `add_*_example(...)` + `start_recruiting()`). ⚠️ Recruiting is **explicit**: a custom audience recruits nobody until you add **≥3 qualification examples** *and then* call `audience.start_recruiting()`. Adding examples does **not** start recruiting on its own; assign a job before recruiting has started and it can never receive responses — `assign_job` logs a warning, and the waiting methods (`get_results()`, `display_progress_bar()`) raise instead of blocking forever. Use `"global"` when you don't need task-specific qualification. +- **Job**: A running instance of a job definition assigned to an audience +- **Flow**: Continuously collect human responses in small batches without full job setup. Two kinds — **ranking flows** (Elo-style comparison) and **classify flows** (sort each datapoint into one of a fixed set of categories) +- **MRI/Benchmark**: Compare and rank AI models on leaderboards + +**Client entry points:** +- `client.job` — create job definitions (classification, comparison, locate, draw, select words, free text, ranking) +- `client.audience` — create and find audiences +- `client.flow` — continuous ranking and classify flows +- `client.mri` — model ranking insights / benchmarks +- `client.signals` — run a labeling job on a repeating schedule +- `client.context` — shorten over-long datapoint contexts against a specific question +- `client.billing` — read the current billing period's cost and remaining credit, and the outstanding balance owed (organization-level) +- `client.validation` — validation sets (`create_classification_set`, `create_compare_set`, …, `get_validation_set_by_id`, `find_validation_sets`); only needed for a flow's `validation_set_id` — jobs qualify labelers through audience examples instead + +## Job Definitions + Audiences + +Job definitions cover **classification**, **comparison**, **locate**, **draw**, **select words**, **free text**, and **ranking**. For continuous, low-latency collection without full job setup, use Flows. + +### Step-by-step workflow + +```python +from rapidata import RapidataClient + +client = RapidataClient() + +# 1. Get an audience. Default: the global pool — ready to go, zero setup. +audience = client.audience.get_audience_by_id("global") +# Curated domain pool (e.g. alignment): client.audience.get_audience_by_id("aud_MU1GZYoESyO") +# Only if you need task-specific qualification: client.audience.create_audience(name="My Evaluators") +# — but recruiting is explicit: you MUST add >=3 qualification examples (add_*_example) AND +# then call audience.start_recruiting() BEFORE assign_job. Adding examples does not start +# recruiting; a job assigned before recruiting starts can never receive responses. +# See "Custom Audiences". + +# 2. Create a job definition +job_def = client.job.create_classification_job_definition( + name="Animal Classification", + instruction="What animal is in this image?", + answer_options=["Cat", "Dog", "Bird"], + datapoints=["https://example.com/img1.jpg", "img2.jpg"], + responses_per_datapoint=10, +) + +# 3. Assign to audience (starts labeling) +job = audience.assign_job(job_def) + +# 4. Watch responses come in (opens browser on the running job) +job.view() + +# 5. Monitor and get results +progress = job.get_progress() # Non-blocking snapshot: state, completion_percentage, recruiting +job.display_progress_bar() # Or block on a live progress bar +results = job.get_results() +df = results.to_pandas() +``` + +### Classification + +Select one category from multiple options. + +```python +from rapidata import NoShuffleSetting + +job_def = client.job.create_classification_job_definition( + name="Image Classification", + instruction="What animal is in this image?", + answer_options=["Cat", "Dog", "Bird", "Other"], + datapoints=["img1.jpg", "img2.jpg"], + data_type="media", # "media" (default) or "text" — text assets are NOT translated + responses_per_datapoint=10, + contexts=["Optional text context per datapoint"], + media_contexts=[["optional_reference.jpg"]], + confidence_threshold=0.99, # Optional: confidence-based early stopping + # quorum_threshold=7, # Alternative: quorum-based early stopping (cannot use both) + settings=[NoShuffleSetting()], # Keep answer order + failure_tolerance=0.01, # Optional: fraction of datapoints allowed to fail the upload + private_metadata=[{"id": "abc"}], +) +``` + +`failure_tolerance` (available on every `create_*_job_definition`, defaults to `rapidata_config.upload.failureTolerance` = `0.0`, i.e. strict) is the fraction of datapoints allowed to fail uploading while the job definition is still created. Above the tolerance **no job definition is created at all** — see gotcha 8. + +### Comparison + +Compare two items and choose the better one. + +```python +from rapidata import AllowNeitherBothSetting + +job_def = client.job.create_compare_job_definition( + name="Image Comparison", + instruction="Which image is higher quality?", + datapoints=[["img_a1.jpg", "img_b1.jpg"], ["img_a2.jpg", "img_b2.jpg"]], + data_type="media", + responses_per_datapoint=10, + contexts=["Prompt that generated these"], + media_contexts=[["reference.jpg"]], + a_b_names=["Model A", "Model B"], + confidence_threshold=0.99, # Optional: confidence-based early stopping + # quorum_threshold=7, # Alternative: quorum-based early stopping (cannot use both) + settings=[AllowNeitherBothSetting()], # Allow "Neither" or "Both" options +) +``` + +### Ranking + +Ranking is available as a job definition or via **continuous ranking flows** (see below). + +```python +job_def = client.job.create_ranking_job_definition( + name="Image Quality Ranking", + instruction="Which image looks better?", + datapoints=[["img1.jpg", "img2.jpg", "img3.jpg"]], # outer list = independent rankings + comparison_budget_per_ranking=50, + responses_per_comparison=1, + random_comparisons_ratio=0.5, + contexts=["Optional context"], +) +job = audience.assign_job(job_def) +job.display_progress_bar() +results = job.get_results() +``` + +**Small rankings are matched exhaustively.** A ranking with **more than 10 datapoints** is matched adaptively (Elo-style) within `comparison_budget_per_ranking`, and `random_comparisons_ratio` applies as usual. A ranking with **10 or fewer datapoints** instead compares every unique pair, spreading the budget evenly across pairs (rounded down to a multiple of the pair count; every pair is compared at least once even if the budget is smaller than the pair count) — here `random_comparisons_ratio` has **no effect**. + +### Locate + +Ask labelers to tap on the points in a datapoint that match your instruction. Results are the set of tapped coordinates per datapoint. No `answer_options`, `a_b_names`, `data_type`, `confidence_threshold`, or `quorum_threshold`. + +```python +from rapidata import LocateMaxPointsSetting + +job_def = client.job.create_locate_job_definition( + name="Artifact Detection", + instruction="Tap on any visual glitches or errors in the image.", + datapoints=["img1.jpg", "img2.jpg"], + responses_per_datapoint=35, + contexts=["Optional text context"], + media_contexts=[["optional_reference.jpg"]], + settings=[LocateMaxPointsSetting(5)], + private_metadata=[{"id": "abc"}], +) +``` + +Locate qualification examples take `truths` as a `list[Box]` (image ratios 0.0–1.0) — see `add_locate_example` under Custom Audiences. + +### Draw + +Ask labelers to draw (color in) regions of an image that match your instruction. Results are the set of drawn lines per datapoint. No `answer_options`, `a_b_names`, `data_type`, `confidence_threshold`, or `quorum_threshold`. + +```python +job_def = client.job.create_draw_job_definition( + name="Artifact Drawing", + instruction="Color in any visual glitches or errors in the image.", + datapoints=["img1.jpg", "img2.jpg"], + responses_per_datapoint=35, + contexts=["Optional text context"], + media_contexts=[["optional_reference.jpg"]], + private_metadata=[{"id": "abc"}], +) +``` + +Draw qualification examples take `truths` as a `list[Box]`; labelers pass when their lines fall within a box — see `add_draw_example` under Custom Audiences. + +### Select Words + +Ask labelers to select the words from a sentence that match your instruction (e.g., words not depicted in an image). Each datapoint is paired with a `sentence` split by spaces. No `contexts`, `media_contexts`, `data_type`, `answer_options`, `a_b_names`, `confidence_threshold`, or `quorum_threshold`. + +```python +job_def = client.job.create_select_words_job_definition( + name="Image-Text Alignment", + instruction="Select the words not correctly depicted in the image.", + datapoints=["img1.jpg", "img2.jpg"], + sentences=["A cat on a red couch [No_mistakes]", "A blue car in the rain [No_mistakes]"], + responses_per_datapoint=15, + private_metadata=[{"id": "abc"}], +) +``` + +Select words qualification examples take `truths` as a `list[int]` of 0-based word indices — see `add_select_words_example` under Custom Audiences. + +### Free Text + +Ask labelers to answer your instruction with free-form text. No `answer_options`, `a_b_names`, `confidence_threshold`, or `quorum_threshold`. Note: free text answers cannot be graded against a ground truth, so audiences cannot be trained with free text qualification examples. + +```python +job_def = client.job.create_free_text_job_definition( + name="Prompt Collection", + instruction="What would you like to ask an AI? Please spell out the question.", + datapoints=["image.jpg"], + responses_per_datapoint=15, + contexts=["Optional text context"], +) +``` + +### Custom Audiences + +Create an audience trained on your specific task. + +⚠️ **Recruiting is explicit — you must start it yourself.** A custom audience recruits nobody until you (1) add **≥3 qualification examples** with `add_*_example(...)` and (2) call `audience.start_recruiting()`. Adding examples does **not** start recruiting; until `start_recruiting()` is called the audience stays in its `Created` state. A job assigned to such an audience is still created (with a warning), but it can never receive responses — `display_progress_bar()` / `get_results()` raise an error explaining that nobody graduated and nobody is being recruited. Call `start_recruiting()` **once**, after all examples are added and reviewed, before `assign_job`. If you don't need task-specific qualification, skip all of this and use the ready-to-go global pool with no setup: `client.audience.get_audience_by_id("global")`. + +**Important:** Every qualification example and its associated truth must be manually and thoroughly reviewed by a human before use. If an example has a wrong or ambiguous truth value, the qualification process will filter out good labelers who answer correctly while letting through bad labelers who happen to match the incorrect answer — completely inverting quality control. Always verify that each example has a clear, unambiguous correct answer. + +```python +from rapidata import Box, NoShuffleSetting, AllowNeitherBothSetting + +audience = client.audience.create_audience( + name="Expert Evaluators", + # target_accuracy=0.8, # Optional: fraction of qualification tasks (0-1) a labeler must get right (server default 0.75) + # min_tasks=12, # Optional: qualification tasks before the accuracy verdict is trusted (server default 10) + # max_tasks=30, # Optional: cap on admission-trial tasks before a verdict is forced (default: no cap) +) + +# Add classification examples +audience.add_classification_example( + instruction="Rate image quality", + answer_options=["Poor", "Good", "Excellent"], + datapoint="example.jpg", + truth=["Excellent"], + context="Optional context", + data_type="media", + explanation="This image is excellent due to its high resolution and sharp focus.", # Shown to labelers who answer incorrectly + settings=[NoShuffleSetting()], # Optional: match job settings so labelers qualify on the same UI +) + +# Add comparison examples +audience.add_compare_example( + instruction="Which image follows the prompt better?", + datapoint=["good.jpg", "bad.jpg"], + truth="good.jpg", + context="A cat on a chair", + data_type="media", + explanation="The first image clearly shows a cat sitting on a chair as described.", # Shown to labelers who answer incorrectly + settings=[AllowNeitherBothSetting()], # Optional: match job settings so labelers qualify on the same UI +) + +# Add locate examples — Box coordinates are image ratios (0.0–1.0) +audience.add_locate_example( + instruction="Tap on any visual glitches or errors in the image.", + datapoint="example_with_artifact.jpg", + truths=[Box(x_min=0.44, y_min=0.42, x_max=0.58, y_max=0.63)], + context="Optional context", + explanation="The artifact is in the highlighted region.", # Shown to labelers who answer incorrectly + settings=[...], # Optional: match job settings so labelers qualify on the same UI +) + +# Add draw examples — graded on whether the drawn lines fall within a box +audience.add_draw_example( + instruction="Color in any visual glitches or errors in the image.", + datapoint="example_with_artifact.jpg", + truths=[Box(x_min=0.44, y_min=0.42, x_max=0.58, y_max=0.63)], + explanation="The artifact is within the highlighted region.", # Shown to labelers who answer incorrectly + settings=[...], # Optional: match job settings so labelers qualify on the same UI +) + +# Add select words examples +audience.add_select_words_example( + instruction="Select the words not correctly depicted in the image.", + datapoint="image.jpg", + sentence="a white cat on a sunny beach [No_mistakes]", + truths=[1], # 0-based word indices to select; index 1 = "white" if the cat is black + # required_precision=1, required_completeness=1, # defaults: no wrong words, all correct words + explanation="The cat in the image is black, not white.", # Shown to labelers who answer incorrectly + settings=[...], # Optional: match job settings so labelers qualify on the same UI +) + +# Inspect the examples currently on the audience +examples_df = audience.get_examples(amount=10, page=1) + +# Once >=3 examples are added and reviewed, start recruiting. This is required and explicit: +# adding examples does not start it, and a job assigned before this can never get responses. +audience.start_recruiting() + +# Watch the funnel fill up (graduated / distilling / dropped / inactive) +metrics = audience.get_recruiting_metrics() +print(metrics.graduated, metrics.distilling) + +# Now the audience recruits against the examples; assign a job as usual. +job = audience.assign_job(job_def) +``` + +**Managing audiences (`client.audience`):** +- `client.audience.create_audience(name, filters=None, target_accuracy=None, min_tasks=None, max_tasks=None)` — create a custom audience. The last three set the admission bar for qualification: `target_accuracy` (0–1, server default `0.75`), `min_tasks` (server default `10`), `max_tasks` (no cap by default). Supplying only some of them is fine — the SDK fills in the defaults. Raises `ValueError` for an accuracy outside 0–1, `min_tasks < 1`, or `max_tasks < min_tasks`. +- `client.audience.get_audience_by_id(audience_id)` — fetch by id; pass `"global"` for the ready-to-go global audience +- `client.audience.find_audiences(name="", amount=10, page=1)` — list your audiences (newest first), optionally filtered by name + +**Audience methods:** +- `audience.start_recruiting()` — begin recruiting/onboarding annotators against the audience's qualification examples. **Required and explicit for custom audiences**: call it once, after adding ≥3 examples and before `assign_job` — adding examples never starts recruiting on its own, and a job assigned before recruiting has started can never receive responses. Calling it again on the same audience object is a no-op; a backend failure raises `RapidataError` rather than being swallowed. Returns the audience, so it chains. Not needed for the global/curated pools. +- `audience.get_recruiting_metrics()` — snapshot of the audience's recruiting funnel as a `RecruitingMetrics` (`graduated` = eligible to work now, `distilling` = still qualifying, `dropped`, `inactive`; one bucket per annotator). All zeros before `start_recruiting()` has pulled anyone in, and for curated audiences. +- `audience.assign_job(job_definition, run_after=None)` — start a job. Never blocks on funds: the job is always created, but if its estimated cost exceeds your account balance a cost warning is logged (with the estimate, your balance, and the shortfall) and the job may pause partway until you top up. A warning is also logged if the audience has no graduated annotators yet. Pass `run_after` (a `RapidataJob`, a job id string, or `None`) to queue this job behind an earlier one: the new job is created immediately in the `Queued` state and only starts once the preceding job **completes or fails**, so a single audience never splits its annotators across two jobs at once. Default `None` starts the job right away. Chain further by pointing each new job at its predecessor. + +```python +first = audience.assign_job(job_def) +second = audience.assign_job(other_job_def, run_after=first) # or run_after="job_id" +``` +- `audience.find_jobs(name="filter", amount=10, page=1)` — find assigned jobs +- `audience.update_filters([...])` — replace the recruitment filters on this audience (audience-supported filters only — see below) +- `audience.filter([filters])` — derive a filtered slice of this audience's graduated labelers without re-onboarding; multiple filters are ANDed. Returns a `RapidataFilteredAudience` (id prefixed `fau_`) that exposes only `assign_job` and `find_jobs` (plus `id`, `name`, `filters`) — no examples, `update_filters`, `delete`, or nested `.filter()`. Its id (or the object) can be passed as a leaderboard's `audience_id` +- `audience.update_name("New Name")` — rename +- `audience.get_examples(amount=10, page=1)` — list qualification examples (returns DataFrame) +- `audience.delete()` — delete the audience + +**Audience-supported filters:** `CountryFilter`, `LanguageFilter`, `AgeFilter` (`AgeGroup`), `GenderFilter` (`Gender`), and `DeviceFilter` (`DeviceType`), plus the `AndFilter`/`OrFilter`/`NotFilter` combinators (also via `&` / `|` / `~`). `UserScoreFilter`, `CampaignFilter`, and `CustomFilter` are **not supported on audiences** and raise `NotImplementedError`. + +**Looking up existing jobs (`client.job`):** `get_job_definition_by_id(id)`, `find_job_definitions(name="", amount=10, page=1)`, `get_job_by_id(id)`, `find_jobs(name="", amount=10, page=1)`. + +**Job / Job Definition methods:** +- `job_def.preview()` — open browser preview of what labelers see +- `job_def.update_dataset(datapoints=..., data_type=..., contexts=..., media_contexts=..., sentences=..., private_metadata=...)` — upload a new dataset as a new revision of the job definition; raises `FailedUploadException` if any datapoint fails +- `job_def.estimated_cost` / `job.estimated_cost` — a `CostEstimate` (`estimated_cost`, `datapoint_count`, `required_responses`) for running to completion; available on a job definition *before* assigning it. The estimate is priced shortly after creation, so the first read blocks briefly until it's ready (raises `TimeoutError` if still unavailable after a few minutes), then caches. It's an estimate, not the final bill — early stopping can lower the actual cost. +- `job_def.delete()` — delete a job definition and all its revisions +- `job.display_progress_bar(refresh_rate=5)` — blocking progress bar +- `job.get_status()` — current status string +- `job.get_progress()` — non-blocking snapshot: a `JobProgress` with `state` (same value as `get_status()`), `completion_percentage` (0–100) and `recruiting` (a `RecruitingMetrics`, or `None` for curated audiences) +- `job.get_results()` — blocks until Completed/Failed (auto-regenerates if `StaleResults`), returns `RapidataResults`. If the job needs manual review (`ManualApproval`) or runs out of funds mid-run (`SpendLimited`) — neither state completes on its own — it raises an informative error naming the state instead of blocking; top up or wait for a reviewer, then call it again. It also raises up front when the job's audience **can never produce responses** (nobody graduated *and* nobody is being recruited); an audience that is merely still distilling does not raise. +- `job.view()` — open the job's details page in the browser +- `job.delete()` — delete a running job + +## Context Management + +Datapoint contexts have a **400-character maximum**; the backend rejects longer ones. An over-long context is therefore **always** shortened against the task instruction before upload — this cannot be turned off — and a warning reports how many contexts were shortened: + +```python +job_def = client.job.create_classification_job_definition( + name="Outfit check", + instruction="Does the main character wear the right clothing?", + answer_options=["Yes", "No"], + datapoints=["scene.jpg"], + contexts=[""], +) +``` + +A context tuned to the question focuses the labeler even when it already fits the limit. Set `rapidata_config.upload.contextShortening = True` to have **every** context shortened, not just the over-long ones: + +```python +from rapidata import rapidata_config + +rapidata_config.upload.contextShortening = True +``` + +Shorten contexts directly without creating a job via `client.context`: + +```python +# Single context +short = client.context.shorten_context( + context="", + question="Does the main character wear the right clothing?", +) + +# Batch: (context, question) pairs, order preserved; sent as concurrent batches of 10 +shortened = client.context.shorten_contexts([ + (context_a, question_a), + (context_b, question_b), +]) +``` + +`ContextManager` is also importable directly: `from rapidata import ContextManager`. + +## Migration from the removed Order API + +The order-based API (`client.order`, `RapidataOrder`, `RapidataOrderManager`, `create_*_order()`) no longer exists — never generate code that uses it. Replace each `create_*_order()` with the matching `create_*_job_definition()` on `client.job`, validation sets with audience qualification examples, and `order.run()` with `audience.assign_job(job_def)`. + +## Settings + +Settings control how a task is rendered and behaves for labelers. All settings are importable from the top-level `rapidata` package. + +Most settings only apply to specific task types. If you add a setting that the job's task type does not support, the SDK logs a non-fatal warning and still sends the flag — it is never dropped and no error is raised. Ranking jobs are treated as Compare for this check. + +```python +from rapidata import ( + NoShuffleSetting, AllowNeitherBothSetting, MarkdownSetting, + MuteVideoSetting, FreeTextMinimumCharactersSetting, FreeTextMaxCharactersSetting, + SwapContextInstructionSetting, PlayPercentageVideoSetting, + OriginalLanguageOnlySetting, NoMistakeOptionSetting, DisableAutoloopSetting, + NoInstructionDisplaySetting, KeyboardNumericSetting, + LocateMaxPointsSetting, LocateMinPointsSetting, + ComparePanoramaSetting, CompareEquirectangularSetting, + ClassifyEquirectangularSetting, + CustomSetting, +) + +settings=[NoShuffleSetting()] # Keep answer options in order (use for Likert scales) +settings=[AllowNeitherBothSetting()] # Comparison: "Unsure" (neither/both) button, shown after delay_ms (default 5000) +settings=[MarkdownSetting()] # Render markdown in text +settings=[MuteVideoSetting()] # Start videos muted +settings=[FreeTextMinimumCharactersSetting(50)] # Min text length for free-text tasks (use with caution — see note below) +settings=[FreeTextMaxCharactersSetting(500)] # Max text length for free-text tasks (default 1024) (use with caution — see note below) +settings=[SwapContextInstructionSetting()] # Swap the positions of context and instruction +settings=[PlayPercentageVideoSetting(percentage=95)] # Require labelers to watch N% of video before answering (0-95) +settings=[OriginalLanguageOnlySetting()] # Do not translate the task (text assets are never translated anyway) +settings=[NoMistakeOptionSetting()] # Add a "No mistakes" button so labelers can confirm the media has no errors +settings=[DisableAutoloopSetting()] # Disable automatic looping of videos +settings=[NoInstructionDisplaySetting()] # Hide instruction on the task screen (hides the context instead if combined with SwapContextInstructionSetting) +settings=[KeyboardNumericSetting()] # Free text: open numeric keyboard on mobile +settings=[LocateMaxPointsSetting(5)] # Locate task: max number of points (default 3) +settings=[LocateMinPointsSetting(1)] # Locate task: min number of points (default 1) +settings=[ComparePanoramaSetting()] # Render comparison media in a panoramic viewer +settings=[CompareEquirectangularSetting()] # Render comparison media as equirectangular 360° view +settings=[ClassifyEquirectangularSetting()] # Render classification media as equirectangular 360° view +settings=[CustomSetting(key="my_flag", value="on")] # Rapid-level flag (target="rapids", default); target="campaign" for campaign-level +``` + +**Note on `FreeTextMinimumCharactersSetting` / `FreeTextMaxCharactersSetting`:** use these with caution. Free-text responses already pass through a reasonableness check by default, so tightening the bounds is usually unnecessary and will reject otherwise valid answers. Only set them when the question genuinely demands a specific length (e.g. a single word, or a full paragraph). + +## Key Gotchas + +1. **Watch early responses** — after assigning, call `job.view()` to open the running job in the browser and check that labelers understand the instruction as intended +2. **Use `NoShuffleSetting()` for Likert scales** — ordered answer options get shuffled by default +3. **Custom audiences need ≥3 examples AND an explicit `start_recruiting()`** — recruiting never begins on its own. Add **≥3 qualification examples** (`add_*_example(...)`), then call `audience.start_recruiting()` **once**, before `assign_job`. Skip either step and the audience recruits nobody: the job is still created (with a warning), but it can never receive responses, so `get_results()` / `display_progress_bar()` raise an error saying the audience can never produce responses. Use `client.audience.get_audience_by_id("global")` when you don't need task-specific qualification. + - **One audience = one task.** A custom audience is qualified for the specific task its examples describe. Reusing it for a different, unrelated task is a misuse — the qualification no longer applies and the quality guarantee is lost. Reuse it only for repeated/scheduled runs of the *same* task; spin up a new audience for a new task. +4. **25-second time limit, 250-character instructions** — labelers have ~25 seconds per task; keep instructions concise. Instructions are capped at **250 characters** — a longer one raises `ValueError: instruction is characters; maximum is 250` when the job definition or qualification example is created +5. **Responses may exceed `responses_per_datapoint`** — concurrent labelers can cause slight overflow +6. **Two early stopping strategies, mutually exclusive** — `confidence_threshold` (statistical, weighted by labeler trust scores) or `quorum_threshold` (stops when N responses agree); cannot use both at once +7. **Early stopping only for unambiguous tasks** — both strategies work best when there's a clear correct answer +8. **Failed uploads abort job-definition creation** — job definitions are created atomically: if more than `failure_tolerance` of the datapoints fail to upload, **no job definition is created** (`e.job_definition` is `None`) and at least one datapoint must always succeed. Fix the failing datapoints and call `e.retry()`, which re-uploads only the failed ones into the *same* dataset and finishes creating the definition; it raises `FailedUploadException` again if failures remain, so it can be looped. Within tolerance, the definition is created and a warning reports how many failed. Inspect failures via `e.failures_by_reason`, `e.failures_by_stage` (grouped by remote-URL ingestion stage — only `internal` is a Rapidata-side fault), and each `FailedUpload`'s `stage` / `http_status` +9. **Preview link printed on job creation** — creating a job definition prints a dashboard preview link; suppress prints with `rapidata_config.logging.silent_mode = True` +10. **Context length limit is 400 characters** — the backend rejects contexts longer than 400 characters, so an over-long context is **always** shortened against the task instruction before upload (not optional; a warning reports how many were shortened). Set `rapidata_config.upload.contextShortening = True` to shorten *every* context, or use `client.context.shorten_context()` / `client.context.shorten_contexts()` to shorten manually. +11. **Jobs can pause for manual review or funds** — `assign_job` always creates the job, but if its estimated cost exceeds your account balance it logs a cost warning and the job may pause until you top up. A job can also enter manual review (`ManualApproval`) or become spend-limited (`SpendLimited`) mid-run; since neither state completes on its own, `get_results()` raises an informative error naming the state instead of blocking — top up or wait for a reviewer, then retry. +12. **Text assets are NOT translated** — datapoints uploaded with `data_type="text"` are shown to labelers exactly as supplied, in their original language, whatever language the labeler views the task in. Only the surrounding task UI may be translated; `OriginalLanguageOnlySetting` does not change this. If labelers must read the text, target speakers of its language with `LanguageFilter` / `CountryFilter`, or supply the text already in the labelers' language. + +## Flows (Continuous Response Collection) + +Flows continuously collect human responses in small batches without full job/audience setup. There are two kinds: **ranking flows** (Elo-style comparison, below) and **classify flows** (sort each datapoint into a category, further below). + +The flow classes are importable from the top level: `from rapidata import RapidataFlow, RapidataRankingFlow, RapidataClassifyFlow`. `RapidataFlow` is a shared base class (only `get_flow_items` and `delete`); the concrete `RapidataRankingFlow` and `RapidataClassifyFlow` subclasses carry the kind-specific `create_new_flow_batch` (and, for ranking, `update_config`). Every flow carries a `flow_type` (`"ranking"` or `"simple"` — classify flows are backed by "simple" flows). `create_ranking_flow` returns a `RapidataRankingFlow` and `create_classify_flow` returns a `RapidataClassifyFlow`, but `get_flow_by_id` and `find_flows` return the union `RapidataRankingFlow | RapidataClassifyFlow` — narrow with `isinstance(flow, RapidataClassifyFlow)` before calling kind-specific batch methods. + +### Ranking Flows + +Lightweight continuous ranking: + +```python +# Create flow +flow = client.flow.create_ranking_flow( + name="Image Quality Ranking", + instruction="Which image looks better?", + max_response_threshold=100, # Target responses per flow item (default 100) + min_response_threshold=50, # Minimum acceptable responses (defaults to max_response_threshold); item is Incomplete if TTL expires below this + # validation_set_id="...", # Optional: run a validation set alongside the flow + # settings=[NoShuffleSetting()], # Optional: flow-wide settings +) + +# Preheat for low-latency responses (call ~5 minutes before time-sensitive batches) +client.flow.preheat() + +# Add items to rank +flow_item = flow.create_new_flow_batch( + datapoints=["img1.jpg", "img2.jpg", "img3.jpg"], + context="Generated by Model X", # A single batch-level context shared across all comparisons + # context_assets=["reference.jpg"], # 1–10 image/video/audio paths/URLs shown alongside instruction (flat list) + time_to_live=300, # Seconds until expiry (45–3600; ranking flows default to 4 minutes when omitted) + # data_type="media", private_metadata=[...], accept_failed_uploads=False, # also accepted +) +# Ranking batches take a batch-level context / context_assets only — no per-datapoint +# contexts (those are a classify-batch feature). + +# Get results (flow items have their own result shape, not RapidataResults) +results = flow_item.get_results() # Blocks until complete; ranking items return FlowItemResult(datapoints={asset key: elo}, total_votes) +status = flow_item.get_status() # Non-blocking check (Pending, Running, Completed, Failed, Stopping, Stopped, or Incomplete) +matrix = flow_item.get_win_loss_matrix() # Pandas DataFrame (blocks until complete); ranking flow items only — raises ValueError otherwise +count = flow_item.get_response_count() + +# Tune a ranking flow after creation (only RapidataRankingFlow has update_config) +flow.update_config( + instruction="New instruction", + starting_elo=1000, + min_responses=40, + max_responses=120, +) + +# Manage flows +all_flows = client.flow.find_flows(name="", amount=10, page=1) +flow = client.flow.get_flow_by_id("flow_id") +items = flow.get_flow_items(amount=10, page=1) # newest first, 10 per page by default +flow.delete() +``` + +Note: `RapidataFlowItem` does **not** have `display_progress_bar()` — poll with `get_status()` or just call `get_results()` to block. + +Using flows as the preference signal for DPO/RLHF or best-of-N? Read [flows-for-preference-data.md](flows-for-preference-data.md) first: TTL, not the max threshold, usually sets a flow item's turnaround. + +### Classify Flows + +Continuously sort each datapoint in a batch into one of the flow's categories: + +```python +# Create flow +flow = client.flow.create_classify_flow( + name="Text Detection", + instruction="Does this image contain text?", # question shown with every datapoint + categories=[("Yes, clearly readable", "yes"), ("No", "no")], # 2–8 options; a plain + # string is shown and returned as-is, a (label, value) tuple shows label but returns value + max_responses_per_datapoint=15, # default 15; accepted responses that close an image + # (collection for that image stops once reached) + min_responses_per_datapoint=10, # default 10, must be >= 1; average responses per image an + # item needs (once it ends by its time to live) to be Completed rather than Incomplete. + # Requires max_responses_per_datapoint >= min_responses_per_datapoint. + # validation_set_id="...", # Optional + # settings=[...], # Optional: flow-wide settings +) +# Time-to-live is set per batch only (create_new_flow_batch below), not on the flow. + +# Add a batch of datapoints to classify +flow_item = flow.create_new_flow_batch( + datapoints=["img1.jpg", "img2.jpg"], + contexts=["Text context for img1", "Text context for img2"], # per-datapoint; exactly one per datapoint + # context_assets=[["ref1.jpg"], ["ref2a.jpg", "ref2b.jpg"]], # Optional: one list of asset paths/URLs per + # datapoint (list[list[str]] — each entry is a list even for a single asset); length must match datapoints + # time_to_live=300, # Optional: seconds this batch may run (45–3600); set per batch + # data_type="media", private_metadata=[...], accept_failed_uploads=False, # also accepted +) +# Classify batches take per-datapoint contexts / context_assets only — no batch-level context. + +# Get results — classify flow items return ClassifyFlowItemResult +result = flow_item.get_results() # Blocks until complete +result.total_responses # int +for key, dp in result.datapoints.items(): # key = source URL, else original filename + dp.majority_value # category value chosen most often (None on a tie) + dp.distribution # {category value: number of responses} for EVERY category defined in the + # flow, in the flow's category order, with 0 for categories nobody chose (any unexpected + # backend values are appended after the blueprint categories). Filling in the zeros costs one + # extra API call to fetch the flow's categories when results are computed. + dp.response_count # responses collected for this datapoint + +count = flow_item.get_response_count() # total responses (blocks until complete) +``` + +Result classes (frozen dataclasses, importable from `rapidata`): `ClassifyFlowItemResult` (`datapoints`, `total_responses`) and `ClassifyDatapointResult` (`majority_value`, `distribution`, `response_count`), alongside the existing `FlowItemResult`. + +## Model Ranking Insights (MRI / Benchmarks) + +Compare and rank AI models on leaderboards. Supports images, videos, audio, and text. + +```python +# Create benchmark +benchmark = client.mri.create_new_benchmark( + name="AI Art Competition", + prompts=["A serene mountain landscape", "A futuristic city"], + # identifiers=[...], # Optional: stable ids for each prompt + # prompt_assets=[["ref1.jpg"], ["ref2.jpg"]], # Optional: one list of reference-media + # # URLs/paths per prompt (list[list[str]]; several entries in a + # # list register as one multi-asset, or None for no asset) + # tags=[...], # Optional: per-prompt tags — str, Tag(value, category=...), or a mix + # origins=[...], # Optional: per-prompt Origin(source) or plain source string +) + +# Add prompts later if needed (one or many, matched up by index) +benchmark.add_prompts( + prompts=["A quiet lake at dawn"], + # identifiers=["dawn_lake"], # Optional: stable id per prompt + # prompt_assets=[["ref.jpg"]], # Optional: one list of reference-media URLs/paths per prompt + # # (list[list[str]]; passing a bare str per prompt still works + # # but is deprecated). Same shape as benchmark.prompt_assets reads back. + # tags=[["landscape"]], # Optional: list of tag lists, one per prompt (str and/or Tag) + # origins=["coco"], # Optional: where each prompt came from (Origin or str) +) + +# Tags carry an optional category; bare strings become Tag(value, category=None) +from rapidata import Tag, Origin + +benchmark.add_prompts( + identifiers=["garage_car"], + prompts=["A car in a garage"], + tags=[[Tag("vehicle", category="object"), "indoor"]], + origins=[Origin("coco")], +) + +# Re-tag / set the origin of an already-registered prompt (a field left None stays unchanged) +benchmark.update_prompt("garage_car", tags=["abstract", "surreal"], origin="wikiart") + +print(benchmark.tags) # Values only, aligned by index with prompts (categories dropped) +print(benchmark.structured_tags) # list[list[Tag]] — preferred, keeps categories +print(benchmark.origins) # list[Origin | None] + +# Create leaderboard +leaderboard = benchmark.create_leaderboard( + name="Realism", + instruction="Which image is more realistic?", + show_prompt=False, + show_prompt_asset=False, + inverse_ranking=False, + # level_of_detail="high", # "debug" (20) | "low" (2000) | "medium" (4000) | "high" (8000) + # # | "very high" (16000), or a positive int response budget + # min_responses_per_matchup=5, + # audience_id="...", # Optional: id string, RapidataAudience, or RapidataFilteredAudience + # settings=[...], + # included_tags=["outdoor"], # Optional: only collect matchups for prompts carrying one of these tags + # excluded_tags=["nsfw"], # Optional: skip prompts carrying one of these tags (always wins) + # vote_aggregation=VoteAggregation.MAJORITY_VOTE, # default; or VoteAggregation.ALL_VOTES — how a matchup's + # # individual responses are aggregated (import VoteAggregation from rapidata) + # skip_initial_run=False, # Optional: when True, skip the initial run that evaluates the models already + # # in the benchmark against each other — you start with no responses/standings. + # # Later add_model still compares against the whole field; create-only (not readable back). +) + +# Evaluate a model (creates participant, uploads media, and submits in one step) +benchmark.evaluate_model( + name="MyModel_v2", + media=["mountain.png", "city.png"], + prompts=["A serene mountain landscape", "A futuristic city"], + data_type="media", # "media" (default) or "text" +) + +# Or add a model without submitting (for more control) +participant = benchmark.add_model( + name="MyModel_v3", + media=["mountain_v3.png", "city_v3.png"], + prompts=["A serene mountain landscape", "A futuristic city"], + data_type="media", +) + +# Upload additional media to the same participant +uploaded, failed = participant.upload_media( + assets=["mountain_v3_extra.png"], + identifiers=["A serene mountain landscape"], + data_type="media", +) +# Returns (identifiers uploaded, list[FailedUpload[SampleUpload]]). Each FailedUpload +# carries the media/identifier pair (.item, a SampleUpload), the reason, and a trace id, +# so a failed pair can be re-submitted directly. Raises ValueError if assets and +# identifiers differ in length. + +# Recover a partial upload (server truth — works for any participant, incl. ones from +# benchmark.participants). add_model already runs this sweep automatically on failure. +missing = participant.missing_counts(identifiers) # Counter[identifier -> samples still short]; empty == done +uploaded, still_failed = participant.retry_missing( # re-sends assets for short identifiers + assets=["mountain_v3_extra.png"], + identifiers=["A serene mountain landscape"], + data_type="media", +) +# Safe to call repeatedly (the backend rejects samples already held, so no duplication). + +# Submit individually or all at once +participant.run() # Submit one participant +benchmark.run() # Submit all unsubmitted participants (status CREATED or + # SUBMITTABLE) in a single batch request. This evaluates them + # symmetrically as one run — each model is compared against every + # other and against the benchmark's already-submitted field — + # rather than as separate per-participant runs. Chunked to <=100 + # participants per request. +# Both run() methods are advisory about under-filled prompts: if the benchmark has a +# minimum-samples-per-prompt gate set (see below) and a submitted participant filled a +# prompt with fewer than the required samples, they log a warning listing each shortfall +# as 'identifier' (asset_count/required), e.g. model-0: 'cat' (2/4). Submission still +# completes and participants are marked SUBMITTED regardless of the warning. + +participant.disable() # Exclude from evaluation and standings (reversible) +participant.enable() # Re-enable a previously disabled participant +participant.get_elo() # Aggregated Elo across all leaderboards (None if not yet computed) +participant.delete() # Delete participant and its uploaded media (cannot be undone) + +# Update participant metadata — adding a model and pricing it are separate calls +participant.rename("New Name") # Rename the participant +participant.set_price(0.04, unit="image") # Set the model's list price in USD per unit ("image" | + # "video_second" | "million_tokens"); both args required. + # ValueError on a non-positive or non-finite price, or an unknown unit +participant.clear_price() # Remove the price (model drops off the cost chart) +participant.price # float | None — USD per price_unit +participant.price_unit # str | None — "image" | "video_second" | "million_tokens" +# Pricing models — a priced participant is plotted on the benchmark's "Score vs. cost" chart. +# After adding a model, set its price if the vendor publishes a list price, and state the unit +# (per image, per second of video, per million tokens). If you don't know the price with +# confidence, leave it unset and tell the user instead of guessing. Unpriced models are hidden +# from the cost chart, and only models quoted in the benchmark's majority unit are plotted. + +# List participants and their status +for p in benchmark.participants: + print(p.name, p.status, p.price, p.price_unit) + +# Prompts — original language and English translation (aligned by index) +print(benchmark.prompts) # As originally provided +print(benchmark.english_prompts) # Server-side English translations, aligned by index +print(benchmark.identifiers, benchmark.prompt_assets) + +# Get results +standings = leaderboard.get_standings() # Pandas DataFrame for one leaderboard +overall = benchmark.get_overall_standings(tags=None, leaderboard_ids=None) # Aggregated ELO across all leaderboards +matrix_lb = leaderboard.get_win_loss_matrix() # Pairwise wins/losses for one leaderboard +matrix_bm = benchmark.get_win_loss_matrix( # Pairwise wins/losses across leaderboards + tags=None, participant_ids=None, leaderboard_ids=None, use_weighted_scoring=None, +) + +# Filter any of the read methods above by voter demographics. All four benchmark reads +# (get_overall_standings, get_win_loss_matrix, get_demographics, get_standings_breakdown) +# and both leaderboard reads (get_standings, get_win_loss_matrix) accept the same optional +# keyword filters, restricting results to votes from matching voters: +from rapidata import Gender, AgeGroup + +overall_filtered = benchmark.get_overall_standings( + country=["US", "GB"], # ISO-2 codes (observed) + language=["en"], # (observed) + gender=[Gender.FEMALE], # list[Gender] (estimated/inferred) + age_bucket=[AgeGroup.BETWEEN_18_29], # list[AgeGroup] (estimated/inferred) + occupation=["Engineer"], # plain strings (estimated/inferred) + run_id="run_...", # restrict to a single evaluation run +) + +# Demographic composition of the benchmark's voters: one row per (dimension, bucket), +# columns dimension / value / votes / share (shares within a dimension sum to 1; each +# dimension includes an "unknown" bucket). dimension holds BenchmarkDemographicDimension values. +demographics = benchmark.get_demographics(tags=None, leaderboard_ids=None) + +# Standings split by a demographic dimension of the voters (dimension is required, first arg): +# one row per (segment, model), columns segment / segment_votes / name / wins / +# total_matches / score. Segments include an "unknown" bucket. +from rapidata import BenchmarkDemographicDimension + +breakdown = benchmark.get_standings_breakdown( + dimension=BenchmarkDemographicDimension.COUNTRY, # AGEBUCKET | GENDER | OCCUPATION | COUNTRY | LANGUAGE +) + +# Access the jobs that ran for a leaderboard (one per run, most recent first) +for job in leaderboard.jobs: + job_results = job.get_results() + +# Update leaderboard config live — all mutation goes through update() (properties are +# read-only). Only the arguments you pass are changed; omitted ones keep their stored value. +leaderboard.update( + name="Realism (Updated)", # non-empty string + level_of_detail="very high", # named level or a positive int budget (e.g. 5000) + min_responses_per_matchup=7, # int >= 3 (bool rejected); takes effect for future evaluations + vote_aggregation=VoteAggregation.MAJORITY_VOTE, # re-counts already-collected responses (no re-evaluation) +) +# Changing level_of_detail / min_responses_per_matchup only affects future evaluations; +# already-computed standings are not recomputed. + +# Reading back the config (all read-only properties) +print(leaderboard.level_of_detail) # A named level only on an exact budget match, otherwise "custom" +print(leaderboard.response_budget) # The exact budget behind it, e.g. 5000 +print(leaderboard.vote_aggregation) # A VoteAggregation member (lazily fetched for leaderboards read from a listing) +print(leaderboard.included_tags, leaderboard.excluded_tags) # Fixed at creation — create a new leaderboard to re-scope + +# Open in browser +benchmark.view() +leaderboard.view() + +# Find existing benchmarks +benchmarks = client.mri.find_benchmarks(name="AI Art", amount=10, page=1) +benchmark = client.mri.get_benchmark_by_id("benchmark_id") + +# Patch the benchmark's configuration — only the arguments you pass are changed; +# omitted ones keep their stored value. +benchmark.update( + min_assets_per_prompt=4, # Minimum samples (assets) each participant should fill per + # prompt. int >= 2 (bool rejected; a smaller value or non-int + # raises ValueError). This is an advisory gate: participant.run() + # and benchmark.run() warn (but do not reject) when a submitted + # participant filled a prompt below this count. +) +``` + +## Signals (Scheduled Labeling) + +A signal runs the same labeling job on a repeating schedule: bind a job definition to an audience and an interval, and Rapidata creates a new job on every tick. + +```python +from rapidata import RapidataClient + +client = RapidataClient() + +audience = client.audience.get_audience_by_id("aud_MU1GZYoESyO") + +job_def = client.job.create_compare_job_definition( + name="Prompt Alignment Job", + instruction="Which image follows the prompt more accurately?", + datapoints=[["flux_book.jpg", "mj_book.jpg"]], + contexts=["A small blue book sitting on a large red book."], +) + +signal = client.signals.create_signal( + name="Daily prompt alignment", + audience=audience, # also accepts id string + job_definition=job_def, # also accepts id string + interval_hours=24, # float + # description="...", # Optional + # revision_number=..., # Optional: pin a specific job-definition revision + # is_public=True, # Optional: let others in your org read the signal +) + +# Inspect jobs created by the signal +for job in signal.get_jobs(page_size=10): + print(job, job.get_status()) + +# Fire one job immediately instead of waiting for the schedule +signal.trigger() +job = signal.wait_for_next_job(timeout=600) # blocks until the job is created +print(job.get_results()) + +# Manage the signal +signal.pause() +signal.resume() +signal.update(name="Hourly prompt alignment", interval_hours=1) # also description= +signal.delete() + +# Look signals up later +signal = client.signals.get_signal_by_id("signal_id") +signals = client.signals.find_signals(name="alignment", amount=10, page=1) +``` + +**Signal properties:** `id`, `name`, `description`, `audience_id`, `job_definition_id`, `revision_number`, `interval_hours`, `next_run_at`, `last_run_at`, `is_paused`, `is_public`, `created_at`. + +Note: `signal.pause()` only affects the scheduler — manual `trigger()` calls still fire on a paused signal. + +## Billing + +`client.billing` reads how much the current billing period has cost so far and how much credit is left. Billing is settled per **organization**, so the figures cover everything the organization spent — not only the jobs this client created. + +```python +from rapidata import RapidataClient, BillingPeriod, RapidataBillingManager + +client = RapidataClient() + +# The billing period currently accruing cost. Raises RapidataError (status 404) +# if the organization has no active period (one only opens once there is something to bill). +period = client.billing.get_current_billing_period() # -> BillingPeriod +``` + +`BillingPeriod` is a frozen dataclass; all amounts are US dollars rounded to the cent, and each read is a snapshot (fetch again for an up-to-date figure): + +| Field | Description | +|---|---| +| `id` | The billing period's id. | +| `start_date` / `end_date` | When the period starts and ends (`datetime`). | +| `status` | `"Open"` while still accruing cost; otherwise one of `"Invoiced"`, `"Void"`, `"Reconciling"`, `"PendingReview"`, `"Closed"`. | +| `outstanding_cost` | Net cost accrued so far (`gross_cost` minus `discount`) — what the period would be invoiced for today. | +| `gross_cost` | Cost accrued so far, before discounts. | +| `discount` | Discounts applied to the period so far. | +| `response_count` | Number of billable responses collected in the period (`int`). | +| `credits` | Prepaid credit still available, or `None` when the organization is billed for usage rather than from a prepaid balance. An organization-level balance that carries across periods. | +| `effective_limit` | The most the organization may spend this period, or `None` when it spends without a cap. On a prepaid plan this is the total credit granted, and `credits` is what remains of it. | + +```python +# Total the organization currently owes, in US dollars rounded to the cent (0.0 when nothing is owed). +# Covers finalized-but-unpaid invoices plus the settled cost of ended periods not yet invoiced; +# excludes the current, still-accruing period. Already net of vouchers and discounts. +owed = client.billing.get_outstanding_balance() # -> float +``` + +## Additional Resources + +- For complete API reference, all parameters, filters, results format, error handling, flows, and MRI: see [reference.md](reference.md) (if this file was installed on its own via `python -m rapidata skill --install`, fetch https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/reference.md) +- For full end-to-end code examples and common patterns: see [examples.md](examples.md) (standalone copy: https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/examples.md) +- For using flows to collect DPO/RLHF preference pairs or run best-of-N, and designing comparisons that train well: see [flows-for-preference-data.md](flows-for-preference-data.md) (standalone copy: https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/flows-for-preference-data.md) diff --git a/tests/conftest.py b/tests/conftest.py index 905fbabfc6..e8609ad4d3 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -13,9 +13,42 @@ """ import os +import sys +from pathlib import Path os.environ["RAPIDATA_DISABLE_OTLP"] = "1" from rapidata.rapidata_client.config import rapidata_config # noqa: E402 rapidata_config.logging.enable_otlp = False + + +import pytest # noqa: E402 + +from rapidata import _agent_hint # noqa: E402 + +_AGENT_VARS = ( + *_agent_hint._AGENT_ENV_VARS, + *_agent_hint._SESSION_ENV_VARS, + "RAPIDATA_AGENT_HINT", + "CLAUDE_CONFIG_DIR", +) + + +@pytest.fixture +def agent_sandbox(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> Path: + """An empty home and project with no agent env, no hint state and no network.""" + for var in _AGENT_VARS: + monkeypatch.delenv(var, raising=False) + home, project = tmp_path / "home", tmp_path / "project" + home.mkdir() + project.mkdir() + monkeypatch.setenv("HOME", str(home)) + monkeypatch.chdir(project) + monkeypatch.setattr( + _agent_hint, "STATE_FILE", home / ".config/rapidata/agent-state.json" + ) + monkeypatch.setattr(_agent_hint, "FALLBACK_STATE_FILE", tmp_path / "tmp-state.json") + monkeypatch.setattr(_agent_hint, "_fetch_live_digest", lambda: None) + monkeypatch.setattr(sys, "orig_argv", ["python", "-c", "import rapidata"]) + return project diff --git a/tests/test_agent_hint.py b/tests/test_agent_hint.py index 36df9db86d..231cd41d67 100644 --- a/tests/test_agent_hint.py +++ b/tests/test_agent_hint.py @@ -1,8 +1,10 @@ from __future__ import annotations +import json import os import subprocess import sys +import time from pathlib import Path import pytest @@ -13,17 +15,40 @@ agent_hint, detected_coding_agent, mark_skill_read, + record_live_skill, running_under_coding_agent, + skill_digest, + stamp_skill, ) +SKILL = "---\nname: rapidata\ndescription: guide\n---\n# Rapidata\n" + @pytest.fixture(autouse=True) -def clean_env(monkeypatch: pytest.MonkeyPatch, tmp_path: Path): - for var in (*_agent_hint._AGENT_ENV_VARS, "RAPIDATA_AGENT_HINT"): - monkeypatch.delenv(var, raising=False) - monkeypatch.chdir(tmp_path) - monkeypatch.setattr(_agent_hint, "SKILL_READ_MARKER", tmp_path / "skill-read") - monkeypatch.setattr(sys, "orig_argv", ["python", "-c", "import rapidata"]) +def sandbox(agent_sandbox: Path) -> Path: + return agent_sandbox + + +def _agent(monkeypatch: pytest.MonkeyPatch, session: str | None = "s1") -> None: + monkeypatch.setenv("CLAUDECODE", "1") + if session: + monkeypatch.setenv("CLAUDE_CODE_SESSION_ID", session) + else: + monkeypatch.delenv("CLAUDE_CODE_SESSION_ID", raising=False) + + +def _install( + base: Path, content: str, rel: str = ".claude/skills/rapidata/SKILL.md" +) -> Path: + target = base / rel + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(content) + return target + + +def _after(monkeypatch: pytest.MonkeyPatch, seconds: float) -> None: + later = time.time() + seconds + monkeypatch.setattr(_agent_hint.time, "time", lambda: later) def test_not_detected_in_a_plain_shell(): @@ -39,11 +64,20 @@ def test_detected_by_agent_env_var(monkeypatch: pytest.MonkeyPatch, var: str): @pytest.mark.parametrize("value", ["0", "false", "NO"]) def test_override_silences(monkeypatch: pytest.MonkeyPatch, value: str): - monkeypatch.setenv("CLAUDECODE", "1") + _agent(monkeypatch) monkeypatch.setenv("RAPIDATA_AGENT_HINT", value) assert agent_hint() is None +def test_override_cannot_force_the_hint_on(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("RAPIDATA_AGENT_HINT", "1") + assert agent_hint() is None + + +def test_hint_does_not_advertise_the_off_switch(): + assert "RAPIDATA_AGENT_HINT" not in AGENT_HINT + + @pytest.mark.parametrize( ("env", "expected"), [ @@ -69,31 +103,158 @@ def test_detection_ignores_hint_override(monkeypatch: pytest.MonkeyPatch): assert running_under_coding_agent() is False -def test_silent_once_the_guide_was_read(monkeypatch: pytest.MonkeyPatch): - monkeypatch.setenv("CLAUDECODE", "1") +def test_silent_once_this_session_read_the_guide(monkeypatch: pytest.MonkeyPatch): + _agent(monkeypatch, "s1") mark_skill_read() assert agent_hint() is None -def test_silent_when_the_skill_is_installed( +def test_a_new_session_on_the_same_machine_is_hinted_again( + monkeypatch: pytest.MonkeyPatch, +): + _agent(monkeypatch, "s1") + mark_skill_read() + _agent(monkeypatch, "s2") + assert agent_hint() == AGENT_HINT + + +def test_codex_sessions_are_keyed_by_thread_id(monkeypatch: pytest.MonkeyPatch): + monkeypatch.setenv("CODEX_THREAD_ID", "t1") + mark_skill_read() + assert agent_hint() is None + monkeypatch.setenv("CODEX_THREAD_ID", "t2") + assert agent_hint() == AGENT_HINT + + +def test_read_without_a_session_id_expires(monkeypatch: pytest.MonkeyPatch): + _agent(monkeypatch, session=None) + mark_skill_read() + assert agent_hint() is None + _after(monkeypatch, _agent_hint.ANON_READ_TTL + 1) + assert agent_hint() == AGENT_HINT + + +def test_read_is_kept_when_home_is_read_only( monkeypatch: pytest.MonkeyPatch, tmp_path: Path ): - monkeypatch.setenv("CLAUDECODE", "1") - target = tmp_path / _agent_hint.SKILL_INSTALL_PATHS["claude"] - target.parent.mkdir(parents=True) - target.write_text("guide") + _agent(monkeypatch) + blocker = tmp_path / "not-a-dir" + blocker.write_text("") + monkeypatch.setattr(_agent_hint, "STATE_FILE", blocker / "agent-state.json") + mark_skill_read() + assert _agent_hint.FALLBACK_STATE_FILE.is_file() assert agent_hint() is None def test_silent_while_running_the_skill_cli(monkeypatch: pytest.MonkeyPatch): - monkeypatch.setenv("CLAUDECODE", "1") + _agent(monkeypatch) monkeypatch.setattr(sys, "orig_argv", ["python", "-m", "rapidata", "skill"]) assert agent_hint() is None +def test_silent_when_the_plugin_is_installed( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +): + _agent(monkeypatch) + config = tmp_path / "claude-config" + (config / "plugins").mkdir(parents=True) + (config / "plugins/installed_plugins.json").write_text( + json.dumps({"plugins": {"rapidata-sdk-plugin@rapidata-sdk-marketplace": []}}) + ) + monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(config)) + assert agent_hint() is None + + +def test_an_unrelated_agents_md_does_not_count_as_installed( + monkeypatch: pytest.MonkeyPatch, sandbox: Path +): + _agent(monkeypatch) + _install(sandbox, "# Our repo conventions\n", "AGENTS.md") + assert agent_hint() == AGENT_HINT + + +@pytest.mark.parametrize("where", ["project", "home"]) +def test_silent_when_an_installed_copy_is_current( + monkeypatch: pytest.MonkeyPatch, sandbox: Path, where: str +): + _agent(monkeypatch) + _install(sandbox if where == "project" else Path.home(), stamp_skill(SKILL)) + record_live_skill(SKILL) + assert agent_hint() is None + + +def test_stale_installed_copy_asks_for_a_reinstall( + monkeypatch: pytest.MonkeyPatch, sandbox: Path +): + _agent(monkeypatch) + path = _install(sandbox, stamp_skill(SKILL)) + record_live_skill(SKILL + "new gotcha\n") + hint = agent_hint() + assert hint is not None and str(path) in hint + assert hint.endswith("python -m rapidata skill --install") + + +def test_stale_user_level_copy_names_its_dir(monkeypatch: pytest.MonkeyPatch): + _agent(monkeypatch) + _install(Path.home(), stamp_skill(SKILL), ".codex/skills/rapidata/SKILL.md") + record_live_skill(SKILL + "new\n") + hint = agent_hint() + assert hint is not None + assert hint.endswith(f"--install --agent codex --dir {Path.home()}") + + +def test_unstamped_copy_from_an_older_install_is_compared_verbatim( + monkeypatch: pytest.MonkeyPatch, sandbox: Path +): + _agent(monkeypatch) + _install(sandbox, SKILL) + record_live_skill(SKILL) + assert agent_hint() is None + record_live_skill(SKILL + "new\n") + assert agent_hint() is not None + + +def test_freshness_is_checked_at_most_once_a_day( + monkeypatch: pytest.MonkeyPatch, sandbox: Path +): + _agent(monkeypatch) + _install(sandbox, stamp_skill(SKILL)) + calls: list[int] = [] + + def fetch() -> str: + calls.append(1) + return skill_digest(SKILL) + + monkeypatch.setattr(_agent_hint, "_fetch_live_digest", fetch) + assert agent_hint() is None + assert agent_hint() is None + assert len(calls) == 1 + _after(monkeypatch, _agent_hint.FRESHNESS_TTL + 1) + agent_hint() + assert len(calls) == 2 + + +def test_offline_freshness_check_stays_silent( + monkeypatch: pytest.MonkeyPatch, sandbox: Path +): + _agent(monkeypatch) + _install(sandbox, stamp_skill(SKILL)) + assert agent_hint() is None + + +def test_stamp_goes_after_the_front_matter(): + stamped = stamp_skill(SKILL) + assert stamped.startswith("---\nname: rapidata\n") + assert f"sha256={skill_digest(SKILL)}" in stamped.split("---\n")[2] + + def test_import_prints_the_hint_to_stderr(tmp_path: Path): - env = {k: v for k, v in os.environ.items() if k not in _agent_hint._AGENT_ENV_VARS} - env.update(CLAUDECODE="1", HOME=str(tmp_path)) + env = { + k: v + for k, v in os.environ.items() + if k not in (*_agent_hint._AGENT_ENV_VARS, *_agent_hint._SESSION_ENV_VARS) + } + env.update(CLAUDECODE="1", HOME=str(tmp_path), TMPDIR=str(tmp_path)) result = subprocess.run( [sys.executable, "-c", "import rapidata"], env=env, diff --git a/tests/test_main.py b/tests/test_main.py index bb84281722..7125e4988d 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -7,19 +7,28 @@ from rapidata import __main__ as cli from rapidata import _agent_hint +from rapidata._agent_hint import skill_digest + +SKILL = "---\nname: rapidata\n---\nguide" @pytest.fixture(autouse=True) -def marker(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> Path: - path = tmp_path / "skill-read" - monkeypatch.setattr(_agent_hint, "SKILL_READ_MARKER", path) - return path +def sandbox(agent_sandbox: Path) -> Path: + return agent_sandbox @pytest.fixture def skill(monkeypatch: pytest.MonkeyPatch) -> str: - monkeypatch.setattr(cli, "fetch_skill", lambda: "---\nname: rapidata\n---\nguide") - return "---\nname: rapidata\n---\nguide" + monkeypatch.setattr(cli, "fetch_skill", lambda: SKILL) + return SKILL + + +@pytest.fixture +def offline(monkeypatch: pytest.MonkeyPatch) -> None: + def boom() -> str: + raise requests.ConnectionError("offline") + + monkeypatch.setattr(cli, "fetch_skill", boom) def test_skill_prints_the_guide(skill: str, capsys: pytest.CaptureFixture[str]): @@ -27,9 +36,14 @@ def test_skill_prints_the_guide(skill: str, capsys: pytest.CaptureFixture[str]): assert capsys.readouterr().out.strip() == skill -def test_skill_install_writes_claude_path_by_default(skill: str, tmp_path: Path): +def test_skill_install_writes_a_stamped_copy_to_the_claude_path( + skill: str, tmp_path: Path +): assert cli.main(["skill", "--install", "--dir", str(tmp_path)]) == 0 - assert (tmp_path / ".claude/skills/rapidata/SKILL.md").read_text() == skill + installed = (tmp_path / ".claude/skills/rapidata/SKILL.md").read_text() + assert installed.startswith("---\nname: rapidata\n---\n") + assert f"sha256={skill_digest(skill)}" in installed + assert installed.endswith("guide") def test_skill_install_honours_agent(skill: str, tmp_path: Path): @@ -37,16 +51,24 @@ def test_skill_install_honours_agent(skill: str, tmp_path: Path): cli.main(["skill", "--install", "--agent", "generic", "--dir", str(tmp_path)]) == 0 ) - assert (tmp_path / "AGENTS.md").read_text() == skill + assert (tmp_path / "AGENTS.md").read_text().endswith("guide") -def test_fetch_failure_points_at_the_online_copy( - monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +def test_offline_falls_back_to_the_bundled_copy( + offline: None, capsys: pytest.CaptureFixture[str] ): - def boom() -> str: - raise requests.ConnectionError("offline") + assert cli.main(["skill"]) == 0 + out = capsys.readouterr() + assert out.out.startswith("---\nname: rapidata\n") + assert "bundled" in out.err - monkeypatch.setattr(cli, "fetch_skill", boom) + +def test_offline_without_a_bundled_copy_points_at_the_online_copy( + offline: None, + monkeypatch: pytest.MonkeyPatch, + capsys: pytest.CaptureFixture[str], +): + monkeypatch.setattr(cli, "bundled_skill", lambda: None) assert cli.main(["skill"]) == 1 assert "llms-full.txt" in capsys.readouterr().err @@ -56,9 +78,19 @@ def test_no_command_prints_help(capsys: pytest.CaptureFixture[str]): assert "skill" in capsys.readouterr().out -def test_reading_the_skill_writes_the_marker(skill: str, marker: Path): +def test_reading_the_skill_marks_this_session( + skill: str, monkeypatch: pytest.MonkeyPatch +): + monkeypatch.setenv("CLAUDECODE", "1") + monkeypatch.setenv("CLAUDE_CODE_SESSION_ID", "s1") + assert _agent_hint.agent_hint() is not None + assert cli.main(["skill"]) == 0 + assert _agent_hint.agent_hint() is None + + +def test_reading_the_skill_records_the_live_digest(skill: str): assert cli.main(["skill"]) == 0 - assert marker.is_file() + assert _agent_hint._load_state()["live_sha"] == skill_digest(skill) @pytest.fixture From b8a4af1a4950c5692f67697260253cfd86edd405 Mon Sep 17 00:00:00 2001 From: RapidPoseidon Date: Mon, 28 Sep 2026 13:28:08 +0000 Subject: [PATCH 2/7] feat(agents): make the SDK the source of the agent skill The skill in src/rapidata/_skill/ (SKILL.md plus reference, examples and the flows guide, moved from RapidataAI/skills) is now edited only here and ships in every wheel. `python -m rapidata skill [guide]` and the new `rapidata` console script print the bundled copy with no network access. --install stamps the SDK version instead of a content hash, and the import hint compares that stamp to rapidata.__version__ locally. The live GitHub fetch, the daily freshness check and the release-time refresh step are gone. A new `Agent Skill` status fails PRs that leave the skill untouched until a reviewer applies skill-unchanged-approved; generator-only and version-bump PRs are exempt. Co-Authored-By: Claude Opus 5.5 Co-Authored-By: lino@rapidata.ai <68745352+LinoGiger@users.noreply.github.com> --- .github/scripts/agent-skill-check.js | 102 ++ .github/workflows/agent-skill.yml | 53 + .github/workflows/release_and_publish.yml | 5 +- README.md | 8 +- docs/ai_agents.md | 69 +- pyproject.toml | 3 + src/rapidata/AGENTS.md | 9 +- src/rapidata/__init__.py | 6 +- src/rapidata/__main__.py | 86 +- src/rapidata/_agent_hint.py | 115 +- src/rapidata/_skill/SKILL.md | 55 +- src/rapidata/_skill/examples.md | 1119 ++++++++++++ .../_skill/flows-for-preference-data.md | 74 + src/rapidata/_skill/reference.md | 1519 +++++++++++++++++ tests/conftest.py | 4 +- tests/test_agent_hint.py | 73 +- tests/test_main.py | 106 +- 17 files changed, 3098 insertions(+), 308 deletions(-) create mode 100644 .github/scripts/agent-skill-check.js create mode 100644 .github/workflows/agent-skill.yml create mode 100644 src/rapidata/_skill/examples.md create mode 100644 src/rapidata/_skill/flows-for-preference-data.md create mode 100644 src/rapidata/_skill/reference.md diff --git a/.github/scripts/agent-skill-check.js b/.github/scripts/agent-skill-check.js new file mode 100644 index 0000000000..6f1342ca7a --- /dev/null +++ b/.github/scripts/agent-skill-check.js @@ -0,0 +1,102 @@ +// Sets the `Agent Skill` commit status on a PR and keeps its sticky comment in +// sync. Called from .github/workflows/agent-skill.yml via actions/github-script. + +const SKILL_DIR = 'src/rapidata/_skill/'; +const MARKER = ''; + +// Must stay a superset of the paths automerge-openapi-client.yml accepts, or +// the daily generator PR is left with a failing status nobody reviews. +const GENERATED = /^(openapi\/|src\/rapidata\/api_client\/|src\/rapidata\/api_client_README\.md$)/; +const VERSION_BUMP_FILES = new Set(['pyproject.toml', 'src/rapidata/__init__.py']); +const VERSION_BUMP_MESSAGE = /^Bump version from \S+ to \S+/; + +function classify(paths, commitMessages) { + if (paths.some((p) => p.startsWith(SKILL_DIR))) return 'changed'; + if (paths.length > 0 && paths.every((p) => GENERATED.test(p))) return 'generated'; + if ( + paths.length > 0 && + paths.every((p) => VERSION_BUMP_FILES.has(p)) && + commitMessages.length > 0 && + commitMessages.every((m) => VERSION_BUMP_MESSAGE.test(m)) + ) { + return 'version-bump'; + } + return 'unchanged'; +} + +function commentBody(outcome, ackLabel, actor) { + const how = [ + `This PR does not modify \`${SKILL_DIR}\`. That directory is the agent skill that ships in every SDK release and that coding agents read through \`python -m rapidata skill\`.`, + '', + 'Before merging, pick one:', + `1. The change affects what an agent needs to know (new or renamed API, changed parameter, default or result field, new gotcha): update \`${SKILL_DIR}SKILL.md\` (or a companion guide next to it) in this PR.`, + `2. Nothing the skill documents changed: a reviewer applies the \`${ackLabel}\` label. New commits remove the label again.`, + ]; + if (outcome === 'acknowledged') { + return [MARKER, `### ✅ No skill update needed — confirmed by @${actor}`, '', ...how].join('\n'); + } + if (outcome === 'resolved') { + return [MARKER, '### ✅ Agent skill check passed', '', 'The skill was updated or the PR only touches generated or release files.'].join('\n'); + } + return [MARKER, '### ⚠ Agent skill not updated', '', ...how].join('\n'); +} + +module.exports = async function run({ github, context, core, ackLabel, statusContext }) { + const { owner, repo } = context.repo; + const pr = context.payload.pull_request; + const action = context.payload.action; + + if (action === 'synchronize') { + try { + await github.rest.issues.removeLabel({ owner, repo, issue_number: pr.number, name: ackLabel }); + core.info(`Removed ${ackLabel}: new commits need a fresh confirmation.`); + } catch (e) { + if (e.status !== 404) throw e; + } + } + + const files = await github.paginate(github.rest.pulls.listFiles, { owner, repo, pull_number: pr.number, per_page: 100 }); + const paths = files.flatMap((f) => [f.filename, f.previous_filename].filter(Boolean)); + const commits = await github.paginate(github.rest.pulls.listCommits, { owner, repo, pull_number: pr.number, per_page: 100 }); + const kind = classify(paths, commits.map((c) => c.commit.message)); + + const { data: issue } = await github.rest.issues.get({ owner, repo, issue_number: pr.number }); + const labelPresent = issue.labels.some((l) => (typeof l === 'string' ? l : l.name) === ackLabel); + + let state, description, outcome; + if (kind === 'changed') { + [state, description, outcome] = ['success', 'Agent skill updated in this PR.', 'resolved']; + } else if (kind === 'generated') { + [state, description, outcome] = ['success', 'Generated API client only; no skill update needed.', 'resolved']; + } else if (kind === 'version-bump') { + [state, description, outcome] = ['success', 'Release version bump; no skill update needed.', 'resolved']; + } else if (labelPresent) { + [state, description, outcome] = ['success', `No skill update needed (${ackLabel}).`, 'acknowledged']; + } else { + [state, description, outcome] = ['failure', `Update ${SKILL_DIR}SKILL.md or apply ${ackLabel}.`, 'pending']; + } + + const comments = await github.paginate(github.rest.issues.listComments, { owner, repo, issue_number: pr.number, per_page: 100 }); + const existing = comments.find((c) => c.body && c.body.includes(MARKER)); + const body = commentBody(outcome, ackLabel, context.actor); + if (existing) { + if (existing.body !== body) { + await github.rest.issues.updateComment({ owner, repo, comment_id: existing.id, body }); + } + } else if (outcome !== 'resolved') { + await github.rest.issues.createComment({ owner, repo, issue_number: pr.number, body }); + } + + await github.rest.repos.createCommitStatus({ + owner, + repo, + sha: pr.head.sha, + state, + context: statusContext, + description: description.slice(0, 140), + target_url: `${context.serverUrl}/${owner}/${repo}/actions/runs/${context.runId}`, + }); + core.info(`${statusContext}: ${state} (${kind}${labelPresent ? ', labeled' : ''})`); +}; + +module.exports.classify = classify; diff --git a/.github/workflows/agent-skill.yml b/.github/workflows/agent-skill.yml new file mode 100644 index 0000000000..fb7d859fa2 --- /dev/null +++ b/.github/workflows/agent-skill.yml @@ -0,0 +1,53 @@ +name: Agent Skill + +# The agent skill in src/rapidata/_skill/ ships in every wheel and is edited only +# here. A PR that leaves it untouched needs a reviewer to confirm, with the ack +# label, that nothing the skill documents changed. +on: + pull_request: + types: [opened, synchronize, reopened, labeled, unlabeled] + +env: + ACK_LABEL: skill-unchanged-approved + STATUS_CONTEXT: Agent Skill + +permissions: + contents: read + issues: write + pull-requests: write + statuses: write + +jobs: + check: + name: Check skill update + if: ${{ github.event.action != 'labeled' && github.event.action != 'unlabeled' }} + runs-on: ubuntu-latest + timeout-minutes: 3 + steps: + - uses: actions/checkout@v4 + with: + sparse-checkout: .github/scripts + - name: Evaluate + uses: actions/github-script@v7 + with: + script: | + const run = require('./.github/scripts/agent-skill-check.js'); + await run({ github, context, core, ackLabel: process.env.ACK_LABEL, statusContext: process.env.STATUS_CONTEXT }); + + ack: + name: Acknowledge label change + if: >- + (github.event.action == 'labeled' || github.event.action == 'unlabeled') + && github.event.label.name == 'skill-unchanged-approved' + runs-on: ubuntu-latest + timeout-minutes: 3 + steps: + - uses: actions/checkout@v4 + with: + sparse-checkout: .github/scripts + - name: Re-evaluate + uses: actions/github-script@v7 + with: + script: | + const run = require('./.github/scripts/agent-skill-check.js'); + await run({ github, context, core, ackLabel: process.env.ACK_LABEL, statusContext: process.env.STATUS_CONTEXT }); diff --git a/.github/workflows/release_and_publish.yml b/.github/workflows/release_and_publish.yml index fe869d0992..9514284eb0 100644 --- a/.github/workflows/release_and_publish.yml +++ b/.github/workflows/release_and_publish.yml @@ -142,12 +142,9 @@ jobs: f.write(f'__version__ = "{version}"\n') EOF - - name: Refresh bundled agent skill - run: curl -fsSL https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/SKILL.md -o src/rapidata/_skill/SKILL.md - - name: Commit and push changes run: | - git add pyproject.toml src/rapidata/__init__.py src/rapidata/_skill/SKILL.md + git add pyproject.toml src/rapidata/__init__.py git commit -m "Bump version from ${{ steps.update_version.outputs.old_version }} to ${{ steps.update_version.outputs.new_version }}" git push origin ${{ github.event.inputs.branch }} diff --git a/README.md b/README.md index e6fd70bd62..5b0f2793d3 100644 --- a/README.md +++ b/README.md @@ -6,11 +6,13 @@ Docs: https://docs.rapidata.ai/ ## Using a coding agent? -Point it at the maintained skill instead of letting it read the installed source: +Point it at the guide that ships with the SDK instead of letting it read the installed source: ```bash -python -m rapidata skill # print the guide -python -m rapidata skill --install # install it into the current project +rapidata skill # print the guide (same as python -m rapidata skill) +rapidata skill --install # install it into the current project ``` +The guide is edited in [`src/rapidata/_skill/`](https://github.com/RapidataAI/rapidata-python-sdk/tree/main/src/rapidata/_skill), so it always matches the installed version. + Details and per-agent install commands: https://docs.rapidata.ai/ai_agents/ diff --git a/docs/ai_agents.md b/docs/ai_agents.md index 2e346c1856..574c572eb6 100644 --- a/docs/ai_agents.md +++ b/docs/ai_agents.md @@ -19,25 +19,16 @@ Pick your agent. One command. Done. Install once. Works in every session after that. That's it. -Already have the SDK installed? It carries the same skill: +The installed skill is a short pointer: it tells the agent to install the SDK and read the full guide that ships with it. Already have the SDK? You can skip the install and read the guide directly: ```bash -python -m rapidata skill # print it -python -m rapidata skill --install # write it to .claude/skills/rapidata/SKILL.md -python -m rapidata skill --install --agent cursor # or cursor, codex, generic (AGENTS.md) +rapidata skill # print the guide (same as python -m rapidata skill) +rapidata skill reference # companion guides: reference, examples, flows-for-preference-data +rapidata skill --install # write it to .claude/skills/rapidata/SKILL.md +rapidata skill --install --agent cursor # or codex, generic (AGENTS.md) ``` -??? note "No install — just the raw SKILL.md" - - If your framework doesn't match any of the above, drop the raw file into your agent's context: - - [**SKILL.md on GitHub**](https://github.com/RapidataAI/skills/blob/main/plugins/rapidata-sdk-plugin/skills/rapidata/SKILL.md) - - Raw URL for fetching: - - ``` - https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/SKILL.md - ``` +The guide is versioned with the SDK, so it always describes the version you have installed. ## Logging in @@ -81,56 +72,22 @@ Other agents follow their own conventions — Cursor rules, Copilot instructions ## Keeping the skill up to date -The Rapidata SDK evolves constantly — new task types, new audience features, better defaults. A skill that lags behind the SDK will describe methods that have changed, so either let Claude Code update it for you or pull it yourself. - -### Automatic — Claude Code - -Claude Code refreshes marketplaces and updates their installed plugins in the background shortly after a session starts. This is off by default for marketplaces outside Anthropic's own, so switch it on once: - -`/plugin` → **Marketplaces** → `rapidata-sdk-marketplace` → **Enable auto-update** - -To set it for everyone on a project, commit it to `.claude/settings.json` — the same block works in your personal `~/.claude/settings.json`: - -```json -{ - "extraKnownMarketplaces": { - "rapidata-sdk-marketplace": { - "source": { "source": "github", "repo": "RapidataAI/skills" }, - "autoUpdate": true - } - }, - "enabledPlugins": { - "rapidata-sdk-plugin@rapidata-sdk-marketplace": true - } -} -``` - -When an update lands mid-session, Claude Code asks you to run `/reload-plugins`. Otherwise it takes effect on your next launch. - -### Manual - -Claude Code: +The full guide ships inside the SDK, so upgrading the SDK upgrades the guide: ```bash -claude plugin marketplace update +pip install -U rapidata # or: uv add -U rapidata ``` -Everything else — the `skills` CLI updates only when you ask it to: +The installed skill from the table above only points the agent at `python -m rapidata skill`, so it rarely needs an update. Pull one anyway with `claude plugin marketplace update` (Claude Code) or `npx skills update rapidata` (everything else). -```bash -npx skills update rapidata -``` - -Or update every skill you've installed at once: +Copies written by `rapidata skill --install` carry the SDK version that wrote them. After an SDK upgrade, the next import by a coding agent tells it to run `rapidata skill --install` again. This check is local and needs no network. -```bash -npx skills update -``` +## Editing the skill -Copies written by `python -m rapidata skill --install` carry a stamp of the version they were made from. When a coding agent imports the SDK, it checks that stamp against the live skill at most once a day and, if it is out of date, tells the agent to run `python -m rapidata skill --install` again. +The guide lives in the SDK repository at [`src/rapidata/_skill/`](https://github.com/RapidataAI/rapidata-python-sdk/tree/main/src/rapidata/_skill) and is edited only there. A pull request that changes SDK behaviour updates it in the same change; the `Agent Skill` check asks the reviewer to confirm when it does not. ## The import-time hint -When the SDK is imported by a coding agent (Claude Code, Codex, Cursor, Gemini CLI) that has no copy of the skill installed, it prints a short pointer to `python -m rapidata skill` on stderr. It shows once per agent session and stops as soon as that session reads the guide. Humans running the SDK directly never see it. +When the SDK is imported by a coding agent (Claude Code, Codex, Cursor, Gemini CLI) that has not read the guide yet, it prints a short pointer to `python -m rapidata skill` on stderr. It shows once per agent session and stops as soon as that session reads the guide. Humans running the SDK directly never see it. Processes an agent merely started — a dev server, a script whose stderr is parsed — inherit its environment and would show the hint too. Set `RAPIDATA_AGENT_HINT=0` for those. diff --git a/pyproject.toml b/pyproject.toml index 34ba8b6999..7649c2c594 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -27,6 +27,9 @@ dependencies = [ "qrcode>=8.0,<9", ] +[project.scripts] +rapidata = "rapidata.__main__:main" + [dependency-groups] dev = [ "python-dotenv>=1.0.1,<2", diff --git a/src/rapidata/AGENTS.md b/src/rapidata/AGENTS.md index d17d559c80..f5233d480f 100644 --- a/src/rapidata/AGENTS.md +++ b/src/rapidata/AGENTS.md @@ -1,16 +1,19 @@ # Rapidata SDK — for coding agents Do not infer how to use this SDK from the source files next to this document. -The maintained guide covers job types, audiences, validation sets, result -fields and the mistakes agents make most often. +The guide that ships with this version covers job types, audiences, validation +sets, result fields and the mistakes agents make most often. Read it first: ```bash -python -m rapidata skill # print the guide +python -m rapidata skill # print the guide (also: rapidata skill) python -m rapidata skill --install # install it into the current project ``` +The guide's source is `_skill/SKILL.md` in this package, edited only in the +RapidataAI/rapidata-python-sdk repository. + ## Logging in `RapidataClient()` needs credentials. With none saved, it opens a browser login diff --git a/src/rapidata/__init__.py b/src/rapidata/__init__.py index dbbeeb0be0..0964342248 100644 --- a/src/rapidata/__init__.py +++ b/src/rapidata/__init__.py @@ -1,8 +1,8 @@ """Rapidata Python SDK: human feedback at scale through ``RapidataClient``. -Coding agents: do not infer usage from this source tree. Read the maintained -guide first with ``python -m rapidata skill`` (``--install`` keeps it in the -project) or at https://docs.rapidata.ai/llms-full.txt. Check authentication with +Coding agents: do not infer usage from this source tree. Read the guide that +ships with this version first: ``python -m rapidata skill`` (``--install`` keeps +it in the project). Check authentication with ``python -m rapidata status`` and log the user in with ``python -m rapidata login``. """ diff --git a/src/rapidata/__main__.py b/src/rapidata/__main__.py index af48ff1676..18849e30eb 100644 --- a/src/rapidata/__main__.py +++ b/src/rapidata/__main__.py @@ -1,8 +1,9 @@ """``python -m rapidata`` — utilities that do not need an authenticated client. -``python -m rapidata skill`` prints the maintained agent skill for this SDK; -``--install`` writes it into the current project so a coding agent loads it on -every session instead of reading the installed source. +``python -m rapidata skill`` (or ``rapidata skill``) prints the agent skill +bundled with this SDK version; ``rapidata/_skill/`` is its source, edited only +in this repository. ``--install`` writes it into the current project so a coding +agent loads it on every session instead of reading the installed source. ``python -m rapidata status`` reports whether this machine can authenticate without starting a login; ``python -m rapidata login`` runs the browser login @@ -17,36 +18,29 @@ from importlib import resources from pathlib import Path -import requests - from rapidata import __version__ from rapidata._agent_hint import ( AGENT_DOCS_URL, - LLMS_FULL_URL, SKILL_INSTALL_PATHS, - SKILL_RAW_URL, mark_skill_read, - record_live_skill, stamp_skill, ) - -def fetch_skill(timeout: float = 10) -> str: - response = requests.get(SKILL_RAW_URL, timeout=timeout) - response.raise_for_status() - return response.text +# `python -m rapidata skill ` -> file in rapidata/_skill/. +SKILL_GUIDES: dict[str, str] = { + "main": "SKILL.md", + "reference": "reference.md", + "examples": "examples.md", + "flows-for-preference-data": "flows-for-preference-data.md", +} -def bundled_skill() -> str | None: - """The copy of the skill refreshed into the wheel at release time, for when GitHub is unreachable.""" - try: - return ( - resources.files("rapidata") - .joinpath("_skill/SKILL.md") - .read_text(encoding="utf-8") - ) - except (OSError, ModuleNotFoundError): - return None +def bundled_skill(guide: str = "main") -> str: + return ( + resources.files("rapidata") + .joinpath(f"_skill/{SKILL_GUIDES[guide]}") + .read_text(encoding="utf-8") + ) def install_skill(root: Path, agent: str, content: str) -> Path: @@ -113,7 +107,7 @@ def login(environment: str) -> int: def _build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser( - prog="python -m rapidata", + prog="rapidata", description=f"Rapidata SDK {__version__}. Agent guide: {AGENT_DOCS_URL}", ) sub = parser.add_subparsers(dest="command") @@ -121,10 +115,17 @@ def _build_parser() -> argparse.ArgumentParser: "skill", help="print the agent skill for this SDK, or install it into the project", ) + skill.add_argument( + "guide", + nargs="?", + choices=list(SKILL_GUIDES), + default="main", + help="which guide to print (default: main, which links the others)", + ) skill.add_argument( "--install", action="store_true", - help="write the skill into the current project instead of printing it", + help="write the main guide into the current project instead of printing it", ) skill.add_argument( "--agent", @@ -164,32 +165,23 @@ def main(argv: list[str] | None = None) -> int: parser.print_help() return 0 - try: - content = fetch_skill() - record_live_skill(content) - except requests.RequestException as e: - bundled = bundled_skill() - if bundled is None: - print(f"Could not fetch the skill ({e}).", file=sys.stderr) - print( - f"Read it online instead: {SKILL_RAW_URL} or {LLMS_FULL_URL}", - file=sys.stderr, - ) - return 1 - print( - f"Could not fetch the latest skill ({e}); using the copy bundled with " - f"rapidata {__version__}. Latest: {SKILL_RAW_URL}", - file=sys.stderr, - ) - content = bundled - - mark_skill_read() if args.install: - target = install_skill(args.dir, args.agent, content) + if args.guide != "main": + parser.error("--install writes the main guide; drop the guide name") + mark_skill_read() + target = install_skill(args.dir, args.agent, bundled_skill()) print(f"Installed the Rapidata skill to {target}") return 0 - print(content) + content = bundled_skill(args.guide) + if args.guide == "main": + mark_skill_read() + try: + print(content) + sys.stdout.flush() + except BrokenPipeError: + # `rapidata skill | head` closes the pipe early; the flush at exit would raise again. + os.dup2(os.open(os.devnull, os.O_WRONLY), sys.stdout.fileno()) return 0 diff --git a/src/rapidata/_agent_hint.py b/src/rapidata/_agent_hint.py index ce169916de..de4d8e11c7 100644 --- a/src/rapidata/_agent_hint.py +++ b/src/rapidata/_agent_hint.py @@ -1,4 +1,4 @@ -"""Point a coding agent at the maintained SDK guide when it imports ``rapidata``. +"""Point a coding agent at the guide bundled with this SDK when it imports ``rapidata``. Agents explore a freshly installed SDK with ``import rapidata`` / ``dir()`` / ``inspect`` before they write a script, and they read stderr but not @@ -8,33 +8,27 @@ ``python -m rapidata skill``. Sessions are told apart by the id the runtime exports (:data:`_SESSION_ENV_VARS`); runtimes without one get a read that expires after :data:`ANON_READ_TTL`. -- Never while a copy of the skill the agent loads on its own is installed - (project, user level or the Claude Code plugin), unless a copy written by - ``--install`` no longer matches the live skill. That is checked at most once - per :data:`FRESHNESS_TTL` with a :data:`FRESHNESS_TIMEOUT` request, and then - the pointer asks for a reinstall instead. +- Never while the Claude Code plugin is installed, or while a copy written by + ``python -m rapidata skill --install`` (project or user level) carries this + SDK's version stamp. A copy stamped with another version asks for a + reinstall instead. Nothing here touches the network. ``RAPIDATA_AGENT_HINT=0`` switches it off for processes an agent merely started. State lives in :data:`STATE_FILE`, falling back to the temp dir when a sandbox -makes the home directory read-only. Kept free of SDK imports: +makes the home directory read-only. Kept free of client imports: ``rapidata/__init__.py`` calls it before loading the client. """ from __future__ import annotations -import hashlib import json import os import re import sys import tempfile import time -import urllib.request -from datetime import datetime, timezone from pathlib import Path -SKILL_RAW_URL = "https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/SKILL.md" -LLMS_FULL_URL = "https://docs.rapidata.ai/llms-full.txt" AGENT_DOCS_URL = "https://docs.rapidata.ai/ai_agents/" PLUGIN_NAME = "rapidata-sdk-plugin" @@ -56,8 +50,6 @@ FALLBACK_STATE_FILE = Path(tempfile.gettempdir()) / "rapidata-agent-state.json" ANON_READ_TTL = 12 * 3600 -FRESHNESS_TTL = 24 * 3600 -FRESHNESS_TIMEOUT = 1.0 _PRUNE_AFTER = 7 * 24 * 3600 # Env var each agent runtime exports -> the name reported in traces. Specific @@ -73,14 +65,13 @@ _SESSION_ENV_VARS = ("CLAUDE_CODE_SESSION_ID", "CODEX_THREAD_ID", "CODEX_SESSION_ID") -_STAMP_RE = re.compile(r"\n") +_STAMP_RE = re.compile(r"\n") AGENT_HINT = ( - "rapidata: coding agent detected. Read the maintained SDK guide before exploring " + "rapidata: coding agent detected. Read the SDK guide for this version before exploring " "the installed source (skip if this session already read it):\n" " python -m rapidata skill # print the guide\n" " python -m rapidata skill --install # keep it in this project\n" - f" {LLMS_FULL_URL}\n" "Before the first RapidataClient(), run `python -m rapidata status`. If it reports not logged in,\n" "run `python -m rapidata login` and show the user the URL it prints (it waits up to 5 minutes)." ) @@ -103,16 +94,16 @@ def running_under_coding_agent() -> bool: return detected_coding_agent() is not None -def skill_digest(content: str) -> str: - return hashlib.sha256(content.encode("utf-8")).hexdigest() +def _sdk_version() -> str: + # Lazy: rapidata/__init__.py imports this module while the package is still loading. + from rapidata import __version__ + return __version__ -def stamp_skill(content: str, now: datetime | None = None) -> str: - """Return ``content`` with a provenance line after its front matter, so staleness can be checked later.""" - fetched = (now or datetime.now(timezone.utc)).strftime("%Y-%m-%dT%H:%M:%SZ") - stamp = ( - f"\n" - ) + +def stamp_skill(content: str, version: str | None = None) -> str: + """Return ``content`` with a version line after its front matter, so a copy made by another SDK version can be told apart.""" + stamp = f"\n" if content.startswith("---\n"): end = content.find("\n---\n", 4) if end != -1: @@ -121,15 +112,10 @@ def stamp_skill(content: str, now: datetime | None = None) -> str: return stamp + content -def _installed_digest(text: str) -> str | None: - """Digest of the live skill this installed copy was made from, or None when it is not the Rapidata skill.""" +def installed_version(text: str) -> str | None: + """SDK version an ``--install``ed copy was written by, or None when ``text`` carries no stamp.""" match = _STAMP_RE.search(text) - if match: - return match.group(1) - # Copies written by --install before stamping existed are verbatim. - if text.startswith("---\nname: rapidata\n"): - return skill_digest(text) - return None + return match.group(1) if match else None def _load_state() -> dict: @@ -187,14 +173,6 @@ def mark_skill_read() -> None: _save_state(state) -def record_live_skill(content: str) -> None: - """Remember the live skill's digest so the next freshness check can skip the network.""" - state = _load_state() - state["live_sha"] = skill_digest(content) - state["live_checked_at"] = time.time() - _save_state(state) - - def _plugin_installed() -> bool: config_dir = Path(os.environ.get("CLAUDE_CONFIG_DIR") or Path.home() / ".claude") try: @@ -209,7 +187,7 @@ def _plugin_installed() -> bool: def installed_copies(root: Path | None = None) -> list[tuple[str, Path, Path, str]]: - """Return ``(agent, install_root, path, digest)`` for each Rapidata skill file in the project ``root`` or the home directory.""" + """Return ``(agent, install_root, path, version)`` for each stamped skill file in the project ``root`` or the home directory.""" root = root or Path.cwd() candidates = [(a, root, rel) for a, rel in SKILL_INSTALL_PATHS.items()] candidates += [(a, Path.home(), rel) for a, rel in USER_SKILL_PATHS.items()] @@ -217,49 +195,35 @@ def installed_copies(root: Path | None = None) -> list[tuple[str, Path, Path, st for agent, base, rel in candidates: path = base / rel try: - digest = _installed_digest(path.read_text(encoding="utf-8")) + version = installed_version(path.read_text(encoding="utf-8")) except (OSError, UnicodeDecodeError): continue - if digest: - copies.append((agent, base, path, digest)) + if version: + copies.append((agent, base, path, version)) return copies -def _fetch_live_digest() -> str | None: - try: - with urllib.request.urlopen(SKILL_RAW_URL, timeout=FRESHNESS_TIMEOUT) as resp: - return skill_digest(resp.read().decode("utf-8")) - except Exception: - return None - - -def _live_digest(state: dict) -> str | None: - if time.time() - state.get("live_checked_at", 0) < FRESHNESS_TTL: - return state.get("live_sha") - live = _fetch_live_digest() - # Recorded on failure too, so an offline machine pays the timeout once a day, not per import. - state["live_checked_at"] = time.time() - if live: - state["live_sha"] = live - _save_state(state) - return state.get("live_sha") - - -def _stale_hint(agent: str, base: Path, path: Path) -> str: +def _stale_hint(agent: str, base: Path, path: Path, version: str) -> str: cmd = "python -m rapidata skill --install" if agent != "claude": cmd += f" --agent {agent}" if base != Path.cwd(): cmd += f" --dir {base}" - return f"rapidata: the installed Rapidata skill at {path} is outdated. Update it with: {cmd}" + return ( + f"rapidata: the Rapidata skill at {path} was installed by rapidata {version}, " + f"but {_sdk_version()} is installed. Update it with: {cmd}" + ) def _running_the_cli() -> bool: argv = getattr(sys, "orig_argv", []) - return any( + if any( a == "-m" and i + 1 < len(argv) and argv[i + 1] == "rapidata" for i, a in enumerate(argv) - ) + ): + return True + # The `rapidata` console script imports the package before main() runs. + return bool(sys.argv) and Path(sys.argv[0]).stem == "rapidata" def agent_hint() -> str | None: @@ -267,15 +231,12 @@ def agent_hint() -> str | None: try: if not running_under_coding_agent() or _running_the_cli(): return None - state = _load_state() copies = installed_copies() - if copies: - live = _live_digest(state) - for agent, base, path, digest in copies: - if live and digest != live: - return _stale_hint(agent, base, path) - return None - if _plugin_installed() or _read_this_session(state): + current = _sdk_version() + for agent, base, path, version in copies: + if version != current: + return _stale_hint(agent, base, path, version) + if copies or _plugin_installed() or _read_this_session(_load_state()): return None return AGENT_HINT except Exception: diff --git a/src/rapidata/_skill/SKILL.md b/src/rapidata/_skill/SKILL.md index 3b4cb22a4e..f378035b4f 100644 --- a/src/rapidata/_skill/SKILL.md +++ b/src/rapidata/_skill/SKILL.md @@ -7,26 +7,11 @@ description: Explains how to use the Rapidata API to get real and fast human ann Rapidata connects you with distributed human labelers worldwide for fast, high-quality data annotation. The SDK lets you create labeling tasks, manage annotator audiences, and retrieve results programmatically. -## Before you start: check the skill is up to date +## This guide ships with the SDK -This skill is pinned to **Rapidata SDK v3.25.7**. Run this check **once at the start of a Rapidata task** (not on every call) to confirm the user's runtime matches the skill: +`python -m rapidata skill` (or `rapidata skill`) prints the copy bundled with the installed `rapidata` package, so it describes exactly that version. After `pip install -U rapidata`, run it again: an upgrade can change what is documented here. -```bash -python -c "import rapidata; print(rapidata.__version__)" 2>/dev/null \ - || pip show rapidata 2>/dev/null | awk -F': ' '/^Version:/{print $2}' -``` - -Compare the output to the pinned version above: - -- **Installed > pinned** — this skill is **outdated**. The SDK may have new features, renamed methods, or changed signatures that this skill does not document. - 1. Suggest updating the skill: in Claude Code, `/plugin marketplace update rapidata-sdk-marketplace` (or `/plugin` → `rapidata-sdk-plugin` → update); outside the plugin, `python -m rapidata skill --install` rewrites the project-local copy. - 2. Tell the user clearly: - > ⚠️ The Rapidata skill is pinned to v3.25.7 but v{installed} is installed — the skill docs may be out of date. I've suggested updating the plugin; if the update isn't available yet, I'll proceed with the documented API and flag any surprises. - 3. Proceed using the documented API. If you hit an unexpected error (missing attribute, changed signature), stop and tell the user the skill is likely the cause — don't guess at the new API. - -- **Installed < pinned** — the user's runtime is older than this skill. Suggest `pip install -U rapidata` so the runtime matches. - -- **Match, or rapidata not installed** — proceed normally. (If not installed, the installation section below is the first step anyway.) +The three companion guides linked at the end print the same way: `python -m rapidata skill reference`, `python -m rapidata skill examples`, `python -m rapidata skill flows-for-preference-data`. ## Installation & Authentication @@ -49,6 +34,17 @@ client = RapidataClient(client_id="...", client_secret="...") # Empty values are treated as unset and fall through to the next resolution layer. ``` +Without saved credentials, `RapidataClient()` blocks on a browser login for up to 5 minutes. As a coding agent, check auth first from the CLI: + +```bash +python -m rapidata status [--environment ENV] # exit 0: "Authenticated for {env} via {source}."; exit 1: not logged in. Never starts a login +python -m rapidata login [--environment ENV] # browser login; saves credentials (no-op if already authenticated). Blocks up to 5 minutes +``` + +- Both check `RAPIDATA_TOKEN_FILE` → `RAPIDATA_CLIENT_ID` + `RAPIDATA_CLIENT_SECRET` → credentials saved for `https://auth.{environment}`. `--environment` defaults to `RAPIDATA_ENVIRONMENT`, else `rapidata.ai`. +- Run `python -m rapidata status` before the first `RapidataClient()`. If it reports not logged in, run `python -m rapidata login` in the background (or with a tool timeout above 5 minutes) and show the user the URL it prints — the login URL is always printed to stderr, even in silent mode. After the user logs in once, every later `RapidataClient()` reuses the saved credentials. +- For headless runs, set `RAPIDATA_CLIENT_ID` and `RAPIDATA_CLIENT_SECRET` instead. + ### Sharing a token across many workers (distributed training) When Rapidata is queried from a large distributed job (e.g. a ranking flow hit from hundreds or thousands of GPU workers), don't let every worker authenticate on its own — all their tokens expire at the same instant and the simultaneous re-auth looks like a coordinated burst that gets rate-limited. Instead, authenticate **once** and share the token via a file: @@ -506,7 +502,7 @@ settings=[CustomSetting(key="my_flag", value="on")] # Rapid-level flag (ta Flows continuously collect human responses in small batches without full job/audience setup. There are two kinds: **ranking flows** (Elo-style comparison, below) and **classify flows** (sort each datapoint into a category, further below). -The flow classes are importable from the top level: `from rapidata import RapidataFlow, RapidataRankingFlow, RapidataClassifyFlow`. `RapidataFlow` is a shared base class (only `get_flow_items` and `delete`); the concrete `RapidataRankingFlow` and `RapidataClassifyFlow` subclasses carry the kind-specific `create_new_flow_batch` (and, for ranking, `update_config`). Every flow carries a `flow_type` (`"ranking"` or `"simple"` — classify flows are backed by "simple" flows). `create_ranking_flow` returns a `RapidataRankingFlow` and `create_classify_flow` returns a `RapidataClassifyFlow`, but `get_flow_by_id` and `find_flows` return the union `RapidataRankingFlow | RapidataClassifyFlow` — narrow with `isinstance(flow, RapidataClassifyFlow)` before calling kind-specific batch methods. +The flow classes are importable from the top level: `from rapidata import RapidataFlow, RapidataRankingFlow, RapidataClassifyFlow`. `RapidataFlow` is a shared base class (only `get_flow_items` and `delete`); the concrete `RapidataRankingFlow` and `RapidataClassifyFlow` subclasses carry the kind-specific `create_new_flow_batch` and `update_config`. Every flow carries a `flow_type` (`"ranking"` or `"simple"` — classify flows are backed by "simple" flows). `create_ranking_flow` returns a `RapidataRankingFlow` and `create_classify_flow` returns a `RapidataClassifyFlow`, but `get_flow_by_id` and `find_flows` return the union `RapidataRankingFlow | RapidataClassifyFlow` — narrow with `isinstance(flow, RapidataClassifyFlow)` before calling kind-specific batch methods. ### Ranking Flows @@ -521,6 +517,7 @@ flow = client.flow.create_ranking_flow( min_response_threshold=50, # Minimum acceptable responses (defaults to max_response_threshold); item is Incomplete if TTL expires below this # validation_set_id="...", # Optional: run a validation set alongside the flow # settings=[NoShuffleSetting()], # Optional: flow-wide settings + # drain_duration=..., serve_timeout=..., # Optional: seconds (drainDurationSeconds / serveTimeoutSeconds) ) # Preheat for low-latency responses (call ~5 minutes before time-sensitive batches) @@ -531,7 +528,7 @@ flow_item = flow.create_new_flow_batch( datapoints=["img1.jpg", "img2.jpg", "img3.jpg"], context="Generated by Model X", # A single batch-level context shared across all comparisons # context_assets=["reference.jpg"], # 1–10 image/video/audio paths/URLs shown alongside instruction (flat list) - time_to_live=300, # Seconds until expiry (45–3600; ranking flows default to 4 minutes when omitted) + time_to_live=300, # Seconds until expiry (10–3600, min 60 with default flow settings; defaults to 4 minutes when omitted) # data_type="media", private_metadata=[...], accept_failed_uploads=False, # also accepted ) # Ranking batches take a batch-level context / context_assets only — no per-datapoint @@ -543,12 +540,13 @@ status = flow_item.get_status() # Non-blocking check (Pending, Running matrix = flow_item.get_win_loss_matrix() # Pandas DataFrame (blocks until complete); ranking flow items only — raises ValueError otherwise count = flow_item.get_response_count() -# Tune a ranking flow after creation (only RapidataRankingFlow has update_config) +# Tune a ranking flow after creation (only the parameters you pass are changed) flow.update_config( instruction="New instruction", starting_elo=1000, min_responses=40, max_responses=120, + # drain_duration=..., serve_timeout=..., # seconds ) # Manage flows @@ -560,7 +558,7 @@ flow.delete() Note: `RapidataFlowItem` does **not** have `display_progress_bar()` — poll with `get_status()` or just call `get_results()` to block. -Using flows as the preference signal for DPO/RLHF or best-of-N? Read [flows-for-preference-data.md](flows-for-preference-data.md) first: TTL, not the max threshold, usually sets a flow item's turnaround. +Using flows as the preference signal for DPO/RLHF or best-of-N? Read the flows guide first (`python -m rapidata skill flows-for-preference-data`): TTL, not the max threshold, usually sets a flow item's turnaround. ### Classify Flows @@ -580,16 +578,21 @@ flow = client.flow.create_classify_flow( # Requires max_responses_per_datapoint >= min_responses_per_datapoint. # validation_set_id="...", # Optional # settings=[...], # Optional: flow-wide settings + # drain_duration=..., serve_timeout=..., # Optional: seconds (drainDurationSeconds / serveTimeoutSeconds) ) # Time-to-live is set per batch only (create_new_flow_batch below), not on the flow. +# After creation only the drain duration can be changed (instruction, categories and +# response thresholds are fixed); omitting it keeps the current value +flow.update_config(drain_duration=20) # seconds + # Add a batch of datapoints to classify flow_item = flow.create_new_flow_batch( datapoints=["img1.jpg", "img2.jpg"], contexts=["Text context for img1", "Text context for img2"], # per-datapoint; exactly one per datapoint # context_assets=[["ref1.jpg"], ["ref2a.jpg", "ref2b.jpg"]], # Optional: one list of asset paths/URLs per # datapoint (list[list[str]] — each entry is a list even for a single asset); length must match datapoints - # time_to_live=300, # Optional: seconds this batch may run (45–3600); set per batch + # time_to_live=300, # Optional: seconds this batch may run (10–3600, min 60 with default flow settings); set per batch # data_type="media", private_metadata=[...], accept_failed_uploads=False, # also accepted ) # Classify batches take per-datapoint contexts / context_assets only — no batch-level context. @@ -921,6 +924,6 @@ owed = client.billing.get_outstanding_balance() # -> float ## Additional Resources -- For complete API reference, all parameters, filters, results format, error handling, flows, and MRI: see [reference.md](reference.md) (if this file was installed on its own via `python -m rapidata skill --install`, fetch https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/reference.md) -- For full end-to-end code examples and common patterns: see [examples.md](examples.md) (standalone copy: https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/examples.md) -- For using flows to collect DPO/RLHF preference pairs or run best-of-N, and designing comparisons that train well: see [flows-for-preference-data.md](flows-for-preference-data.md) (standalone copy: https://raw.githubusercontent.com/RapidataAI/skills/main/plugins/rapidata-sdk-plugin/skills/rapidata/flows-for-preference-data.md) +- For complete API reference, all parameters, filters, results format, error handling, flows, and MRI: run `python -m rapidata skill reference` +- For full end-to-end code examples and common patterns: run `python -m rapidata skill examples` +- For using flows to collect DPO/RLHF preference pairs or run best-of-N, and designing comparisons that train well: run `python -m rapidata skill flows-for-preference-data` diff --git a/src/rapidata/_skill/examples.md b/src/rapidata/_skill/examples.md new file mode 100644 index 0000000000..f7bd6d3487 --- /dev/null +++ b/src/rapidata/_skill/examples.md @@ -0,0 +1,1119 @@ +# Rapidata SDK — Code Examples + +## Simple Classification + +```python +from rapidata import RapidataClient + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_classification_job_definition( + name="Image Classification", + instruction="What's in this image?", + answer_options=["Cat", "Dog", "Bird"], + datapoints=["img1.jpg", "img2.jpg"], +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +df = results.to_pandas() +``` + +## Classification with Likert Scale + +```python +from rapidata import RapidataClient, NoShuffleSetting + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_classification_job_definition( + name="Quality Rating", + instruction="How well does the video match the description?", + answer_options=["1: Poor", "2: Fair", "3: Good", "4: Excellent"], + datapoints=["video1.mp4", "video2.mp4"], + contexts=["A cat playing piano", "A sunset over the ocean"], + responses_per_datapoint=15, + settings=[NoShuffleSetting()], # Critical for ordered scales +) +job = audience.assign_job(job_def) +``` + +## Comparison with Early Stopping + +```python +from rapidata import RapidataClient + +client = RapidataClient() +audience = client.audience.get_audience_by_id("aud_MU1GZYoESyO") + +job_def = client.job.create_compare_job_definition( + name="Model Comparison", + instruction="Which image follows the prompt better?", + datapoints=[ + ["flux_cat.jpg", "mj_cat.jpg"], + ["flux_sunset.jpg", "mj_sunset.jpg"], + ], + contexts=["A cat on a chair", "A sunset over mountains"], + responses_per_datapoint=50, + confidence_threshold=0.99, + a_b_names=["Flux", "Midjourney"], +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +## Classification with Quorum Stopping + +```python +from rapidata import RapidataClient + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_classification_job_definition( + name="Animal Classification with Quorum", + instruction="What animal is in this image?", + answer_options=["Cat", "Dog"], + datapoints=["pet1.jpg", "pet2.jpg"], + responses_per_datapoint=10, # Maximum responses + quorum_threshold=7, # Stop when 7 responses agree +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +## Comparison Allowing "Neither" / "Both" + +```python +from rapidata import RapidataClient, AllowNeitherBothSetting + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_compare_job_definition( + name="Text Comparison", + instruction="Which response is more helpful?", + # With data_type="text" each datapoint is the text itself, not a file path. + datapoints=[[ + "Restart the router, then reconnect.", + "Have you tried turning it off and on again?", + ]], + data_type="text", + settings=[AllowNeitherBothSetting()], +) +job = audience.assign_job(job_def) +``` + +## Locate Job + +```python +from rapidata import RapidataClient + +client = RapidataClient() + +# Simple: use the ready-to-go global pool +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_locate_job_definition( + name="Artifact Detection", + instruction="Tap on any visual glitches or errors in the image.", + datapoints=["img1.jpg", "img2.jpg", "img3.jpg"], + responses_per_datapoint=35, +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +Custom audience with locate qualification examples: + +```python +from rapidata import RapidataClient, Box + +client = RapidataClient() + +audience = client.audience.create_audience(name="Artifact Detection Audience") + +EXAMPLES = [ + ("example1.jpg", [Box(x_min=0.44, y_min=0.42, x_max=0.58, y_max=0.63)]), + ("example2.jpg", [Box(x_min=0.07, y_min=0.37, x_max=0.39, y_max=0.71)]), + ("example3.jpg", [Box(x_min=0.04, y_min=0.10, x_max=0.31, y_max=0.28)]), +] + +for datapoint, truths in EXAMPLES: + audience.add_locate_example( + instruction="Tap on any visual glitches or errors in the image.", + datapoint=datapoint, + truths=truths, + explanation="The artifact is within the highlighted region.", + ) + +audience.start_recruiting() # required — a job assigned before this can never receive responses + +job_def = client.job.create_locate_job_definition( + name="Artifact Detection", + instruction="Tap on any visual glitches or errors in the image.", + datapoints=["img1.jpg", "img2.jpg", "img3.jpg"], + responses_per_datapoint=35, +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +## Draw Job + +```python +from rapidata import RapidataClient + +client = RapidataClient() + +# Simple: use the ready-to-go global pool +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_draw_job_definition( + name="Artifact Drawing", + instruction="Color in any visual glitches or errors in the image.", + datapoints=["img1.jpg", "img2.jpg", "img3.jpg"], + responses_per_datapoint=35, +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +For a custom draw audience, follow the locate custom-audience example above with `audience.add_draw_example(...)` (same `truths=list[Box]` shape). + +## Select Words Job + +```python +from rapidata import RapidataClient + +IMAGES = ["img1.jpg", "img2.jpg", "img3.jpg", "img4.jpg"] +PROMPTS = [ + "The black camera was next to the white tripod.", + "Four cars on the street.", + "Car is bigger than the airplane.", + "One cat and two dogs sitting on the grass.", +] +SENTENCES = [p + " [No_mistakes]" for p in PROMPTS] + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_select_words_job_definition( + name="Image-Text Alignment", + instruction="The image is based on the text below. Select mistakes, i.e., words that are not aligned with the image.", + datapoints=IMAGES, + sentences=SENTENCES, + responses_per_datapoint=15, +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +## Free Text Job + +```python +from rapidata import RapidataClient + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_free_text_job_definition( + name="Prompt Collection", + instruction="What would you like to ask an AI? Please spell out the question.", + datapoints=["image.jpg"], + responses_per_datapoint=15, +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +## Free-Text Length Constraints + +Use length constraints sparingly. Free-text responses are already filtered by a built-in reasonableness check, so these settings are usually unnecessary and will reject otherwise valid answers. Only set them when the question genuinely demands a specific length. + +```python +from rapidata import ( + RapidataClient, + FreeTextMinimumCharactersSetting, + FreeTextMaxCharactersSetting, +) + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_free_text_job_definition( + name="Caption Generation", + instruction="Describe what's happening in this image in one sentence.", + datapoints=["scene1.jpg", "scene2.jpg"], + settings=[ + FreeTextMinimumCharactersSetting(20), + FreeTextMaxCharactersSetting(200), + ], +) +job = audience.assign_job(job_def) +``` + +## Custom Audience with Full Workflow + +```python +from rapidata import RapidataClient + +client = RapidataClient() + +# Create and train audience with diverse examples. +# The admission bar is optional: target_accuracy is the fraction of qualification +# tasks a labeler must get right (server default 0.75), min_tasks how many tasks +# before that verdict is trusted (default 10), max_tasks an optional cap after +# which a verdict is forced. Supplying just one is fine — the rest fall back to +# the defaults. +audience = client.audience.create_audience( + name="Image Quality Experts", + target_accuracy=0.8, + min_tasks=12, +) + +DATAPOINTS = [ + ["good_example.jpg", "bad_example.jpg"], + ["clear.jpg", "blurry.jpg"], + ["accurate.jpg", "inaccurate.jpg"], +] +PROMPTS = [ + "A cat on a chair", + "A sunset over mountains", + "A red sports car", +] + +for prompt, datapoint in zip(PROMPTS, DATAPOINTS): + audience.add_compare_example( + instruction="Which image follows the prompt better?", + datapoint=datapoint, + truth=datapoint[0], + context=prompt, + data_type="media", + explanation="The first image better matches the prompt description.", # Shown to labelers who answer incorrectly + ) + +# Inspect the examples we just added +print(audience.get_examples()) + +# Start recruiting once the examples are added and reviewed — required and explicit. +# A job assigned before this can never receive responses (get_results() raises). +# A backend failure here raises RapidataError instead of being swallowed. +audience.start_recruiting() + +# Follow the recruiting funnel. Counts are mutually exclusive (one bucket per +# annotator) and all-zero for an audience that has not recruited anyone yet. +metrics = audience.get_recruiting_metrics() +print( + f"{metrics.graduated} graduated, {metrics.distilling} distilling, " + f"{metrics.dropped} dropped, {metrics.inactive} inactive" +) + +# Create and run job +job_def = client.job.create_compare_job_definition( + name="Production Comparison", + instruction="Which image follows the prompt better?", + datapoints=[ + ["model_a_1.jpg", "model_b_1.jpg"], + ["model_a_2.jpg", "model_b_2.jpg"], + ], + contexts=["A wizard casting a spell", "A futuristic city"], + responses_per_datapoint=20, + a_b_names=["Model A", "Model B"], +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +## Audience with Demographic Filters + +```python +from rapidata import ( + RapidataClient, CountryFilter, LanguageFilter, +) + +client = RapidataClient() +audience = client.audience.create_audience(name="US English Evaluators") +audience.update_filters([ + CountryFilter(country_codes=["US", "CA"]), + LanguageFilter(language_codes=["en"]), +]) + +# Recruiting is explicit: a custom audience recruits nobody until you add >=3 qualification +# examples AND then call start_recruiting(). A job assigned before recruiting starts can never +# receive responses (get_results() raises). So: add examples, start_recruiting, THEN assign. +EXAMPLES = [ + ("clear.jpg", ["Excellent"]), + ("decent.jpg", ["Good"]), + ("blurry.jpg", ["Poor"]), +] +for datapoint, truth in EXAMPLES: + audience.add_classification_example( + instruction="Rate the image quality", + answer_options=["Poor", "Good", "Excellent"], + datapoint=datapoint, + truth=truth, + data_type="media", + ) + +audience.start_recruiting() # required — without this the job below never gets responses + +job_def = client.job.create_classification_job_definition( + name="Image Quality (US/EN)", + instruction="Rate the image quality", + answer_options=["Poor", "Good", "Excellent"], + datapoints=["img1.jpg", "img2.jpg"], + responses_per_datapoint=10, +) +job = audience.assign_job(job_def) +job.display_progress_bar() +results = job.get_results() + +# If you DON'T need task-specific qualification, skip create_audience + examples entirely +# and use the ready-to-go global pool: client.audience.get_audience_by_id("global"). +# +# update_filters sets recruitment filters: CountryFilter / LanguageFilter (+ And/Or/Not). +# AgeFilter / GenderFilter / DeviceFilter narrow the graduates, so use them with +# audience.filter(...) (see below). +``` + +## Filtered Audience + +Derive a filtered subset of a trained audience without re-onboarding labelers: + +```python +from rapidata import RapidataClient, CountryFilter, LanguageFilter, AgeFilter, AgeGroup + +client = RapidataClient() + +base = client.audience.get_audience_by_id("audience_id") + +# .filter() narrows an audience's graduates. Beyond CountryFilter / LanguageFilter it +# accepts AgeFilter, GenderFilter and DeviceFilter, plus And/Or/Not (& | ~). +# Multiple filters in the list are ANDed. +filtered = base.filter([ + CountryFilter(["US"]), + LanguageFilter(["en"]), + AgeFilter([AgeGroup.BETWEEN_18_29]), +]) + +job_def = client.job.create_classification_job_definition( + name="US Young Adult Classification", + instruction="What product is shown?", + answer_options=["Phone", "Laptop", "Tablet"], + datapoints=["p1.jpg", "p2.jpg"], +) +job = filtered.assign_job(job_def) +job.display_progress_bar() +results = job.get_results() +``` + +## Ranking via Job Definition API + +```python +from rapidata import RapidataClient + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +job_def = client.job.create_ranking_job_definition( + name="Image Quality Ranking", + instruction="Which image looks better?", + datapoints=[["img1.jpg", "img2.jpg", "img3.jpg", "img4.jpg"]], # outer list = independent rankings + comparison_budget_per_ranking=50, + # With >10 datapoints, matchups are chosen adaptively (Elo-style) within the + # budget and random_comparisons_ratio applies. With <=10 (as here), every + # unique pair is compared with the budget spread evenly across pairs, and + # random_comparisons_ratio has no effect. + random_comparisons_ratio=0.5, +) +job = audience.assign_job(job_def) +job.view() +job.display_progress_bar() +results = job.get_results() +``` + +## Continuous Ranking Flow + +```python +from rapidata import RapidataClient + +client = RapidataClient() + +flow = client.flow.create_ranking_flow( + name="Ongoing Quality Ranking", + instruction="Which image looks better?", + max_response_threshold=200, # Aim for 200 responses per item + min_response_threshold=50, # Accept as few as 50; fewer → item marked Incomplete +) + +# Preheat for low-latency responses (call ~5 minutes before time-sensitive batches) +client.flow.preheat() + +# Submit batches over time +batch1 = flow.create_new_flow_batch( + datapoints=["gen1.jpg", "gen2.jpg", "gen3.jpg"], + context="Generated from prompt A", + time_to_live=300, +) + +result1 = batch1.get_results() # Blocks until complete +matrix1 = batch1.get_win_loss_matrix() # Pandas DataFrame +print(result1.datapoints, result1.total_votes) + +# Items sorted from best to worst (scores are Elo-style Bradley-Terry estimates) +ranked = sorted(result1.datapoints.items(), key=lambda item: item[1], reverse=True) + +# Later, submit more batches to the same flow +batch2 = flow.create_new_flow_batch( + datapoints=["gen4.jpg", "gen5.jpg", "gen6.jpg"], + context="Generated from prompt B", + time_to_live=300, +) + +# Tune the flow as you learn more +flow.update_config(instruction="Which image looks better overall?", max_responses=250) +# drain_duration / serve_timeout (seconds) are also updatable; omitted args keep current values +flow.update_config(drain_duration=30, serve_timeout=60) +``` + +## Continuous Classify Flow + +```python +from rapidata import RapidataClient, ClassifyFlowItemResult + +client = RapidataClient() + +flow = client.flow.create_classify_flow( + name="Text Detection", + instruction="Does this image contain text?", + # 2-8 answer options. A plain string is shown and returned as-is; a + # (label, value) tuple shows the label but returns the value in results. + categories=[("Yes, clearly readable", "yes"), ("No", "no")], + max_responses_per_datapoint=15, # accepted responses that close an image; collection stops once reached + min_responses_per_datapoint=10, # avg responses/image needed (once TTL ends) to be Completed vs Incomplete; >= 1 + # Time-to-live is set per batch only (see create_new_flow_batch below). +) + +# Preheat for low-latency responses (call ~5 minutes before time-sensitive batches) +client.flow.preheat() + +# Submit batches over time. Classify flows attach context per datapoint via +# contexts / context_assets (NOT the ranking-only batch-level context=). +# contexts is one text string per datapoint; context_assets is one list of asset +# paths/URLs per datapoint (list[list[str]] — a list even for a single asset). +# Both must have exactly one entry per datapoint. +batch = flow.create_new_flow_batch( + datapoints=["gen1.jpg", "gen2.jpg", "gen3.jpg"], + contexts=["Generated from prompt A", "Generated from prompt B", "Generated from prompt C"], + time_to_live=240, # seconds (up to 3600; min 60 with default flow settings); set per batch only +) + +result: ClassifyFlowItemResult = batch.get_results() # Blocks until complete +print(result.total_responses) + +# datapoints maps each asset (keyed by source URL, else original filename) to its +# classification outcome. +for asset, outcome in result.datapoints.items(): + # majority_value is the category value chosen most often, or None on a tie. + # distribution maps EVERY category value defined in the flow (in flow category order) + # to its response count, with 0 for categories nobody chose, e.g. {"yes": 0, "no": 5}. + print(asset, outcome.majority_value, outcome.distribution, outcome.response_count) + +# Only the drain duration (seconds) can be updated on a classify flow; +# instruction, categories and thresholds are fixed after creation. +flow.update_config(drain_duration=30) +``` + +## Model Benchmark (MRI) + +```python +from rapidata import RapidataClient, Tag, VoteAggregation + +client = RapidataClient() + +benchmark = client.mri.create_new_benchmark( + name="Text-to-Image Benchmark", + identifiers=["mountain", "city", "wizard"], + prompts=[ + "A serene mountain landscape", + "A futuristic city at night", + "A wise wizard portrait", + ], + # Each inner entry may mix Tag objects and bare strings; a bare string becomes + # Tag(value, category=None). + tags=[ + [Tag("outdoor", category="scene"), "nature"], + [Tag("outdoor", category="scene"), "urban"], + [Tag("portrait", category="subject")], + ], + # Per-prompt provenance: an Origin or a plain string (mapped to Origin(source)). + origins=["coco", "coco", "wikiart"], +) + +# Replace the tags of / set the origin on an already-registered prompt. +# A field left as None is not sent and stays unchanged. +benchmark.update_prompt("wizard", tags=["portrait", "fantasy"], origin="wikiart") + +# tags is the values-only view (categories dropped, kept for backwards +# compatibility); structured_tags and origins keep the full objects. All three are +# aligned by index with benchmark.prompts. +print(benchmark.tags, benchmark.structured_tags, benchmark.origins) + +leaderboard = benchmark.create_leaderboard( + name="Prompt Adherence (outdoor)", + instruction="Which image matches the description better?", + show_prompt=True, + # Scope which benchmark prompts this leaderboard collects matchups for: a prompt + # is used if it carries an included tag and no excluded one. excluded_tags always + # wins, and a non-empty included_tags drops untagged prompts. Matching is on the + # tag value only — categories are irrelevant. Applied when a run starts, and fixed + # at creation: to re-scope, create a new leaderboard. + included_tags=["outdoor"], + excluded_tags=["nsfw"], + # level_of_detail also accepts a positive integer response budget instead of a + # named level ("debug" 20, "low" 2000, "medium" 4000, "high" 8000, + # "very high" 16000). + level_of_detail=5000, + # How per-matchup annotator responses are aggregated into a matchup result: + # MAJORITY_VOTE (default) collapses each matchup to one win (ties split 0.5/0.5) + # so every matchup weighs the same; ALL_VOTES counts each response as its own + # matchup, so heavily-answered matchups dominate the standings. + vote_aggregation=VoteAggregation.ALL_VOTES, + # By default an initial run evaluates the models already in the benchmark against + # each other so the leaderboard starts with standings. Set skip_initial_run=True to + # start with no responses and no standings — models added later still compare + # against the whole existing field, and boosting still works. Create-only: it is + # applied at creation and not recorded on the leaderboard, so there's no property to + # read it back. + skip_initial_run=False, +) + +print(leaderboard.included_tags, leaderboard.excluded_tags) # copies; [] when unset +print(leaderboard.response_budget) # 5000 +print(leaderboard.level_of_detail) # "custom" — a name only on an exact budget match +print(leaderboard.vote_aggregation) # VoteAggregation.ALL_VOTES + +# name, level_of_detail, min_responses_per_matchup, and vote_aggregation are +# read-only — assigning to them raises AttributeError. Change any of them through +# update(); only the arguments you pass change, and all go out in one request. +leaderboard.update( + name="Prompt Adherence v2", + level_of_detail="high", # named level or a positive integer budget + min_responses_per_matchup=5, # must be an int >= 3 + vote_aggregation=VoteAggregation.MAJORITY_VOTE, +) + +# Evaluate models (creates, uploads, and submits in one step) +benchmark.evaluate_model( + name="DALL-E 3", + media=["dalle_mountain.png", "dalle_city.png", "dalle_wizard.png"], + prompts=["A serene mountain landscape", "A futuristic city at night", "A wise wizard portrait"], +) + +benchmark.evaluate_model( + name="Midjourney v6", + media=["mj_mountain.png", "mj_city.png", "mj_wizard.png"], + prompts=["A serene mountain landscape", "A futuristic city at night", "A wise wizard portrait"], +) + +# Pricing is separate participant metadata: add the model first, then attach the +# vendor's list price (USD per unit) so it is plotted on the benchmark's +# "Score vs. cost" chart. Skip this if you don't know the price. +midjourney = next(p for p in benchmark.participants if p.name == "Midjourney v6") +midjourney.set_price(0.04, unit="image") # "image" | "video_second" | "million_tokens" + +# Leaderboard-level results +standings = leaderboard.get_standings() +print(standings) + +# Benchmark-level aggregation across leaderboards +overall = benchmark.get_overall_standings() +matrix = benchmark.get_win_loss_matrix() +``` + +## Model Benchmark — Voter Demographics + +Standings, win/loss matrices, and both demographic read methods accept the same +optional voter-demographic filters. `country` (ISO-2 codes), `language`, and +`occupation` take plain strings; `gender` takes the `Gender` enum and `age_bucket` +the `AgeGroup` enum. `run_id` restricts to a single evaluation run. `gender` / +`age_bucket` / `occupation` are estimated (inferred); `country` / `language` are +observed. + +```python +from rapidata import RapidataClient, Gender, AgeGroup, BenchmarkDemographicDimension + +client = RapidataClient() +benchmark = client.mri.get_benchmark_by_id("benchmark_id") + +# Restrict any read to a demographic slice of the voters. +standings = benchmark.get_overall_standings( + country=["US", "CA"], + language=["en"], + gender=[Gender.FEMALE], + age_bucket=[AgeGroup.BETWEEN_18_29], +) +matrix = benchmark.get_win_loss_matrix(country=["US"]) + +# The same filters work on a single leaderboard's reads. +# lb.get_standings(country=["US"], occupation=["student"]) +# lb.get_win_loss_matrix(gender=[Gender.MALE]) + +# Who voted: one row per (dimension, value) with vote counts and shares that sum +# to 1 within each dimension. Every dimension has an "unknown" bucket for votes +# whose attribute could not be determined. +demographics = benchmark.get_demographics() +print(demographics) # columns: dimension, value, votes, share + +# Standings split by one demographic dimension of the voters. +breakdown = benchmark.get_standings_breakdown( + dimension=BenchmarkDemographicDimension.COUNTRY, +) +print(breakdown) # columns: segment, segment_votes, name, wins, total_matches, score +``` + +## Model Benchmark — Staged Submission + +```python +from rapidata import RapidataClient + +client = RapidataClient() + +benchmark = client.mri.get_benchmark_by_id("benchmark_id") + +# Add models without submitting +benchmark.add_model( + name="DALL-E 3", + media=["dalle_mountain.png", "dalle_city.png", "dalle_wizard.png"], + prompts=["A serene mountain landscape", "A futuristic city at night", "A wise wizard portrait"], +) + +benchmark.add_model( + name="Midjourney v6", + media=["mj_mountain.png", "mj_city.png", "mj_wizard.png"], + prompts=["A serene mountain landscape", "A futuristic city at night", "A wise wizard portrait"], +) + +# Update participant metadata after adding: rename, or set / clear the list price +# (USD per unit). Pricing is never part of add_model / evaluate_model. +dalle = next(p for p in benchmark.participants if p.name == "DALL-E 3") +dalle.set_price(0.04, unit="image") +print(dalle.price, dalle.price_unit) # 0.04 image +# dalle.clear_price() # -> None None; model leaves the cost chart + +# Inspect participants before submitting +for p in benchmark.participants: + print(p.name, p.status, p.price, p.price_unit) + +# Optional advisory gate: require every prompt to be filled with at least N samples. +# Only the arguments you pass change; omitted ones keep their stored value. +# min_assets_per_prompt must be an int >= 2 (bool is rejected). +benchmark.update(min_assets_per_prompt=4) + +# Submit all CREATED and SUBMITTABLE participants in a single batch request. They +# are evaluated symmetrically as one run — each model is compared against every other +# and against the benchmark's already-submitted field, rather than as separate +# per-participant runs. +# +# The gate is advisory, not a rejection: submission always completes and participants +# are marked SUBMITTED. If any submitted participant filled a prompt below the required +# samples-per-prompt, run() emits a single aggregated logger.warning listing each +# participant and its shortfall prompts, e.g. "DALL-E 3: 'mountain' (2/4)". +benchmark.run() +``` + +## Model Benchmark — Recovering a Partial Upload + +If some samples fail to upload (e.g. a flaky network), recover without re-sending +everything. `add_model` already runs an automatic recovery sweep on its own +failures, but you can also drive recovery yourself on any participant — including +ones fetched from `benchmark.participants`. + +```python +from rapidata import RapidataClient, SampleUpload + +client = RapidataClient() +benchmark = client.mri.get_benchmark_by_id("benchmark_id") + +MEDIA = ["dalle_mountain.png", "dalle_city.png", "dalle_wizard.png"] +IDENTIFIERS = ["A serene mountain landscape", "A futuristic city at night", "A wise wizard portrait"] + +participant = next(p for p in benchmark.participants if p.name == "DALL-E 3") + +# Ask the server how many samples each identifier is still short (server truth). +# Fully-uploaded identifiers are omitted, so an empty Counter means nothing is +# outstanding. +missing = participant.missing_counts(IDENTIFIERS) +if missing: + print(f"Still short: {dict(missing)}") + + # Re-send only the assets for the short identifiers. Safe to call repeatedly: + # the backend rejects samples the participant already holds, so nothing is + # duplicated, and it stops early once a round stops closing the gap. Returns + # the identifiers uploaded across all rounds and any still short on the last. + uploaded, failures = participant.retry_missing(MEDIA, IDENTIFIERS) + print(f"uploaded {len(uploaded)} across all rounds") + + # Each failure's item is a SampleUpload(media, identifier) you can re-submit + # directly; it also carries the failure reason and backend trace id. + for fu in failures: + sample: SampleUpload = fu.item + print(f"still failed: {sample} — {fu.error_message} (trace {fu.trace_id})") +``` + +## Handling Failed Uploads + +```python +from rapidata import RapidataClient +from rapidata.rapidata_client.exceptions import FailedUploadException + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +datapoints = ["valid1.jpg", "broken_url", "valid2.jpg", "missing.jpg"] + +try: + # failure_tolerance is the fraction of datapoints allowed to fail while still + # creating the definition (default 0.0 = strict). At least one datapoint must + # always upload successfully. + job_def = client.job.create_classification_job_definition( + name="With Failures", + instruction="What's in this image?", + answer_options=["Cat", "Dog"], + datapoints=datapoints, + failure_tolerance=0.1, + ) +except FailedUploadException as e: + # Outside the tolerance NO job definition is created — e.job_definition is None. + print(f"{len(e.failed_uploads)} of {len(datapoints)} failed to upload") + for reason, failed in e.failures_by_reason.items(): + print(f" {reason}: {len(failed)}") + # Remote-URL failures are also grouped by ingestion stage (download, redirect, + # content_type, decode, timeout, size, validation, internal). Only "internal" is + # a Rapidata-side fault; the rest are caller-actionable. Local-file failures have + # no stage, so this dict can be empty. + for stage, failed in e.failures_by_stage.items(): + print(f" stage {stage}: {len(failed)}") + for fu in e.detailed_failures: + print(f" - {fu.item}: {fu.error_type}: {fu.error_message} " + f"(stage={fu.stage}, http_status={fu.http_status})") + + # ...fix the failing datapoints (bad URLs, missing files, ...)... + # retry() re-uploads ONLY the failed datapoints into the SAME dataset and + # finishes creating the definition. It raises FailedUploadException again if + # failures remain outside tolerance, so it can be looped. + job_def = e.retry() + +job = audience.assign_job(job_def) +``` + +## Queueing Jobs on One Audience + +By default `assign_job` starts a job right away. Pass `run_after` to queue a job +that only starts once an earlier job **completes or fails** — this keeps a single +audience from splitting its annotators across two jobs at the same time. The queued +job is created immediately in the `Queued` state. + +```python +from rapidata import RapidataClient + +client = RapidataClient() +audience = client.audience.get_audience_by_id("global") + +first_def = client.job.create_classification_job_definition( + name="Batch 1", + instruction="What's in this image?", + answer_options=["Cat", "Dog", "Bird"], + datapoints=["img1.jpg", "img2.jpg"], +) +second_def = client.job.create_classification_job_definition( + name="Batch 2", + instruction="What's in this image?", + answer_options=["Cat", "Dog", "Bird"], + datapoints=["img3.jpg", "img4.jpg"], +) + +first = audience.assign_job(first_def) +# Starts only after `first` completes or fails. +second = audience.assign_job(second_def, run_after=first) + +# You can also queue behind a job id (e.g. from an earlier session), and chain +# further by pointing each new job at its predecessor. +third_def = client.job.create_classification_job_definition( + name="Batch 3", + instruction="What's in this image?", + answer_options=["Cat", "Dog", "Bird"], + datapoints=["img5.jpg", "img6.jpg"], +) +third = audience.assign_job(third_def, run_after=second.id) +``` + +## Updating a Job Definition's Dataset + +```python +# Keep the job definition (and its config / audience) but swap in new datapoints. +job_def.update_dataset( + datapoints=["new1.jpg", "new2.jpg", "new3.jpg"], + data_type="media", + contexts=["ctx 1", "ctx 2", "ctx 3"], +) +``` + +## Checking Job Progress Without Blocking + +`get_progress()` returns immediately with the current state, unlike `display_progress_bar()` / `get_results()`. + +```python +from rapidata import RapidataClient + +client = RapidataClient() +job = client.job.get_job_by_id("job_id") + +progress = job.get_progress() +print(f"{progress.state}: {progress.completion_percentage:.1f}% done") + +# recruiting is None for curated audiences (e.g. "global"), which don't recruit. +if progress.recruiting: + print(f"{progress.recruiting.graduated} graduated, " + f"{progress.recruiting.distilling} still distilling") + +# get_results() / display_progress_bar() raise up front if the job's audience can never +# produce responses — nobody graduated AND nobody is being recruited. An audience +# that is still distilling keeps waiting normally. +results = job.get_results() +``` + +## Estimating Job Cost Before Launch + +Check what a run is expected to cost before committing to it. `estimated_cost` is available on a job definition (before assigning it to an audience) and on a running job. + +```python +from rapidata import RapidataClient + +client = RapidataClient() +audience = client.audience.get_audience_by_id("aud_MU1GZYoESyO") + +job_def = client.job.create_compare_job_definition( + name="Example Image Prompt Alignment", + instruction="Which image matches the description better?", + datapoints=[["midjourney.jpg", "flux.jpg"]], + contexts=["A small blue book sitting on a large red book."], +) + +# Reading estimated_cost blocks briefly until the estimate is priced. +estimate = job_def.estimated_cost +print( + f"About {estimate.estimated_cost} for {estimate.required_responses} responses " + f"across {estimate.datapoint_count} datapoints" +) + +# The same property is available once the job is running. +job = audience.assign_job(job_def) +print(job.estimated_cost.estimated_cost) +``` + +## Checking Billing Spend and Remaining Credit + +Read how much the current billing period has cost so far and how much credit is +left. Billing is settled per **organization**, so these figures cover everything +the organization spent, not only the jobs this client created. + +```python +from rapidata import RapidataClient +from rapidata.rapidata_client.exceptions import RapidataError + +client = RapidataClient() + +try: + # Returns the period currently accruing cost. All amounts are in US dollars, + # rounded to the cent, and are a snapshot — fetch again for an up-to-date figure. + period = client.billing.get_current_billing_period() +except RapidataError as e: + # A period only opens once there is something to bill. + if e.status_code == 404: + print("No active billing period yet.") + raise + raise + +# outstanding_cost is gross_cost minus discount — what the period would be +# invoiced for today. status is "Open" while still accruing cost. +print( + f"[{period.status}] {period.start_date:%Y-%m-%d} → {period.end_date:%Y-%m-%d}: " + f"${period.outstanding_cost} outstanding " + f"(${period.gross_cost} gross - ${period.discount} discount) " + f"over {period.response_count} responses" +) + +# credits is the prepaid balance still available (an org-level balance that carries +# across periods), or None when billed for usage rather than from a prepaid balance. +# effective_limit is the most the org may spend this period, or None when uncapped. +if period.credits is not None: + print(f"${period.credits} credit remaining of ${period.effective_limit} granted") + +# The outstanding balance is what the organization currently owes: finalized-but-unpaid +# invoices plus the settled cost of ended periods not yet invoiced. It excludes the +# current, still-accruing period, is already net of vouchers and discounts, and is +# returned in US dollars rounded to the cent (0.0 when nothing is owed). +owed = client.billing.get_outstanding_balance() +print(f"Outstanding balance: ${owed}") +``` + +## Context Shortening + +Contexts longer than 400 characters are **always** shortened automatically at job creation time (a warning reports how many were shortened) — this cannot be turned off. Use `client.context` to shorten them yourself beforehand, or opt into shortening *every* context. + +```python +from rapidata import RapidataClient, rapidata_config + +client = RapidataClient() + +# Shorten a single context for a specific question +short = client.context.shorten_context( + context="", + question="Does the main character wear the right clothing?", +) + +# Shorten a batch of (context, question) pairs in one request +shortened = client.context.shorten_contexts([ + ("Long scene description A ...", "What is the dominant color?"), + ("Long scene description B ...", "How many people are visible?"), +]) + +# Or shorten EVERY context (not just over-long ones) at job creation time +rapidata_config.upload.contextShortening = True + +job_def = client.job.create_classification_job_definition( + name="Outfit check", + instruction="Does the main character wear the right clothing?", + answer_options=["Yes", "No"], + datapoints=["scene.jpg"], + contexts=[""], +) +``` + +## Signal (Scheduled Labeling) + +```python +from rapidata import RapidataClient + +client = RapidataClient() + +audience = client.audience.get_audience_by_id("aud_MU1GZYoESyO") + +job_def = client.job.create_compare_job_definition( + name="Prompt Alignment Job", + instruction="Which image follows the prompt more accurately?", + datapoints=[["flux_book.jpg", "mj_book.jpg"]], + contexts=["A small blue book sitting on a large red book."], +) + +# Create a signal that fires every 24 hours +signal = client.signals.create_signal( + name="Daily prompt alignment", + audience=audience, + job_definition=job_def, + interval_hours=24, +) + +# Inspect jobs created by the signal so far +for job in signal.get_jobs(page_size=10): + print(job, job.get_status()) + +# Trigger a job immediately without waiting for the schedule +signal.trigger() +job = signal.wait_for_next_job(timeout=600) +results = job.get_results() + +# Pause / resume / update +signal.pause() +signal.resume() +signal.update(name="Hourly prompt alignment", interval_hours=1) + +# Look up later +signal = client.signals.get_signal_by_id("signal_id") +signals = client.signals.find_signals(name="alignment") + +# Delete when no longer needed +signal.delete() +``` + +## Sharing a Token Across Workers (Distributed Training) + +Authenticate once in a coordinator process and share the token with many workers via a file, so thousands of workers don't all re-authenticate at the same instant. + +```python +# --- Coordinator (holds the client credentials) --- +from rapidata import RapidataClient + +coordinator = RapidataClient(leeway=300) # renew 5 min before expiry +# Writes the file now, then keeps it fresh from a background thread. +# .join() blocks forever; drop it if the coordinator also does other work. +coordinator.maintain_token_file("/shared/rapidata_token.json").join() +``` + +```python +# --- Worker (never sees the client secret) --- +from rapidata import RapidataClient + +client = RapidataClient(token_file="/shared/rapidata_token.json") +# or set RAPIDATA_TOKEN_FILE and construct with no arguments. +# The SDK re-reads the file whenever the in-memory token nears expiry. +``` + +Rolling your own file writer with `get_token()`: + +```python +import json, os, time +from rapidata import RapidataClient + +TOKEN_FILE = "/shared/rapidata_token.json" +coordinator = RapidataClient(leeway=300) +os.makedirs(os.path.dirname(TOKEN_FILE), exist_ok=True) + +def write_token(token: dict) -> None: + tmp = TOKEN_FILE + ".tmp" + with open(tmp, "w") as f: + json.dump(token, f) # keep the absolute expires_at field + os.replace(tmp, TOKEN_FILE) # atomic: workers never read a half-written file + +while True: + write_token(coordinator.get_token()) # only re-auths when near expiry + time.sleep(60) +``` + +The file is just one transport. To move the token over any transport (key-value store, RPC, secret manager, message queue), pair `get_token()` on the coordinator with `set_token()` on the worker — no shared file needed: + +```python +from rapidata import RapidataClient + +# --- Coordinator: export the current token (refreshes it first if near expiry) --- +coordinator = RapidataClient(leeway=300) +token = coordinator.get_token() +# ... distribute `token` through any transport you like ... + +# --- Worker: bootstrap directly from a token object, then renew in place --- +worker = RapidataClient(token=token) +# Later, when a fresh token arrives, inject it without reconstructing the client: +fresh_token = coordinator.get_token() # in practice received over your transport +worker.set_token(fresh_token) # used from the next request on +``` diff --git a/src/rapidata/_skill/flows-for-preference-data.md b/src/rapidata/_skill/flows-for-preference-data.md new file mode 100644 index 0000000000..8aac3ddbac --- /dev/null +++ b/src/rapidata/_skill/flows-for-preference-data.md @@ -0,0 +1,74 @@ +# Flows for Preference Data (DPO / RLHF / Best-of-N) + +Guidance for using ranking flows as the human-preference signal in a training loop: collecting DPO/RLHF pairs while a model generates, or picking the best of N candidates at inference time. The flow API itself is documented in the "Flows" section of the main guide (`python -m rapidata skill`). + +## Flows or jobs? + +- **Use a flow** when candidates are generated on the fly and each group should go to evaluation as soon as it exists: DPO data collected alongside generation, online RLHF, best-of-N. +- **Use a ranking job definition** when all candidates exist upfront and you can submit one large batch. + +Before choosing settings, answer four questions: + +1. **Turnaround:** what latency per group is desired, and what is the maximum you can tolerate? A few minutes per group is a sensible DPO target. Online RL and best-of-N need tighter bounds. +2. **Volume:** how many groups, and how often? +3. **Parallelism:** flows get faster in aggregate with more groups in flight. One flow item takes at least a minute or two, but many items running in parallel finish in about the same time as one. +4. **Bandwidth:** turnaround × volume gives the sustained responses per minute that training needs. If that rate is high, talk to Rapidata about a guaranteed-throughput setup before scaling. + +## How a flow item completes + +A flow item ends when one of these happens: + +- It reaches the flow's `max_response_threshold` (for classify flows, `max_responses_per_datapoint`). +- Its `time_to_live` (set per batch in `create_new_flow_batch`, up to 3600 s, default 240 s; at least 60 s with the default flow settings) expires. + +**Most items are expected to end by TTL.** (`min_response_threshold` defaults to the max; set it lower explicitly.) To avoid overflow, Rapidata stops handing out an item somewhere between the min and max thresholds. The item then waits for its TTL without collecting more votes. If it ended below the minimum it is `Incomplete`, but its results are still returned. + +**The consequence:** the TTL, not the max threshold, usually sets turnaround. A long TTL makes items sit idle for most of their lifetime. Set `time_to_live` close to your turnaround target (e.g. `240`–`300` for a few minutes), not to the 3600 s ceiling. + +## DPO / RLHF collection + +- **Match turnaround to generation time.** With several generator workers in parallel, one flow item's turnaround should roughly equal the time to generate one group, so generators never wait on ratings. Tune `time_to_live` to that. +- **Keep many items in flight.** Submit each group as soon as it is generated rather than batching groups up. +- **Turn results into pairs.** A ranking item's `get_results()` returns an Elo score per candidate (`FlowItemResult.datapoints`). Take the top and bottom as chosen/rejected, or use `get_win_loss_matrix()` for per-pair preference counts, which lets you drop pairs with a narrow margin. +- **Validation tasks** mixed into the flow (`validation_set_id=` on `create_ranking_flow`, built with `client.validation`) strengthen the signal. For help tuning them, contact Rapidata. + +```python +flow = client.flow.create_ranking_flow( + name="DPO preference collection", + instruction="Which image looks more realistic?", + max_response_threshold=30, + min_response_threshold=20, +) + +# Per generated group, as soon as it exists: +item = flow.create_new_flow_batch( + datapoints=candidate_paths, + time_to_live=300, +) +result = item.get_results() # blocks until the item completes +ranked = sorted(result.datapoints.items(), key=lambda kv: kv[1], reverse=True) +chosen, rejected = ranked[0][0], ranked[-1][0] +``` + +## Best-of-N + +Latency matters most here, even at the cost of efficiency: + +- **Low thresholds.** Set a small `min_response_threshold` / `max_response_threshold` so an item can finish quickly. +- **Short TTL.** Set it to the latency you can tolerate. +- **Preheat.** Call `client.flow.preheat()` ~5 minutes before a latency-sensitive burst. +- **Aggressive distribution.** Rapidata can serve an item to many more annotators than needed upfront, trading overflow for speed, and can load tasks one at a time so they are never stale when shown. Neither is exposed in the SDK; ask Rapidata to enable them. + +## Designing the comparison + +These tips apply to any Rapidata task. They matter most when the signal trains a model, because a confounded preference is learned as faithfully as a real one. + +- **Show only the context the dimension needs.** Unnecessary context is the most common confounder. + - Rating photorealism? Leave out the prompt, or ratings drift toward prompt adherence. + - Rating counting? Include only the counting part of the prompt, not the scene, colours or lighting. + - Rating text rendering? Crop to the text (e.g. a collage of the crops) so nothing else in the image competes. +- **One criterion per task.** Several criteria folded into one question reduce consistency. Split them into separate flows or leaderboards. +- **Plain, short instructions.** No jargon; labelers answer in ~25 seconds, mostly on phones. +- **Match annotators to the content.** When rating text in a script, target people who read that script. Flows take no audience or filters in the SDK: ask Rapidata to restrict a flow's annotators, or use a ranking job definition on a filtered audience (`LanguageFilter` / `CountryFilter`). +- **Check the layout with a preview** before scaling. Some content renders better with a setting (see Settings in the main guide, `python -m rapidata skill`). +- **Pilot small.** Try a few setups (instruction wording, context, thresholds) at small scale, compare agreement, then roll out the best one. diff --git a/src/rapidata/_skill/reference.md b/src/rapidata/_skill/reference.md new file mode 100644 index 0000000000..60b0a77f44 --- /dev/null +++ b/src/rapidata/_skill/reference.md @@ -0,0 +1,1519 @@ +# Rapidata SDK — Full API Reference + +## Job Definition Parameters + +The new job-definition API exposes **classification**, **comparison**, **locate**, **draw**, **select words**, **free text**, and **ranking** publicly. + +### Common parameters (classification & comparison) + +| Parameter | Type | Description | +|-----------|------|-------------| +| `name` | str | Job identifier (not shown to labelers) | +| `instruction` | str | Task description shown to labelers (max 250 characters — longer raises `ValueError`) | +| `datapoints` | list | Data to label (URLs or local paths) | +| `data_type` | `"media"` \| `"text"` | `"media"` (default, covers image/video/audio) or `"text"`. **Text assets are NOT translated** — labelers see them verbatim in their original language | +| `responses_per_datapoint` | int | Responses per item (default 10) | +| `contexts` | list[str] \| None | Text context per datapoint (max 400 characters each; contexts over the limit are always shortened against the instruction before upload — set `rapidata_config.upload.contextShortening = True` to shorten every context, or use `client.context` to shorten manually) | +| `media_contexts` | list[list[str]] \| None | Reference images per datapoint; each entry is a list of image URLs/paths (one inner list per datapoint) | +| `confidence_threshold` | float \| None | Confidence-based early stopping threshold (0-1); cannot combine with `quorum_threshold` | +| `quorum_threshold` | int \| None | Quorum-based early stopping: stop when this many responses agree; cannot combine with `confidence_threshold` | +| `settings` | `Sequence[RapidataSetting] \| None` | Display/behavior settings | +| `failure_tolerance` | float \| None | Fraction of datapoints (0.0–1.0) allowed to fail upload while the definition is still created; `None` falls back to `rapidata_config.upload.failureTolerance` (default `0.0` = strict). See Error Handling | +| `private_metadata` | `list[dict[str, str]] \| None` | Hidden metadata per datapoint | + +### Instruction length + +`instruction` is capped at 250 characters (`Workflow.MAX_INSTRUCTION_LENGTH`) for every job definition type and for every audience qualification example. Over-long values raise `ValueError: instruction is characters; maximum is 250` at construction time. + +### Classification-specific + +| Parameter | Type | Description | +|-----------|------|-------------| +| `answer_options` | list[str] | Categories to choose from | + +### Comparison-specific + +| Parameter | Type | Description | +|-----------|------|-------------| +| `datapoints` | list[list[str]] | Pairs: `[["a1.jpg","b1.jpg"], ...]` | +| `a_b_names` | list[str] \| None | Custom labels for results, e.g. `["Model A","Model B"]` | + +### Locate-specific + +Locate has no job-specific parameters — only the core parameters apply. `data_type`, `answer_options`, `a_b_names`, `confidence_threshold`, and `quorum_threshold` are not available for locate jobs. `datapoints` is `list[str]` (one item per row). + +```python +job_definition = client.job.create_locate_job_definition( + name="Artifact Detection", + instruction="Tap on any visual glitches or errors in the image.", + datapoints=["image1.jpg", "image2.jpg"], + responses_per_datapoint=35, + contexts=["Optional context"], + settings=[LocateMaxPointsSetting(5)], +) +``` + +For locate audience examples, use `audience.add_locate_example(instruction, datapoint, truths, context=None, media_context=None, explanation=None, settings=None)` where `truths` is a `list[Box]` (import `Box` from `rapidata`); coordinates are image ratios (0.0–1.0). + +### Draw-specific + +Draw has no job-specific parameters — only the core parameters apply. `data_type`, `answer_options`, `a_b_names`, `confidence_threshold`, and `quorum_threshold` are not available for draw jobs. `datapoints` is `list[str]`. + +```python +job_definition = client.job.create_draw_job_definition( + name="Object Marking", + instruction="Color in all the blue books", + datapoints=["image1.jpg", "image2.jpg"], + responses_per_datapoint=35, +) +``` + +For draw audience examples, use `audience.add_draw_example(instruction, datapoint, truths, context=None, media_context=None, explanation=None, settings=None)` where `truths` is a `list[Box]` (import `Box` from `rapidata`); coordinates are image ratios (0.0–1.0). + +### Select Words-specific + +| Parameter | Type | Description | +|-----------|------|-------------| +| `sentences` | list[str] | One sentence per datapoint, split by spaces for labelers to select words from (must have same length as `datapoints`) | + +`contexts`, `media_contexts`, `data_type`, `answer_options`, `a_b_names`, `confidence_threshold`, and `quorum_threshold` are not available for select words jobs. + +```python +job_definition = client.job.create_select_words_job_definition( + name="Prompt Alignment", + instruction="Select the words that are not depicted in the image.", + datapoints=["image1.jpg", "image2.jpg"], + sentences=["A cat on a red couch [No_mistakes]", "A blue car in the rain [No_mistakes]"], + responses_per_datapoint=15, +) +``` + +For select words audience examples, use `audience.add_select_words_example(instruction, datapoint, sentence, truths, required_precision=1, required_completeness=1, explanation=None, settings=None)` where `truths` is a `list[int]` of 0-based word indices to select. `required_precision` is the minimum share of selected words that must be correct, `required_completeness` the minimum share of correct words that must be selected (both default `1` = exact match). + +### Free Text-specific + +Free Text has no job-specific parameters — only the core parameters apply. `answer_options`, `a_b_names`, `confidence_threshold`, and `quorum_threshold` are not available. Note: free text answers cannot be graded against a ground truth; audiences cannot be trained with free text qualification examples. + +```python +job_definition = client.job.create_free_text_job_definition( + name="Prompt Collection", + instruction="What would you like to ask an AI?", + datapoints=["image1.jpg"], + responses_per_datapoint=15, +) +``` + +### Ranking (`client.job.create_ranking_job_definition`) + +| Parameter | Type | Description | +|-----------|------|-------------| +| `datapoints` | list[list[str]] | Groups: `[["img1","img2","img3"], ...]`; each inner list is one independent ranking set | +| `comparison_budget_per_ranking` | int | Total comparisons per ranking group | +| `responses_per_comparison` | int | Responses per individual comparison (default 1); replaces `responses_per_datapoint` for ranking | +| `random_comparisons_ratio` | float | Ratio of random vs targeted comparisons (0-1, default 0.5). Ignored for rankings of ≤10 datapoints (see below) | +| `data_type` | `"media"` \| `"text"` | Default `"media"` | +| `contexts` / `media_contexts` | list \| None | One entry per ranking group (not per datapoint) | + +`responses_per_datapoint`, `answer_options`, `a_b_names`, `confidence_threshold`, `quorum_threshold`, and `private_metadata` are not available for ranking jobs. + +**Matchup behavior by ranking size.** How a ranking group is compared depends on how many datapoints it holds: + +- **More than 10 datapoints:** matched adaptively (Elo-style) within `comparison_budget_per_ranking`; `random_comparisons_ratio` applies as described above. +- **10 or fewer datapoints:** every unique pair is compared, with the budget spread evenly across pairs (the total is rounded down to a multiple of the pair count; every pair is compared at least once even if the budget is smaller than the pair count). `random_comparisons_ratio` does **not** apply in this case. + +### Finding and updating job definitions and jobs + +```python +job_def = client.job.get_job_definition_by_id("job_definition_id") +job_defs = client.job.find_job_definitions(name="", amount=10, page=1) +job = client.job.get_job_by_id("job_id") +jobs = client.job.find_jobs(name="", amount=10, page=1) + +job_def.preview() # open the labeler preview in the browser +job_def.update_dataset( # replace the datapoints + datapoints=[...], data_type="media", contexts=None, media_contexts=None, + sentences=None, # select-words definitions only + private_metadata=None, +) +``` + +## Audiences + +**Goal.** An audience selects a specific group of annotators for a **specific task**. You tailor the pool to that task two ways — by training it on **qualification examples** (tasks with a known-correct answer; only labelers who answer them correctly are recruited) and/or by attaching **recruitment filters** (country, language, demographics). The point is to get the *right* annotators onto *that* task. + +A task-specific audience is meant for that task and its repeated or scheduled runs — **not** for reuse on a different, unrelated task. The qualification examples encode what "good" means for the original task; once the task changes they no longer describe the work, so reusing the audience silently loses the quality it was built for. Create a new audience per distinct task. (The `global` audience is the exception: it's the generic baseline pool for tasks that need no special qualification.) + +**Three kinds:** + +| Kind | How to get it | When to use | +|------|---------------|-------------| +| global | `client.audience.get_audience_by_id("global")` | Instant, baseline quality, no setup | +| curated | `client.audience.get_audience_by_id("aud_MU1GZYoESyO")` (alignment) | Pre-trained on a domain | +| custom | `client.audience.create_audience(name=...)` + `add_*_example(...)` | You need labelers qualified on *your* task | + +**Lifecycle / management:** + +```python +# Create / fetch / discover +audience = client.audience.create_audience( + name="Expert Evaluators", + filters=None, + # target_accuracy=0.8, # Optional: fraction of qualification tasks (0.0–1.0) a labeler must get right + # min_tasks=12, # Optional: qualification tasks before the accuracy verdict is trusted + # max_tasks=30, # Optional: cap on admission-trial tasks before a verdict is forced +) +audience = client.audience.get_audience_by_id("global") # or "aud_..." / any audience id +audiences = client.audience.find_audiences(name="", amount=10, page=1) # your audiences, newest first + +# Train a custom audience (every truth must be human-reviewed): +audience.add_classification_example(instruction=..., answer_options=[...], datapoint=..., truth=[...]) +audience.add_compare_example(instruction=..., datapoint=[...], truth=...) +audience.add_locate_example(instruction=..., datapoint=..., truths=[Box(...)]) # requires: from rapidata import Box +audience.add_draw_example(instruction=..., datapoint=..., truths=[Box(...)]) +audience.add_select_words_example(instruction=..., datapoint=..., sentence=..., truths=[1]) +df = audience.get_examples(amount=10, page=1) # inspect examples (DataFrame) + +# Start recruiting — REQUIRED and EXPLICIT for a custom audience. Recruiting begins only when +# you call this, once >=3 examples are added and reviewed. Adding examples does NOT start it; an +# audience left un-recruited stays in `Created` and a job assigned to it can never get responses +# (get_results()/display_progress_bar() raise — see "Jobs on an audience that can never respond"). +# Skip all of this and use get_audience_by_id("global") when you need no task-specific qualification. +audience.start_recruiting() # returns self; calling again is a no-op. + # A backend failure raises RapidataError — it is + # not swallowed, so recruiting never starts silently. +metrics = audience.get_recruiting_metrics() # snapshot of the recruiting funnel + +# Manage +audience.update_name("New Name") +audience.update_filters([CountryFilter(["US"]), LanguageFilter(["en"])]) # audience-supported filters only +filtered = audience.filter([CountryFilter(["US"])]) # slim subset, reuses the pool (no re-recruiting); + # a RapidataFilteredAudience only has assign_job / find_jobs +audience.delete() + +# Use +job = audience.assign_job(job_def) # start a job on the pool (after start_recruiting) +jobs = audience.find_jobs(name="", amount=10, page=1) # jobs assigned to this audience +``` + +Note: free-text answers can't be graded against a ground truth, so there is no `add_free_text_example` — custom audiences cannot be trained for free-text tasks. + +### Admission bar (`create_audience`) + +Three optional parameters set how strict qualification is: + +| Parameter | Type | Description | +|-----------|------|-------------| +| `target_accuracy` | float \| None | Fraction of qualification tasks (0.0–1.0) a labeler must answer correctly. Server default `0.75` | +| `min_tasks` | int \| None | Qualification tasks a labeler must complete before the accuracy verdict is trusted. Server default `10` | +| `max_tasks` | int \| None | Upper bound on admission-trial tasks before a verdict is forced. Default `None` (no cap) | + +Supplying only one of the three is fine — the SDK fills the others in from the defaults (`0.75` / `10`). Passing none of them sends no graduation rule at all and lets the server default apply. Client-side `ValueError`s: `target_accuracy` outside `0.0..1.0`, `min_tasks < 1`, `max_tasks < min_tasks`. + +### `RecruitingMetrics` + +`audience.get_recruiting_metrics()` returns a frozen `RecruitingMetrics` dataclass (importable from the top-level `rapidata` package) — a snapshot of the recruiting funnel. All counts are zero for audiences that have not recruited anyone and for curated audiences. + +| Field | Type | Meaning | +|-------|------|---------| +| `graduated` | int | Passed qualification, eligible to work now | +| `distilling` | int | Still going through qualification | +| `dropped` | int | Removed from the pool (score too low, limits hit, …) | +| `inactive` | int | Previously graduated/distilling, went quiet | + +Buckets are mutually exclusive — each annotator is counted exactly once. + +### Queueing jobs (`assign_job(..., run_after=...)`) + +`assign_job` takes an optional `run_after` parameter that queues a job to start only +after an earlier job finishes, so a single audience never splits its annotators across +two jobs at the same time. + +```python +def assign_job( + self, + job_definition: RapidataJobDefinition, + run_after: RapidataJob | str | None = None, +) -> RapidataJob: + ... +``` + +| Parameter | Type | Description | +|-----------|------|-------------| +| `run_after` | `RapidataJob \| str \| None` | Job (or job id) the new job must wait for. `None` (default) starts the job right away | + +When `run_after` is set, the new job is created immediately in the `Queued` state and +begins once the preceding job **completes or fails**. A `RapidataJob` contributes its +`.id`; a string is used directly as the job id. The value is sent to the API as the +`precedingJobId` field on the create-job request (`None` sends no preceding job). + +```python +# Queue behind a returned job object +first = audience.assign_job(job_def) +second = audience.assign_job(other_job_def, run_after=first) + +# Queue behind a job id (e.g. from an earlier session) +second = audience.assign_job(other_job_def, run_after="job_id") +``` + +Jobs can be chained further by pointing each new job at its predecessor. + +### Warnings on `assign_job` + +The job is always created, but three advisory warnings may be logged afterwards: + +- the estimated cost exceeds the account balance — the warning gives the estimate, the balance and the shortfall; the job runs as far as the balance allows (see "Jobs under review or out of funds"); +- the explicit-content-check skip requested via `rapidata_config.upload.checkForExplicitContent = False` was denied by the account (the check still runs); +- the audience has **no graduated annotators yet** — the warning names the audience, how many are still distilling, and the job, and points at adding examples + `start_recruiting()`, or at using the `"global"` audience. Only `RapidataAudience` emits this; filtered audiences reuse their base pool. + +## Demographic Filters + +All filters are importable from the top-level `rapidata` package. + +```python +from rapidata import ( + CountryFilter, LanguageFilter, UserScoreFilter, + AgeFilter, GenderFilter, DeviceFilter, CampaignFilter, CustomFilter, + AgeGroup, Gender, DeviceType, + NotFilter, OrFilter, AndFilter, +) + +# --- Recruitment filters on an audience: CountryFilter and LanguageFilter +# (plus the And/Or/Not combinators). UserScoreFilter/CampaignFilter/CustomFilter +# raise NotImplementedError here; demographic/device targeting belongs on .filter() (below). --- +audience.update_filters([ + CountryFilter(country_codes=["US", "CA", "GB"]), # 2-letter ISO codes (uppercased) + LanguageFilter(language_codes=["en", "fr"]), # 2-letter ISO language codes +]) + +# Combine filters with logic operators +combined = OrFilter([filter1, filter2]) +audience.update_filters([NotFilter(combined)]) + +# Derive a filtered subset of a trained audience without re-onboarding labelers. +# Supported filters for .filter(): CountryFilter, LanguageFilter, AgeFilter, +# GenderFilter, DeviceFilter (plus And/Or/Not combinators). Multiple filters are ANDed. +filtered = base_audience.filter([ + CountryFilter(["US"]), + LanguageFilter(["en"]), + AgeFilter([AgeGroup.BETWEEN_18_29]), +]) +job = filtered.assign_job(job_def) # filtered is a RapidataFilteredAudience + +# Combine filters with &, |, ~ operators +us_or_ca_not_fr = base_audience.filter([ + (CountryFilter(["US"]) | CountryFilter(["CA"])) & ~LanguageFilter(["fr"]), +]) +``` + +### Filter signatures + +| Filter | Signature | Works on audiences? | +|--------|-----------|---------------------| +| `CountryFilter` | `(country_codes: list[str])` | yes | +| `LanguageFilter` | `(language_codes: list[str])` | yes | +| `UserScoreFilter` | `(lower_bound: float = 0.0, upper_bound: float = 1.0, dimension: str \| None = None)` — bounds 0–1 | no (raises `NotImplementedError`) | +| `AgeFilter` | `(age_groups: list[AgeGroup])` | `.filter()` only | +| `GenderFilter` | `(genders: list[Gender])` | `.filter()` only | +| `DeviceFilter` | `(device_types: list[DeviceType])` | `.filter()` only | +| `CampaignFilter` | `(campaign_ids: list[str])` | no (raises `NotImplementedError`) | +| `CustomFilter` | `(identifier: str, values: list[str])` | no (raises `NotImplementedError`) | +| `NotFilter` | `(filter: RapidataFilter)` | both | +| `OrFilter` | `(filters: list[RapidataFilter])` | both | +| `AndFilter` | `(filters: list[RapidataFilter])` | both | + +Note: use `CountryFilter`, `LanguageFilter`, and the `And`/`Or`/`Not` combinators as recruitment filters (`create_audience(filters=...)` / `audience.update_filters(...)`). Target by age, gender or device with `audience.filter(...)` (deriving a filtered audience from graduates) using `AgeFilter`, `GenderFilter`, and `DeviceFilter`. There is no `DemographicFilter` class. `AgeGroup` members: `UNDER_18`, `BETWEEN_18_29`, `BETWEEN_30_39`, `BETWEEN_40_49`, `BETWEEN_50_64`, `OVER_65`; `Gender`: `MALE`, `FEMALE`, `OTHER`; `DeviceType`: `UNKNOWN`, `PHONE`, `TABLET`. `UserScoreFilter`, `CampaignFilter`, and `CustomFilter` cannot be attached to audiences at all (they raise `NotImplementedError`). + +## Results Format + +### Classification Results + +```json +{ + "results": { + "globalAggregatedData": { "Cat": 15, "Dog": 8 }, + "data": [ + { + "originalFileName": "image1.jpg", + "aggregatedResults": { "Cat": 15, "Dog": 8 }, + "summedUserScores": { "Cat": 9.5, "Dog": 4.2 }, + "confidencePerCategory": { "Cat": 0.989, "Dog": 0.011 }, + "detailedResults": [...] + } + ] + } +} +``` + +### Comparison Results + +```json +{ + "info": { "type": "Compare", "name": "Image Comparison", "instruction": "Which image is higher quality?", "version": "4.1.0" }, + "results": [ + { + "context": "A small blue book...", + "winner": "model_b.jpg", + "winnerIndex": 1, + "weightedWinner": "model_b.jpg", + "weightedWinnerIndex": 1, + "winner_index": 1, + "assetUrls": { + "model_a.jpg": "https://assets.rapidata.ai/.jpg", + "model_b.jpg": "https://assets.rapidata.ai/.jpg" + }, + "aggregatedResults": { "model_a.jpg": 3, "model_b.jpg": 5 }, + "aggregatedResultsRatios": { "model_a.jpg": 0.375, "model_b.jpg": 0.625 }, + "summedUserScores": { "model_a.jpg": 1.0, "model_b.jpg": 2.5 }, + "summedUserScoresRatios": { "model_a.jpg": 0.286, "model_b.jpg": 0.714 }, + "detailedResults": [ + { + "votedFor": "model_b.jpg", + "userDetails": { + "country": "US", "language": "en", + "userScores": { "global": 0.75 }, + "demographics": { "age": "25-34", "gender": "Female" } + } + } + ] + } + ], + "summary": { "A_wins_total": 5, "B_wins_total": 3 } +} +``` + +### Key Result Fields + +| Field | Meaning | +|-------|---------| +| `info.type` / `info.name` / `info.instruction` | Task type (e.g. `Compare`, `Classify`), the job name, and the instruction shown to labelers | +| `assetUrls` | Maps each option to the Rapidata-hosted URL of the exact file shown to labelers (random-UUID filenames; not encrypted) | +| `winner` | Most-voted option by raw vote count (`argmax` of `aggregatedResults`); `null` when nothing was voted or the top count is tied | +| `winnerIndex` | Position of `winner` in the ordered option list (`0` = first asset, `1` = second; `Both`/`Neither` appear as trailing indexes when voted); `null` under the same conditions as `winner` | +| `weightedWinner` / `weightedWinnerIndex` | Reliability-weighted winner (`argmax` of `summedUserScores`) and its index, same index space and `null`-on-tie behavior. Can differ from `winner` on close votes | +| `winner_index` | **Deprecated** alias of `winnerIndex`, kept for backwards compatibility; will be removed in a future release | +| `aggregatedResults` | Raw vote counts | +| `aggregatedResultsRatios` | Vote percentages | +| `summedUserScores` | Per-option sum of each choosing labeler's aggregated `userScore` | +| `summedUserScoresRatios` | `summedUserScores` normalized to sum to 1 | +| `confidencePerCategory` | Confidence level per category (with early stopping) | +| `userScore` | 0-1 value indicating individual labeler reliability | + +A labeler's `demographics` may be empty when no demographic data was collected for them. + +`winnerIndex`, `weightedWinner` and `weightedWinnerIndex` are only emitted by aggregator version `4.1.0`+ (`info.version`); results produced by older versions carry only `winner` and `winner_index`. + +### Working with Results + +```python +results = job.get_results() # RapidataResults — a dict subclass holding the raw JSON +df = results.to_pandas() # one row per datapoint; compare results get A_/B_ columns +df = results.to_pandas(split_details=True) # one row per individual response +results.to_json("results.json") # writes the file (default "./results.json"); returns None +``` + +Flow items have a different result shape — see the Flows section. + +## Early Stopping + +Two mutually exclusive strategies are available for Classification and Comparison jobs. You cannot set both on the same job. + +### Confidence Stopping + +Stop collecting responses once a statistical confidence threshold (weighted by labeler trust scores) is reached. + +```python +job_def = client.job.create_classification_job_definition( + name="Animal Classification", + instruction="What animal is in this image?", + answer_options=["Cat", "Dog"], + datapoints=["pet1.jpg", "pet2.jpg"], + responses_per_datapoint=50, # Maximum + confidence_threshold=0.99, # Stop at 99% confidence +) +``` + +- System calculates confidence using labeler userScores +- Stops when target confidence is reached +- Saves cost by collecting fewer responses when consensus is clear +- Best for unambiguous tasks with clear correct answers +- Not recommended for subjective preference tasks + +### Quorum Stopping + +Stop collecting responses once a fixed number of responses agree on the same answer. + +```python +job_def = client.job.create_classification_job_definition( + name="Animal Classification", + instruction="What animal is in this image?", + answer_options=["Cat", "Dog"], + datapoints=["pet1.jpg", "pet2.jpg"], + responses_per_datapoint=10, # Maximum + quorum_threshold=7, # Stop when 7 responses agree +) +``` + +A datapoint stops when: +1. `quorum_threshold` responses agree on the same answer, **OR** +2. Quorum becomes mathematically impossible (e.g. votes are too split to reach the threshold), **OR** +3. `responses_per_datapoint` total votes are collected + +- Simpler than confidence stopping — based on raw vote counts, not statistics +- Good when you want predictable cost bounds with early termination +- Best for unambiguous tasks with a clear correct answer + +## Job Progress + +`job.get_progress()` returns a frozen `JobProgress` dataclass immediately — it never blocks, unlike `get_results()` / `display_progress_bar()`. `JobProgress` is importable from the top-level `rapidata` package. + +| Field | Type | Description | +|-------|------|-------------| +| `state` | str | Same value as `job.get_status()` | +| `completion_percentage` | float | 0–100 | +| `recruiting` | `RecruitingMetrics \| None` | Recruiting funnel of the job's audience; `None` for curated audiences | + +```python +progress = job.get_progress() +print(f"{progress.state}: {progress.completion_percentage:.1f}% done") +if progress.recruiting: + print(progress.recruiting.graduated, progress.recruiting.distilling) +``` + +## Cost Estimates + +Both `RapidataJobDefinition` and `RapidataJob` expose an `estimated_cost` property returning a `CostEstimate` — an approximate estimate of what the job will cost to run to completion. Reading it from a job definition lets you check the cost of a run **before** assigning it to an audience. `CostEstimate` is importable from the top-level `rapidata` package. + +The estimate is priced shortly after the definition/job is created, so the first read blocks briefly while polling until the estimate is available (raising `TimeoutError` if it is still not ready after a few minutes), then caches the result. + +```python +job_def = client.job.create_compare_job_definition( + name="Example Image Prompt Alignment", + instruction="Which image matches the description better?", + datapoints=[["midjourney.jpg", "flux.jpg"]], + contexts=["A small blue book sitting on a large red book."], +) + +estimate = job_def.estimated_cost # blocks briefly until priced +print(f"About {estimate.estimated_cost} for {estimate.required_responses} responses") + +# The same property is available once the job is running +job = audience.assign_job(job_def) +print(job.estimated_cost.estimated_cost) +``` + +### `CostEstimate` fields + +| Field | Type | Description | +|-------|------|-------------| +| `estimated_cost` | float | Estimated total cost of running the job to completion, in your account's billing currency | +| `datapoint_count` | int | Number of datapoints the job will label | +| `required_responses` | int | Total number of responses the job collects to complete | + +This is an **estimate, not the final bill**: it is based on a sample of the job's tasks scaled to the number of responses requested, so the amount actually charged can differ. Early stopping can also lower the final cost by collecting fewer responses than the maximum. + +## Billing + +`client.billing` (a `RapidataBillingManager`, created during client construction) reads how much the current billing period has cost so far and how much credit is left. Billing is settled per **organization**, so its figures cover everything the organization spent — not only the jobs this client created. + +```python +period = client.billing.get_current_billing_period() # BillingPeriod +print(f"${period.outstanding_cost} accrued over {period.response_count} responses") +``` + +### `client.billing.get_current_billing_period() → BillingPeriod` + +Returns the billing period currently accruing cost. Raises `RapidataError` with status `404` if the organization has no active billing period (a period only opens once there is something to bill). + +### `client.billing.get_outstanding_balance() → float` + +Returns the total the organization currently owes, in US dollars rounded to the cent (`0.0` when nothing is owed). Covers finalized-but-unpaid invoices plus the settled cost of ended periods not yet invoiced; it does **not** include the current, still-accruing period. The figure is already net of vouchers and discounts, and is settled per organization. + +```python +owed = client.billing.get_outstanding_balance() # e.g. 42.50 +print(f"${owed} outstanding") +``` + +### `BillingPeriod` fields + +A frozen dataclass. All amounts are in **US dollars**, rounded to the cent. Values are a snapshot — fetch again for an up-to-date figure. `BillingPeriod` (and `RapidataBillingManager`) are importable from the top-level `rapidata` package (and re-exported from `rapidata.rapidata_client`). + +```python +from rapidata import BillingPeriod, RapidataBillingManager +``` + +| Field | Type | Description | +|-------|------|-------------| +| `id` | str | The billing period's id | +| `start_date` / `end_date` | datetime | When the period starts and ends | +| `status` | str | Lifecycle status: `"Open"` while still accruing cost; otherwise one of `"Invoiced"`, `"Void"`, `"Reconciling"`, `"PendingReview"`, `"Closed"` | +| `outstanding_cost` | float | Net cost accrued so far (`gross_cost` minus `discount`) — what the period would be invoiced for today | +| `gross_cost` | float | Cost accrued so far, before discounts | +| `discount` | float | Discounts applied to the period so far | +| `response_count` | int | Number of billable responses collected in the period | +| `credits` | `float \| None` | Prepaid credit still available, or `None` when the organization is billed for usage rather than from a prepaid balance. An organization-level balance that carries across periods | +| `effective_limit` | `float \| None` | The most the organization may spend this period, or `None` when it spends without a cap. On a prepaid plan this is the total credit granted, and `credits` is what remains of it | + +## Settings Reference + +All settings inherit from `RapidataSetting` and are importable from `rapidata`. + +Most settings only apply to specific task types. If you add a setting that the job's task type does not support, the SDK logs a non-fatal warning and still sends the flag — it is never dropped and no error is raised. Ranking jobs are treated as Compare for this check. + +| Class | Constructor | Effect | +|-------|-------------|--------| +| `NoShuffleSetting` | `(value: bool = True)` | Disable shuffling of answer options (Likert scales) | +| `AllowNeitherBothSetting` | `(delay_ms: int = 5000)` — ≥ 0 | Comparison: show an "Unsure" button (answers "Neither" / "Both") after `delay_ms` milliseconds | +| `MarkdownSetting` | `(value: bool = True)` | Render markdown in text | +| `MuteVideoSetting` | `(value: bool = True)` | Start videos muted | +| `FreeTextMinimumCharactersSetting` | `(value: int)` — must be ≥ 1 (prints a warning above 40) | Min chars for free-text tasks. Use with caution — see note below the table | +| `FreeTextMaxCharactersSetting` | `(value: int = 1024)` — must be ≥ 1 | Max chars for free-text tasks. Use with caution — see note below the table | +| `SwapContextInstructionSetting` | `(value: bool = True)` | Swap positions of context and instruction | +| `PlayPercentageVideoSetting` | `(percentage: int = 95)` — 0–95 | Require labelers to watch N% of video | +| `OriginalLanguageOnlySetting` | `(value: bool = True)` | Skip translation; show task in original language. Text assets (`data_type="text"`) are never translated, with or without this setting | +| `NoMistakeOptionSetting` | `(value: bool = True)` | Hide the "mark as mistake" option | +| `DisableAutoloopSetting` | `(value: bool = True)` | Disable automatic media looping | +| `NoInstructionDisplaySetting` | `(value: bool = True)` | Hide instruction from task screen | +| `KeyboardNumericSetting` | `(value: bool = True)` | Open numeric keyboard on mobile | +| `LocateMaxPointsSetting` | `(value: int = 3)` — ≥ 1 | Locate tasks: max points per labeler | +| `LocateMinPointsSetting` | `(value: int = 1)` — ≥ 1 | Locate tasks: min points per labeler | +| `ComparePanoramaSetting` | `(value: bool = True)` | Render comparison media as 360° panorama | +| `CompareEquirectangularSetting` | `(value: bool = True)` | Render comparison media as equirectangular VR | +| `ClassifyEquirectangularSetting` | `(value: bool = True)` | Render classification media as equirectangular 360° view | +| `CustomSetting` | `(key: str, value: str, target: "rapids" \| "campaign" = "rapids")` | Pass a custom key/value through to the backend; `target` controls whether the flag is applied at the rapid level (`"rapids"`) or campaign level (`"campaign"`) | + +**Note on `FreeTextMinimumCharactersSetting` / `FreeTextMaxCharactersSetting`:** use these with caution. Free-text responses already pass through a reasonableness check by default, so tightening the bounds is usually unnecessary and will reject otherwise valid answers. Only set them when the question genuinely demands a specific length (e.g. a single word, or a full paragraph). + +## Error Handling + +### FailedUploadException + +Job-definition creation is **atomic**: the remote definition is persisted only once the datapoint upload lands within the failure tolerance. If too many datapoints fail, **no job definition is created** and `e.job_definition` is `None` — recover with `e.retry()`, which re-uploads only the failed datapoints into the *same* dataset (never a new one) and finishes creating the definition. + +```python +from rapidata import FailedUploadException + +try: + job_def = client.job.create_classification_job_definition( + name="My Job", + instruction="...", + answer_options=[...], + datapoints=["valid.jpg", "missing.jpg", "valid2.jpg"], + failure_tolerance=0.01, # Fraction allowed to fail (default 0.0 = strict) + ) +except FailedUploadException as e: + print(f"Failed: {len(e.failed_uploads)}") + for reason, dps in e.failures_by_reason.items(): + print(f" {reason}: {len(dps)} datapoints") + for stage, dps in e.failures_by_stage.items(): + print(f" stage {stage}: {len(dps)} datapoints") + for fu in e.detailed_failures: + print(fu.item, fu.stage, fu.http_status, fu.error_message, fu.error_type) + # ...fix the failing datapoints... + job_def = e.retry() # Raises FailedUploadException again if failures remain — loopable +``` + +Tolerance behaviour: + +- Within tolerance but with some failures: the definition **is** created and a warning reports `n/total` failed and the tolerance in effect. +- Outside tolerance: nothing is created; `job_definition` is `None`. +- Regardless of tolerance, at least one datapoint must upload successfully — a definition over an empty dataset is never created. +- The failure ratio is always measured against the **original** datapoint count, so it stays meaningful across `retry()` calls. + +**Properties:** `failed_uploads` (list[Datapoint]), `detailed_failures` (list[FailedUpload[Datapoint]]), `failures_by_reason` (dict[str, list[Datapoint]]), `failures_by_stage` (dict[str, list[Datapoint]] — grouped by remote-URL ingestion stage; failures without a stage, e.g. local files, are omitted, so this can be empty), `job_definition`, `dataset`, `machine` (the creation state machine backing `retry()`). + +**`retry()`** raises `RuntimeError` when the exception did not come from job-definition creation — for those, use `dataset.add_datapoints(exception.failed_uploads)` instead. + +The exception message annotates each item with `stage=…`, `http_status=…` and `trace_id=…`, appends a `Too many open files` hint (naming `ulimit -n`, `RAPIDATA_cacheShards`, `RAPIDATA_maxWorkers`) when a failure looks like file-descriptor exhaustion, and points at `exception.retry()` whenever a creation machine is attached. + +### `FailedUpload` fields + +| Field | Type | Description | +|-------|------|-------------| +| `item` | Datapoint \| SampleUpload | The item that failed. A `Datapoint` for job uploads; a `SampleUpload` (media/identifier pair) for benchmark participant uploads (`upload_media` / `retry_missing`) | +| `error_message` / `error_type` | str | Failure reason and exception type | +| `stage` | `str \| None` | Remote-URL ingestion stage: `"download"`, `"redirect"`, `"content_type"`, `"decode"`, `"timeout"`, `"size"`, `"validation"`, `"internal"`. `None` for local-file and datapoint-creation failures | +| `http_status` | `int \| None` | Origin server's HTTP status, e.g. `403` | +| `trace_id` | `str \| None` | Backend trace id for the failure, taken from the `RapidataError` (the `x-trace-id` response header, falling back to the `traceId` in the problem+json body) | + +Only `internal` is a Rapidata-side fault — every other stage is caller-actionable. `format_error_details()` emits `Stage:` and `HTTP Status:` lines when these are present. Datapoint-level asset failures only propagate `stage` / `http_status` when all blocking asset failures agree on a single value; otherwise both are `None`. + +### `AssetWarning` + +Non-fatal advisories the backend attaches to **successful** uploads (e.g. a video longer than annotators can solve). Importable from `rapidata.rapidata_client.exceptions`. + +```python +@dataclass(frozen=True) +class AssetWarning(Generic[T]): + item: T # the asset (file path or URL) + message: str # backend advisory text, surfaced verbatim +``` + +Collected from both single-asset and batch upload paths, de-duplicated on `(item, message)`, and logged once at the end of an upload as `Upload warning for '': `. They never fail the upload. + +**Recovery docs:** https://docs.rapidata.ai/3.x/error_handling/ + +### Jobs under review or out of funds + +`assign_job` never blocks on funds: the job is always created. If its estimated cost exceeds your account balance, `assign_job` logs a warning with the estimate, your balance, and the expected shortfall — the job still runs, but may pause partway until you top up. + +Some jobs don't go straight to running. A job can enter manual review (`ManualApproval`) or, once out of funds mid-run, become spend-limited (`SpendLimited`). Neither state completes on its own, so `get_results()` raises an informative error naming the state (and the review reason, when available) instead of blocking indefinitely — top up or wait for a reviewer, then call it again. + +### Jobs on an audience that can never respond + +`get_results()` and `display_progress_bar()` also raise up front when the job's audience can never produce responses — nobody graduated **and** nobody is being recruited (recruiting was never started, or the audience is `Ready` with an empty pool). This catches the case where `start_recruiting()` was forgotten, instead of blocking forever at 0 responses. + +An audience that is still distilling, an audience in `Pending`/`Recruiting`, a curated audience, or a failed metrics read do **not** raise — those can still deliver responses. + +## Flows (Continuous Response Collection) + +Flows continuously collect human responses in small batches without full job/audience setup. Two flow types exist: **ranking flows** (`create_ranking_flow`) and **classify flows** (`create_classify_flow`), backed by two concrete subclasses of the shared base `RapidataFlow`: + +```python +from rapidata import RapidataFlow, RapidataRankingFlow, RapidataClassifyFlow +``` + +- `RapidataFlow` — shared base class; holds only the shared listing/deletion behavior (`get_flow_items`, `delete`). It does **not** expose `create_new_flow_batch` or `update_config`. +- `RapidataRankingFlow` — concrete ranking flow; adds `create_new_flow_batch` (batch-level shared context) and `update_config` (instruction, thresholds, starting Elo, drain duration, serve timeout). +- `RapidataClassifyFlow` — concrete classify flow; adds `create_new_flow_batch` (per-datapoint context) and `update_config` (drain duration only). + +Each flow and `RapidataFlowItem` carries a `flow_type` (`"ranking"` or `"simple"` — classify flows are `"simple"`, since they run on the backend's simple-flow routes), which determines the result shape and which methods are available. `get_flow_by_id` and `find_flows` populate the type from the API and return the matching subclass. Because `create_ranking_flow` → `RapidataRankingFlow`, `create_classify_flow` → `RapidataClassifyFlow`, and `get_flow_by_id` / `find_flows` return `RapidataRankingFlow | RapidataClassifyFlow`, narrow a retrieved flow (e.g. `isinstance(flow, RapidataClassifyFlow)`) before passing kind-specific batch arguments. + +### Ranking Flows + +Lightweight continuous ranking: + +```python +# Create flow +flow = client.flow.create_ranking_flow( + name="Image Quality Ranking", + instruction="Which image looks better?", + max_response_threshold=100, # Target responses per flow item (default 100) + min_response_threshold=50, # Minimum acceptable responses; item is Incomplete if TTL expires below this + # validation_set_id="...", # Optional: validation-set id to interleave validation rapids + # settings=[...], # Optional: flow-wide RapidataSettings + # drain_duration=30, # Optional: drain duration in seconds (sent as drainDurationSeconds) + # serve_timeout=60, # Optional: serve timeout in seconds (sent as serveTimeoutSeconds) +) + +# Add items to rank (RapidataRankingFlow.create_new_flow_batch — batch-level shared context) +flow_item = flow.create_new_flow_batch( + datapoints=["img1.jpg", "img2.jpg", "img3.jpg"], + context="Generated by Model X", # Optional: batch-level text context shared by all comparisons + context_assets=["reference.jpg"], # Optional: 1–10 image/video/audio paths/URLs shown alongside instruction + data_type="media", # "media" (default) or "text" + private_metadata=[...], # Optional + accept_failed_uploads=False, # If True, proceed even if some uploads fail + time_to_live=300, # Seconds until expiry (up to 3600; defaults to 4 minutes for ranking flows). + # Client-side check is 10–3600 (else ValueError("Time to live must be + # between 10 seconds and 1 hour.")); with default flow settings the minimum is 60 +) +# context (singular) and context_assets are the shared, batch-level context for all comparisons. +# Ranking batches take no per-datapoint `contexts` (TypeError: unexpected keyword argument). + +# Get results — ranking flow items return FlowItemResult, NOT RapidataResults +result = flow_item.get_results() # Blocks until completed/failed/stopped/incomplete +# result.datapoints: dict[str, int] # asset → Bradley-Terry strength estimate on Elo-scale (default start 1200); +# # keyed by source URL when provided, otherwise by original filename +# result.total_votes: int # total pairwise comparisons collected across all items + +# Items sorted from best to worst +ranked = sorted(result.datapoints.items(), key=lambda item: item[1], reverse=True) + +status = flow_item.get_status() # Non-blocking check; one of Pending, Running, Completed, + # Failed, Stopping, Stopped, Incomplete +matrix = flow_item.get_win_loss_matrix() # Pandas DataFrame (blocks until completed). Ranking flow items only — + # raises ValueError on a classify flow item +count = flow_item.get_response_count() # responses collected (waits for completion) + +# Query flow items (returned newest first; defaults: 10 per page, page 1) +items = flow.get_flow_items(amount=10, page=1) + +# Update a ranking flow after creation. Every argument is optional; drain_duration and +# serve_timeout (seconds) are sent only when not None, so omitting them keeps the current values +flow.update_config( + instruction="New instruction", + starting_elo=1000, + min_responses=40, + max_responses=120, + drain_duration=30, + serve_timeout=60, +) + +# Preheat for low-latency responses (call ~5 minutes before time-sensitive batches) +client.flow.preheat() + +# Manage flows +all_flows = client.flow.find_flows(name="", amount=10, page=1) +flow = client.flow.get_flow_by_id("flow_id") +flow.delete() + +# Delete other resources +job_def.delete() # Deletes the job definition and all its revisions +job.delete() # Deletes a running job +audience.delete() # Deletes the audience +``` + +(`preheat()`, `find_flows`, `get_flow_by_id` and `delete()` apply to both flow types.) + +### Classify Flows + +Continuously sort each datapoint of every flow item into one of the flow's categories: + +```python +flow = client.flow.create_classify_flow( + name="Text Detection", + instruction="Does this image contain text?", # question shown with every datapoint + categories=[("Yes, clearly readable", "yes"), ("No", "no")], # 2–8 options; a plain str is shown and + # returned as-is, a (label, value) tuple + # shows label but returns value + max_responses_per_datapoint=15, # default 15; accepted responses that close an image — collection for + # that image stops once reached + min_responses_per_datapoint=10, # default 10, must be >= 1; average responses per image an item needs + # (once it ends by its time to live) to be Completed rather than Incomplete. + # max_responses_per_datapoint must be >= min_responses_per_datapoint + # validation_set_id="...", # Optional: validation-set id + # settings=[...], # Optional: flow-wide RapidataSettings + # drain_duration=30, # Optional: drain duration in seconds (sent as drainDurationSeconds) + # serve_timeout=60, # Optional: serve timeout in seconds (sent as serveTimeoutSeconds) +) +# Time-to-live is controlled per batch only — set it on each create_new_flow_batch call (below). + +# Add a batch — one flow item classifying each of its datapoints into a category +# (RapidataClassifyFlow.create_new_flow_batch — per-datapoint context) +flow_item = flow.create_new_flow_batch( + datapoints=["img1.jpg", "img2.jpg"], + contexts=["Per-datapoint text", "..."], # Optional: one entry per datapoint (len must match) + context_assets=[["ref1.jpg"], ["ref2a.jpg", "ref2b.jpg"]], # Optional: one list of asset paths/URLs per datapoint + # (each entry is a list even for a single asset; len must match) + data_type="media", # "media" (default) or "text" + private_metadata=[...], # Optional + accept_failed_uploads=False, # If True, proceed even if some uploads fail + time_to_live=300, # Seconds until expiry (up to 3600; defaults to 4 minutes; + # client-side check 10–3600; with default flow settings the minimum is 60) +) +# Classify batches take no batch-level `context` (singular) and no `media_contexts` (TypeError). +# Validation: `contexts` must be a list of strings matching the datapoint count +# (else ValueError); `context_assets` must be a list of lists of strings +# (else ValueError("Context assets must be a list of lists of strings.")) with matching length +# (else ValueError("Number of context assets entries must match number of datapoints.")). + +# Get results — classify flow items return ClassifyFlowItemResult +result = flow_item.get_results() # Blocks until terminal state +# result.datapoints: dict[str, ClassifyDatapointResult] # asset identifier → outcome; keyed by source URL +# # when available, else original filename +# result.total_responses: int + +for key, dp in result.datapoints.items(): + print(key, dp.majority_value, dp.distribution, dp.response_count) + +# Update the drain duration (seconds) after creation — the only updatable classify setting. +# drain_duration=None (the default) is not sent, so the current value stays. +flow.update_config(drain_duration=30) +``` + +Validation performed before any API call: 2–8 categories (else `ValueError("Categories must contain between 2 and 8 entries.")`), unique category values, `min_responses_per_datapoint >= 1` (else `ValueError("Min responses per datapoint must be at least 1.")`), and `max_responses_per_datapoint >= min_responses_per_datapoint` (else `ValueError("Max responses per datapoint must be at least min responses per datapoint.")`). Time-to-live is set per batch on `create_new_flow_batch`, not on the flow. + +`get_win_loss_matrix()` is ranking-only; calling it on a classify flow item raises `ValueError`. `update_config()` on `RapidataClassifyFlow` accepts only `drain_duration`; the instruction, categories and response thresholds can't be changed after creation. `get_response_count()` works on both — for classify flow items it returns `total_responses`. + +A classify batch becomes `Incomplete` when its `time_to_live` expires with total responses below `min_responses_per_datapoint × number of images` (an average per image); otherwise it is `Completed` — including when every image already reached `max_responses_per_datapoint`. (For ranking flows, `Incomplete` instead occurs when `time_to_live` expires with fewer than `min_response_threshold` responses.) + +**Classify result classes** — both are frozen dataclasses importable from the top-level `rapidata` package (alongside `FlowItemResult`): + +```python +from rapidata import FlowItemResult, ClassifyFlowItemResult, ClassifyDatapointResult +``` + +`ClassifyDatapointResult` — classification outcome of a single datapoint: + +| Field | Type | Meaning | +|-------|------|---------| +| `majority_value` | `str \| None` | Category value chosen most often, or `None` on a tie | +| `distribution` | `dict[str, int]` | Category value → number of responses that chose it. Includes **every** category value defined in the flow, in the flow's category order, with `0` for categories nobody chose (e.g. `{"yes": 0, "no": 5}`); any unexpected backend values not in the blueprint are appended after the blueprint categories. Computing results makes an extra API call to fetch the flow blueprint's categories | +| `response_count` | int | Responses collected for this datapoint | + +`ClassifyFlowItemResult` — result of a classify flow item: + +| Field | Type | Meaning | +|-------|------|---------| +| `datapoints` | `dict[str, ClassifyDatapointResult]` | Asset identifier → outcome | +| `total_responses` | int | Total responses collected across the item | + +## Model Ranking Insights (MRI / Benchmarks) + +Compare and rank AI models on leaderboards. Supports images, videos, audio, and text. + +```python +# Create benchmark +benchmark = client.mri.create_new_benchmark( + name="AI Art Competition", + prompts=["A serene mountain landscape", "A futuristic city"], + # identifiers=[...], # Optional: stable ids for each prompt + # prompt_assets=[["ref1.jpg"], ["ref2.jpg"]], # Optional: one list of asset URLs/paths per prompt + # # (several entries in a list = one multi-asset), or None + # tags=[...], # Optional: per-prompt tag lists; entries may be str, Tag, or a mix + # origins=[...], # Optional: per-prompt Origin or plain source string +) + +# Add prompts later if needed (one or many, matched up by index) +benchmark.add_prompts( + prompts=["A quiet lake at dawn"], + # identifiers=["dawn_lake"], # Optional: stable id per prompt + # prompt_assets=[["ref.jpg"]], # Optional: one list of asset URLs/paths per prompt (or None) + # tags=[["landscape"]], # Optional: list of tag lists, one per prompt (str and/or Tag) + # origins=["coco"], # Optional: Origin / source string / None, one per prompt +) + +# Replace tags and/or set the origin of an already-registered prompt +benchmark.update_prompt("dawn_lake", tags=["abstract", "surreal"], origin="wikiart") + +# Create leaderboard +leaderboard = benchmark.create_leaderboard( + name="Realism", + instruction="Which image is more realistic?", + show_prompt=False, + show_prompt_asset=False, + inverse_ranking=False, + # level_of_detail="high", # "debug" | "low" | "medium" | "high" | "very high", or a positive int budget + # min_responses_per_matchup=5, + # audience_id="...", # Optional: id string, RapidataAudience, or RapidataFilteredAudience + # settings=[...], + # included_tags=["outdoor"], # Optional: only collect matchups for prompts carrying one of these tags + # excluded_tags=["nsfw"], # Optional: skip prompts carrying any of these tags (always wins) + # vote_aggregation=VoteAggregation.MAJORITY_VOTE, # VoteAggregation.MAJORITY_VOTE (default) or ALL_VOTES — how matchup votes are aggregated + # skip_initial_run=False, # Optional: when True, skip the initial run that evaluates the models already in the benchmark against each other (start with no responses/standings; later models still compare against the whole field). Create-only — not readable back +) + +# Evaluate a model (creates participant, uploads media, and submits in one step). +# Pair each media item with its prompt via prompts=[...] or identifiers=[...] (one is required). +benchmark.evaluate_model( + name="MyModel_v2", + media=["mountain.png", "city.png"], + prompts=["A serene mountain landscape", "A futuristic city"], + # identifiers=["mountain", "city"], # alternative to prompts + data_type="media", # "media" (default) or "text" +) + +# Or add a model without submitting (for more control). If any sample fails to +# upload, add_model automatically runs a recovery sweep (retry_missing) that diffs +# intended samples against server state and re-uploads only the difference; any +# still-failing samples are logged individually (first 5, then "… and N more") and +# the warning points at participant.retry_missing(...) / participant.missing_counts(...). +participant = benchmark.add_model( + name="MyModel_v3", + media=["mountain_v3.png", "city_v3.png"], + prompts=["A serene mountain landscape", "A futuristic city"], + data_type="media", +) + +# Upload additional media to the same participant. Returns +# (identifiers uploaded, failures) where failures is list[FailedUpload[SampleUpload]]. +# Raises ValueError if assets and identifiers differ in length. +uploaded, failed = participant.upload_media( + assets=["mountain_v3_extra.png"], + identifiers=["A serene mountain landscape"], + data_type="media", +) + +# Recover a partial upload — ask the server which identifiers are still short and +# re-send every asset belonging to those identifiers. Safe to call repeatedly (the +# backend rejects samples the participant already holds, so no duplication) and stops +# early once a round stops closing the gap. Works for any participant, including ones +# fetched from benchmark.participants. Raises ValueError on assets/identifiers length +# mismatch. Returns (identifiers uploaded across all rounds, failures still short on +# the last round → list[FailedUpload[SampleUpload]]). +uploaded, still_failed = participant.retry_missing( + assets=["mountain_v3.png", "city_v3.png"], + identifiers=["A serene mountain landscape", "A futuristic city"], + data_type="media", +) + +# Per-identifier count (server truth) of how many samples are still outstanding. +# Fully-uploaded identifiers are omitted, so an empty Counter means nothing is short. +missing = participant.missing_counts( + identifiers=["A serene mountain landscape", "A futuristic city"], +) # Counter[str] + +# Submit individually or all at once +participant.run() # Submit one participant (via the batch endpoint as a batch of one). + # Submission always completes; if the benchmark has a minimum-samples-per-prompt + # gate, any prompt filled below it is logged via logger.warning (advisory, not a rejection). +benchmark.run() # Submit all unsubmitted (CREATED or SUBMITTABLE) participants in a single batch request + # (chunked at 100 ids). Batching evaluates them symmetrically as one run — + # each model compared against every other and against the already-submitted + # field — rather than as separate per-participant runs. + # Emits a single aggregated logger.warning listing any participant that filled a + # prompt below the benchmark's minimum-samples-per-prompt gate (advisory; submission still completes). + +# Update benchmark configuration (only passed args change; omitted ones keep their stored value) +benchmark.update( + min_assets_per_prompt=4, # int >= 2 (bool rejected); ValueError otherwise +) + +participant.disable() # Exclude from evaluation and standings (reversible) +participant.enable() # Re-enable a previously disabled participant +participant.get_elo() # Aggregated Elo across all leaderboards (None if not yet computed) +participant.delete() # Delete participant and its uploaded media (cannot be undone) + +# Update participant metadata — adding a model and pricing it are separate calls (see "Participant pricing") +participant.rename("New Name") # Rename the participant +participant.set_price(0.04, unit="image") # List price in USD per unit ("image" | "video_second" | "million_tokens"); + # both required. ValueError on a non-positive or non-finite price, or an unknown unit +participant.clear_price() # Remove the price +participant.price # float | None — USD per price_unit +participant.price_unit # str | None — "image" | "video_second" | "million_tokens" + +# List participants and their status (p.price / p.price_unit are None for unpriced models) +for p in benchmark.participants: + print(p.id, p.name, p.status, p.price, p.price_unit) + +for lb in benchmark.leaderboards: # list[RapidataLeaderboard] + print(lb.id, lb.name) + +# Prompts — original language and English translation (aligned by index) +print(benchmark.prompts) # As originally provided +print(benchmark.english_prompts) # Server-side English translations, aligned by index +print(benchmark.identifiers) # Prompt identifiers, aligned by index +print(benchmark.structured_tags) # list[list[Tag]] — tags with categories, aligned by index +print(benchmark.origins) # list[Origin | None], aligned by index +print(benchmark.tags) # list[list[str]] — values only (categories dropped) +print(benchmark.prompt_assets) # list[list[str] | None], aligned by index — the reference asset(s) + # of each prompt. Each entry is the list of assets for that prompt + # (one element for a single asset, several for a multi-asset), or None + # for text/null prompts. This is the same shape add_prompts / + # create_new_benchmark take, so a value read back can be fed straight in + +# Get results +standings = leaderboard.get_standings() # Pandas DataFrame for one leaderboard +overall = benchmark.get_overall_standings(tags=None, leaderboard_ids=None) # Aggregated ELO across all leaderboards +matrix_lb = leaderboard.get_win_loss_matrix() # Pairwise wins/losses for one leaderboard +matrix_bm = benchmark.get_win_loss_matrix( # Pairwise wins/losses across leaderboards + tags=None, participant_ids=None, leaderboard_ids=None, use_weighted_scoring=None, +) + +# All of the read methods above (plus the two below) accept voter-demographic filters +# and run_id — see "Voter demographic filtering". +demographics = benchmark.get_demographics() # Demographic composition of the voters +breakdown = benchmark.get_standings_breakdown( # Standings split by a voter dimension + dimension=BenchmarkDemographicDimension.COUNTRY, +) + +# Access the jobs that ran for a leaderboard (one RapidataJob per run, most recent first) +for job in leaderboard.jobs: + job_results = job.get_results() + +# Update leaderboard config live — every mutable setting goes through update(); +# only the arguments passed change, all in one PATCH request. +leaderboard.update( + name="Realism (Updated)", # non-empty str + level_of_detail="very high", # Named level or a positive int response budget + min_responses_per_matchup=7, # int >= 3 (bool rejected) + vote_aggregation=VoteAggregation.MAJORITY_VOTE, # re-counts already-collected responses +) + +# Read-only leaderboard properties (mutate via update(), never by assignment — +# assigning to any of these raises AttributeError) +print(leaderboard.name) # str +print(leaderboard.level_of_detail) # Named level or "custom" +print(leaderboard.min_responses_per_matchup) +print(leaderboard.vote_aggregation) # VoteAggregation.MAJORITY_VOTE / ALL_VOTES +print(leaderboard.response_budget) # Exact budget behind level_of_detail +print(leaderboard.included_tags) # Copies; empty list when unset. Fixed at creation — +print(leaderboard.excluded_tags) # create a new leaderboard to re-scope + +# Open in browser +benchmark.view() +leaderboard.view() + +# Find existing benchmarks +benchmarks = client.mri.find_benchmarks(name="AI Art", amount=10) +benchmark = client.mri.get_benchmark_by_id("benchmark_id") +``` + +### Minimum samples per prompt (`benchmark.update`) + +A benchmark can require a minimum number of samples (assets) per prompt. Set it with `benchmark.update(min_assets_per_prompt=...)`: + +```python +def update(self, min_assets_per_prompt: int | None = None) -> None: ... +``` + +- Only the arguments you pass are changed; anything omitted keeps its stored value. +- `min_assets_per_prompt`, when provided, must be an `int` and **≥ 2** (`bool` is explicitly rejected), else `ValueError`. + +The gate is **advisory**, not a rejection. When a participant is submitted (`participant.run()` or `benchmark.run()`) with any prompt filled below the required count, the submission still completes and the participant is still marked `SUBMITTED`, but a `logger.warning` reports the shortfall. Each shortfall prompt is formatted as `'identifier' (asset_count/required)` — e.g. `model-0: 'cat' (2/4)`. `benchmark.run()` emits one aggregated warning listing every affected participant (by display name) and its shortfall prompts. + +### Participant pricing (`set_price` / `clear_price`) + +A benchmark participant can carry the model's list price so it appears on the benchmark's **"Score vs. cost"** chart. Pricing is participant metadata, like `rename`: add the model first (`add_model` / `evaluate_model` take no price arguments), then price the returned participant. + +```python +def set_price(self, price: float, unit: Literal["image", "video_second", "million_tokens"]) -> None: ... +def clear_price(self) -> None: ... + +participant.price # float | None — USD per price_unit +participant.price_unit # str | None — "image" | "video_second" | "million_tokens" +``` + +- `price` is in **USD per unit** and must be a finite number greater than 0; `unit` must be one of the three literals. Both are required — an invalid price or an unknown unit raises `ValueError`. +- `clear_price()` removes the price; `price` and `price_unit` read back `None` afterwards. +- Unpriced participants are **hidden** from the cost chart, and only participants quoted in the benchmark's **majority unit** are plotted. Set the price only when the vendor publishes a list price you are confident in; otherwise leave it unset and say so. + +### `SampleUpload` + +Importable from the top-level `rapidata` package. A frozen dataclass representing one media/identifier pair as submitted to a participant. It is the `item` carried by each `FailedUpload` returned from `upload_media` / `retry_missing`, so a caller can re-submit a failed pair directly. + +```python +from rapidata import SampleUpload + +@dataclass(frozen=True) +class SampleUpload: + media: str # media asset (local path or URL) or text content + identifier: str # the benchmark identifier/prompt the media was paired with + + def __str__(self) -> str: ... # "identifier (media)" +``` + +### `Tag` and `Origin` + +Both are importable from the top-level `rapidata` package (and from `rapidata.types`). + +```python +from rapidata import Tag, Origin + +@dataclass +class Tag: + value: str + category: str | None = None + +@dataclass +class Origin: + source: str +``` + +`tags` on `create_new_benchmark`, `add_prompts` and `update_prompt` accepts plain strings, `Tag`s, or a mix — a bare string becomes `Tag(value, category=None)`. `origins` accepts an `Origin`, a plain string (mapped to `Origin(source)`), or `None`, one per prompt. + +```python +benchmark = client.mri.create_new_benchmark( + name="Tagged Benchmark", + identifiers=["scene_1", "scene_2"], + prompts=["A sunny beach", "A car in a garage"], + tags=[ + [Tag("beach", category="scene"), "outdoor"], + [Tag("vehicle", category="object"), "indoor"], + ], + origins=["coco", "coco"], +) +``` + +`identifiers`, `prompts`, `prompt_assets`, `tags` and `origins` must all have the same length or be `None`. + +### Prompt assets shape (`prompt_assets`) + +On `create_new_benchmark` and `add_prompts`, `prompt_assets` is a **list with one entry per prompt**, each entry being a `list[str]` of image / video / audio URLs or file paths shown alongside that prompt (or `None` for no asset). A single asset is a one-element list; several entries in one list are registered together as one multi-asset. This matches the `media_contexts` shape of job definitions, and the write shape equals the read shape — a value read from `benchmark.prompt_assets` can be fed straight back in. + +```python +prompt_assets = [ + ["https://assets.rapidata.ai/prompt_1.jpg"], # single asset + ["https://example.com/pan_left.gif", "https://example.com/street.jpg"], # one multi-asset + None, # no asset (e.g. text prompt) +] +``` + +Passing a bare `str` per prompt is still accepted but **deprecated** — it is wrapped in a single-element list and logs one warning per call. Passing anything that is not a list raises `ValueError` ("Prompt assets must be a list with one entry per prompt, each a list of strings or None."); an empty list or empty-string entry also raises `ValueError`. + +### `benchmark.update_prompt(identifier, tags=None, origin=None)` + +Replaces the tags and/or sets the origin of an already-registered prompt. A field left as `None` is not sent and stays unchanged; local caches are updated in place. Raises `ValueError` if both are `None` ("Provide tags and/or origin to update."), on bad tag/origin types, or if the identifier is not registered on the benchmark. + +### Response budgets (`level_of_detail`) + +`level_of_detail` accepts a named level or a positive integer response budget. Named levels map to fixed budgets: + +| Level | Budget | +|-------|--------| +| `"debug"` | 20 | +| `"low"` | 2,000 | +| `"medium"` | 4,000 | +| `"high"` | 8,000 | +| `"very high"` | 16,000 | + +`leaderboard.response_budget` always returns the exact budget. The `level_of_detail` getter returns a named level only on an **exact** budget match and `"custom"` otherwise. Booleans are rejected; a non-positive or non-integer budget raises "Response budget must be a positive integer". Changing the budget applies to future evaluations — already-computed standings are not recomputed. + +```python +print(leaderboard.level_of_detail) # "low" +leaderboard.update(level_of_detail=5000) +print(leaderboard.level_of_detail) # "custom" +print(leaderboard.response_budget) # 5000 +``` + +### Updating a leaderboard (`leaderboard.update()`) + +`update()` is the single entry point for every mutable leaderboard setting. The properties (`name`, `level_of_detail`, `min_responses_per_matchup`, …) are read-only — assigning to them raises `AttributeError`. + +```python +def update( + self, + name: str | None = None, + level_of_detail: LevelOfDetail | int | None = None, + min_responses_per_matchup: int | None = None, + vote_aggregation: VoteAggregation | None = None, +) -> None: ... +``` + +Only the arguments you pass are changed; anything omitted keeps its stored value, and all changes go out in a single PATCH request. A no-argument `update()` sends an empty patch (it does not resend the current state). + +| Argument | Validation / effect | +|----------|---------------------| +| `name` | Non-empty string (≥ 1 char), else `ValueError` | +| `level_of_detail` | Named level (`"debug"`/`"low"`/`"medium"`/`"high"`/`"very high"`) or a positive int budget. Takes effect for future evaluations; already-computed standings are not recomputed | +| `min_responses_per_matchup` | `int` and ≥ `3`; `bool` is explicitly rejected, else `ValueError` | +| `vote_aggregation` | A `VoteAggregation` member, else `ValueError`. Because standings are derived from raw responses on every read, changing this re-counts already-collected responses (no re-evaluation needed) | + +```python +leaderboard.update(level_of_detail="high") +leaderboard.update(min_responses_per_matchup=5) +leaderboard.update(name="Realism v2") + +# Multiple fields in one request: +leaderboard.update( + name="Realism v2", + level_of_detail="high", + min_responses_per_matchup=5, + vote_aggregation=VoteAggregation.MAJORITY_VOTE, +) +``` + +### Vote aggregation (`VoteAggregation`) + +`VoteAggregation` controls how the individual annotator responses on a single matchup (one comparison of two models on one prompt) are aggregated into that matchup's result. Importable from the top-level `rapidata` package (and from `rapidata.types`). + +```python +from rapidata import VoteAggregation +``` + +| Member | Meaning | +|--------|---------| +| `VoteAggregation.MAJORITY_VOTE` | Collapses each matchup to a single win for the side the majority of responses picked, splitting ties 0.5/0.5. Every matchup weighs the same regardless of how many responses it collected. **Default.** | +| `VoteAggregation.ALL_VOTES` | Counts every individual response as its own matchup, so heavily-answered matchups dominate the standings | + +- Set at creation via `benchmark.create_leaderboard(..., vote_aggregation=...)` (defaults to `VoteAggregation.MAJORITY_VOTE`). +- Read back via the read-only `leaderboard.vote_aggregation` property (returns a `VoteAggregation`). For a leaderboard read from the benchmark's listing the value is lazily fetched on first access and cached. +- Change afterwards via `leaderboard.update(vote_aggregation=...)`. + +```python +from rapidata import VoteAggregation + +leaderboard = benchmark.create_leaderboard( + name="Realism", + instruction="Which image is more realistic?", + vote_aggregation=VoteAggregation.ALL_VOTES, +) +print(leaderboard.vote_aggregation) # VoteAggregation.ALL_VOTES +``` + +### Prompt-tag scoping (`included_tags` / `excluded_tags`) + +These restrict **which benchmark prompts the leaderboard collects matchups for**. A prompt is used when it carries at least one `included_tags` value and no `excluded_tags` value; `excluded_tags` always wins, and a non-empty `included_tags` drops untagged prompts. Matching is on the tag value only — the category is irrelevant. The filter is applied when a run starts, not snapshotted at creation, and is fixed for the life of the leaderboard. + +Distinct from `get_standings(tags=...)`, which filters what you read back rather than what gets collected. + +### Win/loss matrix + +`leaderboard.get_win_loss_matrix(tags=None, use_weighted_scoring=None)` and `benchmark.get_win_loss_matrix(tags=None, participant_ids=None, leaderboard_ids=None, use_weighted_scoring=None)` return a square pandas DataFrame indexed by participant name on both axes. Cell `[i, j]` is how often row model `i` beat column model `j`; the diagonal is always 0. + +- `tags=None` includes every matchup; `tags=[]` includes none. +- `use_weighted_scoring=True` weights each matchup by annotator reliability (`userScore`), so cells hold weighted float sums; `False` gives raw win counts; `None` uses the server-configured default. +- Both also accept the voter-demographic filters and `run_id` described below. + +### Voter demographic filtering + +Every benchmark and leaderboard read method that returns standings or a matrix accepts the same six optional voter-demographic filters plus `run_id`, restricting the result to votes cast by matching voters: + +- **Benchmark:** `get_overall_standings`, `get_win_loss_matrix`, `get_demographics`, `get_standings_breakdown`. +- **Leaderboard:** `get_standings`, `get_win_loss_matrix`. + +| Parameter | Type | Notes | +|-----------|------|-------| +| `country` | `list[str] \| None` | ISO-2 country codes; **observed** | +| `language` | `list[str] \| None` | Language codes; **observed** | +| `gender` | `list[Gender] \| None` | SDK `Gender` enum; **estimated** (inferred) | +| `age_bucket` | `list[AgeGroup] \| None` | SDK `AgeGroup` enum; **estimated** (inferred) | +| `occupation` | `list[str] \| None` | Occupation strings; **estimated** (inferred) | +| `run_id` | `str \| None` | Restrict to a single evaluation run | + +```python +from rapidata import Gender, AgeGroup, BenchmarkDemographicDimension + +# Standings from US/GB voters aged 18–29 +overall = benchmark.get_overall_standings( + country=["US", "GB"], + age_bucket=[AgeGroup.BETWEEN_18_29], +) +``` + +`gender`/`age_bucket` enum values are converted to backend values internally. + +### `benchmark.get_demographics(...)` + +Returns the demographic composition of the benchmark's voters. Accepts `tags`, `leaderboard_ids`, and the six demographic filters plus `run_id` above. The DataFrame has one row per `(dimension, bucket)`: + +| Column | Meaning | +|--------|---------| +| `dimension` | Which attribute the row describes (a `BenchmarkDemographicDimension` value) | +| `value` | The bucket within that dimension | +| `votes` | Raw vote count in the bucket | +| `share` | Fraction of the dimension's votes; shares within a dimension sum to 1 | + +Every dimension includes an `"unknown"` bucket for votes whose attribute could not be determined. + +### `benchmark.get_standings_breakdown(dimension, ...)` + +Returns standings split by a demographic dimension of the voters. `dimension` (required, first positional arg) is a `BenchmarkDemographicDimension`; the method also accepts `tags`, `leaderboard_ids`, and the six demographic filters plus `run_id`. The DataFrame has one row per `(segment, model)`: + +| Column | Meaning | +|--------|---------| +| `segment` | The voter segment within the chosen dimension (includes an `"unknown"` bucket) | +| `segment_votes` | Raw vote count for the segment | +| `name` | Model / participant name | +| `wins` | Wins for that model within the segment | +| `total_matches` | Matches the model took part in within the segment | +| `score` | Score rounded to 2 decimals, or `None` | + +### `BenchmarkDemographicDimension` + +Importable from the top-level `rapidata` package. Selects which voter attribute `get_standings_breakdown` splits on and identifies the `dimension` column of `get_demographics`. Members: `AGEBUCKET`, `GENDER`, `OCCUPATION`, `COUNTRY`, `LANGUAGE` (rendered as `AgeBucket`, `Gender`, `Occupation`, `Country`, `Language`). + +```python +from rapidata import BenchmarkDemographicDimension +``` + +## Signals (Scheduled Labeling) + +A signal runs a job definition against an audience on a repeating schedule. Each firing creates one `RapidataJob`. + +### `client.signals.create_signal` + +| Parameter | Type | Description | +|-----------|------|-------------| +| `name` | str | Human-readable name for the signal | +| `audience` | `RapidataAudience` \| str | Audience (or id string) the spawned jobs will target | +| `job_definition` | `RapidataJobDefinition` \| str | Job definition (or id string) each firing creates a job from | +| `interval_hours` | float | Hours between consecutive firings; must be positive (`ValueError` otherwise) | +| `description` | str \| None | Optional description | +| `revision_number` | int \| None | Optional: pin a specific job-definition revision; omit for "latest at fire time" | +| `is_public` | bool | Default `False`. If `True`, the signal is readable by every authenticated user in your org | + +```python +signal = client.signals.create_signal( + name="Daily prompt alignment", + audience=audience, # RapidataAudience or id string + job_definition=job_def, # RapidataJobDefinition or id string + interval_hours=24, + # revision_number=2, # Optional: pin a revision + # is_public=True, # Optional: org-wide visibility +) +``` + +### Signal methods + +| Method | Description | +|--------|-------------| +| `signal.get_jobs(page=1, page_size=20, sort_descending=True)` | List `RapidataJob` objects created by this signal (newest first by default) | +| `signal.trigger()` | Fire one job immediately; returns right away — job created asynchronously | +| `signal.wait_for_next_job(timeout=300, poll_interval=5.0)` | Block until the next firing creates its job and return it | +| `signal.pause()` | Pause the scheduler (manual `trigger()` calls still fire); returns the signal | +| `signal.resume()` | Resume a paused signal; returns the signal | +| `signal.update(name=..., description=..., interval_hours=...)` | Keyword-only; update any of name, description, cadence; returns the signal | +| `signal.delete()` | Delete the signal and all its runs | + +### Signal manager methods + +| Method | Description | +|--------|-------------| +| `client.signals.get_signal_by_id("signal_id")` | Look up a signal by id | +| `client.signals.find_signals(name="", amount=10, page=1)` | Find signals by name | + +### Signal properties + +| Property | Description | +|----------|-------------| +| `id` | Unique signal id | +| `name` / `description` | Display name and optional description | +| `audience_id` | The audience each job targets | +| `job_definition_id` | The job definition each job is created from | +| `revision_number` | Pinned revision, or `None` for "latest at fire time" | +| `interval_hours` | How often the signal fires, in hours (float) | +| `next_run_at` / `last_run_at` | Timestamps of the next and most recent firings | +| `is_paused` | Whether the scheduler is currently skipping this signal | +| `is_public` | Whether other users can discover and read it | +| `created_at` | When the signal was created | + +## Configuration + +```python +from rapidata import rapidata_config, logger, CompressionConfig + +# Logging +rapidata_config.logging.level = "INFO" # DEBUG, INFO, WARNING, ERROR, CRITICAL +rapidata_config.logging.log_file = "/path/to/log.txt" +rapidata_config.logging.silent_mode = False # also suppresses the dashboard preview link printed on job creation +rapidata_config.logging.enable_otlp = True # OpenTelemetry tracing (auto-disabled for environments without an OTLP collector — only rapidata.ai and rabbitdata.ch have one). Defaults to True, except under pytest (where it defaults to False); can also be disabled via RAPIDATA_DISABLE_OTLP=1, or forced on by passing it explicitly +rapidata_config.logging.environment = "rapidata.ai" # API environment; derives the OTLP collector host (otlp-sdk.). Set automatically by RapidataClient from its environment + +# Upload tuning +rapidata_config.upload.maxWorkers = 25 # Concurrent upload threads (warns above 200) +rapidata_config.upload.maxRetries = 3 +rapidata_config.upload.cacheToDisk = True +rapidata_config.upload.cacheTimeout = 1.0 +rapidata_config.upload.batchSize = 1000 # URLs per batch (100–5000; below 100 raises ValueError) +rapidata_config.upload.batchPollInterval = 0.5 +rapidata_config.upload.compression = CompressionConfig( + enabled=True, + quality=70, # WebP quality 1–100 + max_dimension=1024, # Max width or height in pixels (images only) +) # Optional: per-upload compression override for images and videos (None = server default) +rapidata_config.upload.contextShortening = False # When True, shorten EVERY context (over-long ones are always shortened) +rapidata_config.upload.failureTolerance = 0.0 # Fraction of a job's datapoints allowed to fail (0.0–1.0, 0.0 = strict) +rapidata_config.upload.checkForExplicitContent = None # None = account default, True = force on, False = request skip + +# Client-level maintenance +client.clear_all_caches() +client.reset_credentials() +``` + +`CompressionConfig` fields (all default `None` = defer to server): + +| Field | Type | Description | +|-------|------|-------------| +| `enabled` | `bool \| None` | Force compression on or off for **both images and videos**. `False` preserves the original image *and* video (resolution and bitrate) | +| `quality` | `int \| None` | WebP quality (1–100) when image compression runs. **Images only** | +| `max_dimension` | `int \| None` | Max width or height in pixels (≥ 1) when image compression runs. **Images only** (videos have no equivalent knob) | + +Governs compression of images **and** videos. Applies to single-asset uploads (`/asset/file` and `/asset/url`) and batched URL uploads (`/asset/batch-upload`). + +`failureTolerance` is validated to `0.0..1.0` (`ValueError` otherwise) and is overridden per call by `failure_tolerance` on `create_*_job_definition`. + +`checkForExplicitContent = False` only *requests* a skip of the server-side explicit-content check applied on job assignment — it is honored only if your account is permitted; otherwise the check still runs and `assign_job` logs a warning. + +`cacheLocation` (`~/.cache/rapidata/upload_cache`) and `cacheShards` (default 32) are immutable at runtime — don't try to assign them; set `cacheShards` via `RAPIDATA_cacheShards`. Each shard holds open file handles, and 32 comfortably covers the default `maxWorkers` of 25. + +**`OSError: [Errno 24] Too many open files`:** raise `ulimit -n`, lower `RAPIDATA_cacheShards` / `RAPIDATA_maxWorkers`, or set `cacheToDisk = False` (in-memory cache, no cache file descriptors, but you lose cross-run upload dedup). + +All config fields support environment-variable overrides with the `RAPIDATA_` prefix (e.g., `RAPIDATA_maxWorkers=10`, `RAPIDATA_failureTolerance=0.0`, `RAPIDATA_DISABLE_OTLP=1`). + +**Client authentication** is also resolved from environment variables: `RAPIDATA_CLIENT_ID` and `RAPIDATA_CLIENT_SECRET` are used before falling back to `~/.config/rapidata/credentials.json` and browser login. Without saved credentials, `RapidataClient()` blocks on a browser login for up to 5 minutes; the login URL is always printed to stderr (even in silent mode). `RAPIDATA_ENVIRONMENT` overrides the API endpoint (default: `rapidata.ai`). `RAPIDATA_TOKEN_FILE` points the client at a shared access-token file (equivalent to `token_file=`). Empty values are treated as unset and fall through to the next resolution layer. + +**Checking and establishing auth from the CLI** (`--environment` defaults to `RAPIDATA_ENVIRONMENT`, else `rapidata.ai`). Both commands check `RAPIDATA_TOKEN_FILE`, then `RAPIDATA_CLIENT_ID` + `RAPIDATA_CLIENT_SECRET`, then the credentials saved for `https://auth.{environment}`: + +```bash +python -m rapidata status [--environment ENV] # exit 0 "Authenticated for {env} via {source}."; exit 1 if not logged in. Never starts a login +python -m rapidata login [--environment ENV] # browser login; saves credentials (exit 0), or "Login did not complete..." on stderr (exit 1). + # Exits 0 immediately if already authenticated; otherwise blocks up to 5 minutes +``` + +Run `status` before the first `RapidataClient()`. If it reports not logged in, run `login` in the background (or with a tool timeout above 5 minutes) and show the user the URL it prints. Every later `RapidataClient()` reuses the saved credentials. For headless runs, set `RAPIDATA_CLIENT_ID` / `RAPIDATA_CLIENT_SECRET` instead. + +## Shared Token Files (Distributed Training) + +When Rapidata is queried from a large distributed job (e.g. hundreds or thousands of GPU workers hitting a ranking flow), don't let every worker authenticate on its own: each `RapidataClient()` exchanges the client credentials for an access token that expires ~1 hour later, so all workers re-auth in the same instant and the burst gets rate-limited. Authenticate **once** and share the token via a file that all workers can read. + +### `RapidataClient` authentication parameters + +| Parameter | Type | Description | +|-----------|------|-------------| +| `leeway` | int | Seconds before token expiry at which the SDK refreshes/re-reads the token (default 60). A coordinator typically uses a larger value, e.g. `leeway=300`, to renew well before workers need a fresh token | +| `token_file` | str | Path to a shared token file to read the access token from; the SDK re-reads it whenever the in-memory token is within `leeway` of expiry. Also settable via the `RAPIDATA_TOKEN_FILE` env var | +| `token` | dict | An access-token dict passed directly; the SDK never re-reads it, so inject a fresh one with `client.set_token(...)` (or construct a new client) once the token expires | + +### `client.maintain_token_file(path, interval=60) → threading.Thread` + +Writes the token file at `path` immediately, then keeps rewriting it atomically from a background daemon thread every `interval` seconds, creating the directory if needed. Returns the thread; `.join()` blocks the process forever (drop it if the coordinator also does other work). + +### `client.get_token() → dict` + +Returns the current access token as a dict. Cheap to call at any frequency: it only contacts the auth server once the token is within `leeway` of expiry. Use it to write the shared token file yourself (write atomically, and keep the absolute `expires_at` field so workers know when to re-read). + +### `client.set_token(token) → None` + +The counterpart to `get_token()`: replace the token a running client authenticates with, effective from its next request, without reconstructing the client. Expects the complete token object (`access_token`, `token_type`, and an absolute `expires_at` timestamp — pass what `get_token()` returned). Together, `get_token()` and `set_token()` let you move the token over any transport (key-value store, RPC, secret manager, message queue) — a **push** system (the coordinator distributes a fresh token to every worker before the old one expires, each worker applies it with `set_token`) or a **pull** system (each worker periodically fetches the current token from your own endpoint). + +```python +from rapidata import RapidataClient + +# Coordinator: holds the client credentials and keeps the file fresh +coordinator = RapidataClient(leeway=300) +coordinator.maintain_token_file("/shared/rapidata_token.json").join() + +# Worker: reads the shared token, never sees the client secret +client = RapidataClient(token_file="/shared/rapidata_token.json") + +# Any transport: export from the coordinator, inject into a worker +token = coordinator.get_token() # refreshes first if near expiry +worker = RapidataClient(token=token) # bootstrap a worker from a token object +worker.set_token(coordinator.get_token()) # renew a running worker later +``` + +## Validation Sets (`client.validation`) + +Validation sets hold tasks with known answers used to check labeler quality. Attach one to a flow by passing its id as `validation_set_id` to `create_ranking_flow` / `create_classify_flow`. For job definitions, train an audience with qualification examples instead. + +```python +vs = client.validation.create_classification_set( + name="Animal check", + instruction="What animal is in this image?", + answer_options=["Cat", "Dog"], + datapoints=["cat.jpg", "dog.jpg"], + truths=[["Cat"], ["Dog"]], # list of correct answers per datapoint + # data_type="media", contexts=None, media_contexts=None, explanations=None, dimensions=[], +) +flow = client.flow.create_classify_flow(..., validation_set_id=vs.id) +``` + +Other constructors: `create_compare_set(name, instruction, datapoints, truths: list[str], ...)`, `create_select_words_set(name, instruction, truths: list[list[int]], datapoints, sentences, required_precision=1.0, required_completeness=1.0, ...)`, `create_locate_set(...)` / `create_draw_set(...)` (`truths: list[list[Box]]`), `create_timestamp_set(...)` (`truths: list[list[tuple[int, int]]]`). Look up with `client.validation.get_validation_set_by_id(id)` and `client.validation.find_validation_sets(name="", amount=10, page=1)`. A `RapidataValidationSet` has `view()`, `delete()`, `update_dimensions(...)`, `update_should_alert(bool)`, `update_can_be_flagged(bool)`. + +## Context Management + +Datapoint contexts have a backend maximum of **400 characters** (`MAX_CONTEXT_LENGTH`). Contexts over that limit are shortened automatically before upload — this cannot be turned off. + +`ContextManager` is importable from the top-level `rapidata` package and is exposed as `client.context` on every `RapidataClient` instance. + +### `client.context.shorten_context(context, question) → str` + +Shorten a single context for the given question. Results are cached server-side. + +### `client.context.shorten_contexts(pairs) → list[str]` + +Shorten a batch of `(context, question)` pairs. Returns shortened contexts in the same order as `pairs`. Batches of ≤10 pairs go out as a single request; larger batches are split into chunks of 10 and sent concurrently using `rapidata_config.upload.maxWorkers`, with a `Shortening contexts` progress bar (suppressed by `rapidata_config.logging.silent_mode`). + +```python +# Single context +short = client.context.shorten_context( + context="", + question="Does the main character wear the right clothing?", +) + +# Batch +shortened = client.context.shorten_contexts([ + (context_a, question_a), + (context_b, question_b), +]) +``` + +### Automatic shortening at job creation + +Any context exceeding 400 characters is **always** shortened against the task instruction before upload — there is no way to disable this. A warning reports how many contexts were shortened, and per-context before/after lengths are logged at info level. If shortening returns an empty result the original context is kept and a warning is logged. + +Set `rapidata_config.upload.contextShortening = True` (default `False`) to shorten **every** context, not just over-long ones. + +```python +from rapidata import rapidata_config +rapidata_config.upload.contextShortening = True +``` + +## Human Prompting Best Practices + +- **Be concise** — labelers have ~25 seconds per task +- **Positive framing** — "Which is more realistic?" not "Which is less AI-generated?" +- **Distinct options** — "Poor / Acceptable / Excellent" not "Bad / Not Good / Fine / Good / Great" +- **Single criterion per task** — "What animal is in the image?" not "Does this contain a rabbit, dog, or cat?" +- **Use `NoShuffleSetting()` for scales** — always for Likert or ordered answer options diff --git a/tests/conftest.py b/tests/conftest.py index e8609ad4d3..1178d6e225 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -37,7 +37,7 @@ @pytest.fixture def agent_sandbox(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> Path: - """An empty home and project with no agent env, no hint state and no network.""" + """An empty home and project with no agent env and no hint state.""" for var in _AGENT_VARS: monkeypatch.delenv(var, raising=False) home, project = tmp_path / "home", tmp_path / "project" @@ -49,6 +49,6 @@ def agent_sandbox(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> Path: _agent_hint, "STATE_FILE", home / ".config/rapidata/agent-state.json" ) monkeypatch.setattr(_agent_hint, "FALLBACK_STATE_FILE", tmp_path / "tmp-state.json") - monkeypatch.setattr(_agent_hint, "_fetch_live_digest", lambda: None) monkeypatch.setattr(sys, "orig_argv", ["python", "-c", "import rapidata"]) + monkeypatch.setattr(sys, "argv", ["-c"]) return project diff --git a/tests/test_agent_hint.py b/tests/test_agent_hint.py index 231cd41d67..512c7af5c1 100644 --- a/tests/test_agent_hint.py +++ b/tests/test_agent_hint.py @@ -9,15 +9,14 @@ import pytest -from rapidata import _agent_hint +from rapidata import __version__, _agent_hint from rapidata._agent_hint import ( AGENT_HINT, agent_hint, detected_coding_agent, + installed_version, mark_skill_read, - record_live_skill, running_under_coding_agent, - skill_digest, stamp_skill, ) @@ -174,78 +173,70 @@ def test_an_unrelated_agents_md_does_not_count_as_installed( @pytest.mark.parametrize("where", ["project", "home"]) -def test_silent_when_an_installed_copy_is_current( +def test_silent_when_an_installed_copy_matches_this_version( monkeypatch: pytest.MonkeyPatch, sandbox: Path, where: str ): _agent(monkeypatch) _install(sandbox if where == "project" else Path.home(), stamp_skill(SKILL)) - record_live_skill(SKILL) assert agent_hint() is None -def test_stale_installed_copy_asks_for_a_reinstall( +def test_a_copy_from_another_version_asks_for_a_reinstall( monkeypatch: pytest.MonkeyPatch, sandbox: Path ): _agent(monkeypatch) - path = _install(sandbox, stamp_skill(SKILL)) - record_live_skill(SKILL + "new gotcha\n") + path = _install(sandbox, stamp_skill(SKILL, version="3.0.0")) + mark_skill_read() hint = agent_hint() assert hint is not None and str(path) in hint + assert "3.0.0" in hint and __version__ in hint assert hint.endswith("python -m rapidata skill --install") -def test_stale_user_level_copy_names_its_dir(monkeypatch: pytest.MonkeyPatch): +def test_outdated_user_level_copy_names_its_dir(monkeypatch: pytest.MonkeyPatch): _agent(monkeypatch) - _install(Path.home(), stamp_skill(SKILL), ".codex/skills/rapidata/SKILL.md") - record_live_skill(SKILL + "new\n") + _install( + Path.home(), + stamp_skill(SKILL, version="3.0.0"), + ".codex/skills/rapidata/SKILL.md", + ) hint = agent_hint() assert hint is not None assert hint.endswith(f"--install --agent codex --dir {Path.home()}") -def test_unstamped_copy_from_an_older_install_is_compared_verbatim( +def test_an_unstamped_copy_does_not_count_as_installed( monkeypatch: pytest.MonkeyPatch, sandbox: Path ): _agent(monkeypatch) _install(sandbox, SKILL) - record_live_skill(SKILL) - assert agent_hint() is None - record_live_skill(SKILL + "new\n") - assert agent_hint() is not None - - -def test_freshness_is_checked_at_most_once_a_day( - monkeypatch: pytest.MonkeyPatch, sandbox: Path -): - _agent(monkeypatch) - _install(sandbox, stamp_skill(SKILL)) - calls: list[int] = [] - - def fetch() -> str: - calls.append(1) - return skill_digest(SKILL) - - monkeypatch.setattr(_agent_hint, "_fetch_live_digest", fetch) - assert agent_hint() is None - assert agent_hint() is None - assert len(calls) == 1 - _after(monkeypatch, _agent_hint.FRESHNESS_TTL + 1) - agent_hint() - assert len(calls) == 2 + assert agent_hint() == AGENT_HINT -def test_offline_freshness_check_stays_silent( - monkeypatch: pytest.MonkeyPatch, sandbox: Path -): +def test_silent_while_running_the_console_script(monkeypatch: pytest.MonkeyPatch): _agent(monkeypatch) - _install(sandbox, stamp_skill(SKILL)) + monkeypatch.setattr(sys, "orig_argv", ["python", "/venv/bin/rapidata", "skill"]) + monkeypatch.setattr(sys, "argv", ["/venv/bin/rapidata", "skill"]) assert agent_hint() is None def test_stamp_goes_after_the_front_matter(): stamped = stamp_skill(SKILL) assert stamped.startswith("---\nname: rapidata\n") - assert f"sha256={skill_digest(SKILL)}" in stamped.split("---\n")[2] + assert f"version={__version__}" in stamped.split("---\n")[2] + assert installed_version(stamped) == __version__ + + +def test_hint_needs_no_network(monkeypatch: pytest.MonkeyPatch, sandbox: Path): + import socket + + def no_network(*args, **kwargs): + raise AssertionError("the import hint must not touch the network") + + monkeypatch.setattr(socket, "create_connection", no_network) + _agent(monkeypatch) + _install(sandbox, stamp_skill(SKILL, version="3.0.0")) + assert agent_hint() is not None def test_import_prints_the_hint_to_stderr(tmp_path: Path): diff --git a/tests/test_main.py b/tests/test_main.py index 7125e4988d..e495fe564b 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -1,15 +1,12 @@ from __future__ import annotations +import os from pathlib import Path import pytest -import requests from rapidata import __main__ as cli -from rapidata import _agent_hint -from rapidata._agent_hint import skill_digest - -SKILL = "---\nname: rapidata\n---\nguide" +from rapidata import __version__, _agent_hint @pytest.fixture(autouse=True) @@ -17,60 +14,75 @@ def sandbox(agent_sandbox: Path) -> Path: return agent_sandbox -@pytest.fixture -def skill(monkeypatch: pytest.MonkeyPatch) -> str: - monkeypatch.setattr(cli, "fetch_skill", lambda: SKILL) - return SKILL +def test_skill_prints_the_bundled_guide(capsys: pytest.CaptureFixture[str]): + assert cli.main(["skill"]) == 0 + out = capsys.readouterr().out + assert out.startswith("---\nname: rapidata\n") + assert "python -m rapidata skill reference" in out -@pytest.fixture -def offline(monkeypatch: pytest.MonkeyPatch) -> None: - def boom() -> str: - raise requests.ConnectionError("offline") +@pytest.mark.parametrize( + "guide", ["reference", "examples", "flows-for-preference-data"] +) +def test_skill_prints_each_companion_guide( + guide: str, capsys: pytest.CaptureFixture[str] +): + assert cli.main(["skill", guide]) == 0 + assert capsys.readouterr().out.startswith("# ") + - monkeypatch.setattr(cli, "fetch_skill", boom) +def test_skill_does_not_touch_the_network(monkeypatch: pytest.MonkeyPatch): + import socket + def no_network(*args, **kwargs): + raise AssertionError("`rapidata skill` must not touch the network") -def test_skill_prints_the_guide(skill: str, capsys: pytest.CaptureFixture[str]): + monkeypatch.setattr(socket, "create_connection", no_network) assert cli.main(["skill"]) == 0 - assert capsys.readouterr().out.strip() == skill -def test_skill_install_writes_a_stamped_copy_to_the_claude_path( - skill: str, tmp_path: Path +def test_skill_survives_a_closed_pipe(tmp_path: Path): + import subprocess + import sys + + result = subprocess.run( + f'"{sys.executable}" -m rapidata skill | head -1', + shell=True, + cwd=tmp_path, + env={**os.environ, "HOME": str(tmp_path), "RAPIDATA_AGENT_HINT": "0"}, + capture_output=True, + text=True, + ) + assert result.stdout == "---\n" + assert "BrokenPipeError" not in result.stderr + + +def test_skill_install_writes_a_version_stamped_copy_to_the_claude_path( + tmp_path: Path, ): assert cli.main(["skill", "--install", "--dir", str(tmp_path)]) == 0 installed = (tmp_path / ".claude/skills/rapidata/SKILL.md").read_text() - assert installed.startswith("---\nname: rapidata\n---\n") - assert f"sha256={skill_digest(skill)}" in installed - assert installed.endswith("guide") + assert installed.startswith("---\nname: rapidata\n") + assert f"" in installed + assert ( + installed.replace(f"\n", "") + == cli.bundled_skill() + ) -def test_skill_install_honours_agent(skill: str, tmp_path: Path): +def test_skill_install_honours_agent(tmp_path: Path): assert ( cli.main(["skill", "--install", "--agent", "generic", "--dir", str(tmp_path)]) == 0 ) - assert (tmp_path / "AGENTS.md").read_text().endswith("guide") - + assert _agent_hint.installed_version((tmp_path / "AGENTS.md").read_text()) == ( + __version__ + ) -def test_offline_falls_back_to_the_bundled_copy( - offline: None, capsys: pytest.CaptureFixture[str] -): - assert cli.main(["skill"]) == 0 - out = capsys.readouterr() - assert out.out.startswith("---\nname: rapidata\n") - assert "bundled" in out.err - -def test_offline_without_a_bundled_copy_points_at_the_online_copy( - offline: None, - monkeypatch: pytest.MonkeyPatch, - capsys: pytest.CaptureFixture[str], -): - monkeypatch.setattr(cli, "bundled_skill", lambda: None) - assert cli.main(["skill"]) == 1 - assert "llms-full.txt" in capsys.readouterr().err +def test_install_rejects_a_companion_guide(tmp_path: Path): + with pytest.raises(SystemExit): + cli.main(["skill", "reference", "--install", "--dir", str(tmp_path)]) def test_no_command_prints_help(capsys: pytest.CaptureFixture[str]): @@ -78,9 +90,7 @@ def test_no_command_prints_help(capsys: pytest.CaptureFixture[str]): assert "skill" in capsys.readouterr().out -def test_reading_the_skill_marks_this_session( - skill: str, monkeypatch: pytest.MonkeyPatch -): +def test_reading_the_skill_marks_this_session(monkeypatch: pytest.MonkeyPatch): monkeypatch.setenv("CLAUDECODE", "1") monkeypatch.setenv("CLAUDE_CODE_SESSION_ID", "s1") assert _agent_hint.agent_hint() is not None @@ -88,9 +98,13 @@ def test_reading_the_skill_marks_this_session( assert _agent_hint.agent_hint() is None -def test_reading_the_skill_records_the_live_digest(skill: str): - assert cli.main(["skill"]) == 0 - assert _agent_hint._load_state()["live_sha"] == skill_digest(skill) +def test_reading_a_companion_guide_does_not_mark_the_session( + monkeypatch: pytest.MonkeyPatch, +): + monkeypatch.setenv("CLAUDECODE", "1") + monkeypatch.setenv("CLAUDE_CODE_SESSION_ID", "s1") + assert cli.main(["skill", "examples"]) == 0 + assert _agent_hint.agent_hint() is not None @pytest.fixture From da88509bc30fcc6d38612f66705d99132b49a887 Mon Sep 17 00:00:00 2001 From: RapidPoseidon Date: Mon, 28 Sep 2026 15:35:05 +0000 Subject: [PATCH 3/7] feat(agents): record the SDK version in the installed skill's front matter `--install` now writes `metadata.rapidata-sdk-version` into the front matter and opens the copy with a version check, so an agent that loads the installed skill without importing the SDK is still told to reinstall on a mismatch. The import hint reads the version from the front matter only. Co-Authored-By: Claude Opus 5.5 Co-Authored-By: lino@rapidata.ai <68745352+LinoGiger@users.noreply.github.com> --- docs/ai_agents.md | 2 +- src/rapidata/__main__.py | 3 ++- src/rapidata/_agent_hint.py | 52 +++++++++++++++++++++++++++---------- tests/test_agent_hint.py | 18 ++++++++++--- tests/test_main.py | 7 ++--- 5 files changed, 58 insertions(+), 24 deletions(-) diff --git a/docs/ai_agents.md b/docs/ai_agents.md index 574c572eb6..5b9f346b7b 100644 --- a/docs/ai_agents.md +++ b/docs/ai_agents.md @@ -80,7 +80,7 @@ pip install -U rapidata # or: uv add -U rapidata The installed skill from the table above only points the agent at `python -m rapidata skill`, so it rarely needs an update. Pull one anyway with `claude plugin marketplace update` (Claude Code) or `npx skills update rapidata` (everything else). -Copies written by `rapidata skill --install` carry the SDK version that wrote them. After an SDK upgrade, the next import by a coding agent tells it to run `rapidata skill --install` again. This check is local and needs no network. +Copies written by `rapidata skill --install` record the SDK version that wrote them in their front matter (`metadata.rapidata-sdk-version`) and open with a short check. That check tells the agent to compare the recorded version with `rapidata.__version__` and reinstall on a mismatch. After an SDK upgrade, the next import by a coding agent also prints the reinstall command. Both checks are local and need no network. ## Editing the skill diff --git a/src/rapidata/__main__.py b/src/rapidata/__main__.py index 18849e30eb..c031146b32 100644 --- a/src/rapidata/__main__.py +++ b/src/rapidata/__main__.py @@ -46,7 +46,8 @@ def bundled_skill(guide: str = "main") -> str: def install_skill(root: Path, agent: str, content: str) -> Path: target = root / SKILL_INSTALL_PATHS[agent] target.parent.mkdir(parents=True, exist_ok=True) - target.write_text(stamp_skill(content), encoding="utf-8") + user_level = root.expanduser().resolve() == Path.home().resolve() + target.write_text(stamp_skill(content, agent, user_level), encoding="utf-8") return target diff --git a/src/rapidata/_agent_hint.py b/src/rapidata/_agent_hint.py index de4d8e11c7..52bce668d9 100644 --- a/src/rapidata/_agent_hint.py +++ b/src/rapidata/_agent_hint.py @@ -10,7 +10,7 @@ expires after :data:`ANON_READ_TTL`. - Never while the Claude Code plugin is installed, or while a copy written by ``python -m rapidata skill --install`` (project or user level) carries this - SDK's version stamp. A copy stamped with another version asks for a + SDK's version in its front matter. A copy from another version asks for a reinstall instead. Nothing here touches the network. ``RAPIDATA_AGENT_HINT=0`` switches it off for processes an agent merely started. @@ -65,7 +65,7 @@ _SESSION_ENV_VARS = ("CLAUDE_CODE_SESSION_ID", "CODEX_THREAD_ID", "CODEX_SESSION_ID") -_STAMP_RE = re.compile(r"\n") +_STAMP_RE = re.compile(r'^ rapidata-sdk-version: "([^"]+)"$', re.MULTILINE) AGENT_HINT = ( "rapidata: coding agent detected. Read the SDK guide for this version before exploring " @@ -101,20 +101,48 @@ def _sdk_version() -> str: return __version__ -def stamp_skill(content: str, version: str | None = None) -> str: - """Return ``content`` with a version line after its front matter, so a copy made by another SDK version can be told apart.""" - stamp = f"\n" +def _install_command(agent: str, directory: str | None = None) -> str: + cmd = "python -m rapidata skill --install" + if agent != "claude": + cmd += f" --agent {agent}" + if directory: + cmd += f" --dir {directory}" + return cmd + + +def stamp_skill( + content: str, + agent: str = "claude", + user_level: bool = False, + version: str | None = None, +) -> str: + """Return ``content`` with the SDK version in its front matter and a version check as its first lines. + + The front matter field is what :func:`installed_version` reads on import; + the check lets an agent that loads the copy without importing the SDK + notice that it drifted from the installed version. + """ + version = version or _sdk_version() + where = "" if user_level else " from the project root" + cmd = _install_command(agent, "~" if user_level else None) + check = ( + f"> **Installed copy, written by rapidata {version}.** Before relying on it, run\n" + f'> `python -c "import rapidata; print(rapidata.__version__)"`. If that does not print\n' + f"> `{version}`, run `{cmd}`{where} to update this file, then read it again.\n" + ) + field = f'metadata:\n rapidata-sdk-version: "{version}"\n' if content.startswith("---\n"): end = content.find("\n---\n", 4) if end != -1: cut = end + len("\n---\n") - return content[:cut] + stamp + content[cut:] - return stamp + content + return content[: end + 1] + field + "---\n" + check + content[cut:] + return f"---\n{field}---\n{check}{content}" def installed_version(text: str) -> str | None: - """SDK version an ``--install``ed copy was written by, or None when ``text`` carries no stamp.""" - match = _STAMP_RE.search(text) + """SDK version an ``--install``ed copy was written by, or None when its front matter carries no stamp.""" + end = text.find("\n---\n", 4) if text.startswith("---\n") else -1 + match = _STAMP_RE.search(text[:end]) if end != -1 else None return match.group(1) if match else None @@ -204,11 +232,7 @@ def installed_copies(root: Path | None = None) -> list[tuple[str, Path, Path, st def _stale_hint(agent: str, base: Path, path: Path, version: str) -> str: - cmd = "python -m rapidata skill --install" - if agent != "claude": - cmd += f" --agent {agent}" - if base != Path.cwd(): - cmd += f" --dir {base}" + cmd = _install_command(agent, None if base == Path.cwd() else str(base)) return ( f"rapidata: the Rapidata skill at {path} was installed by rapidata {version}, " f"but {_sdk_version()} is installed. Update it with: {cmd}" diff --git a/tests/test_agent_hint.py b/tests/test_agent_hint.py index 512c7af5c1..8375bb8ef6 100644 --- a/tests/test_agent_hint.py +++ b/tests/test_agent_hint.py @@ -220,11 +220,23 @@ def test_silent_while_running_the_console_script(monkeypatch: pytest.MonkeyPatch assert agent_hint() is None -def test_stamp_goes_after_the_front_matter(): +def test_stamp_puts_the_version_in_the_front_matter(): stamped = stamp_skill(SKILL) - assert stamped.startswith("---\nname: rapidata\n") - assert f"version={__version__}" in stamped.split("---\n")[2] + front_matter, body = stamped.split("---\n")[1:3] + assert front_matter.startswith("name: rapidata\n") + assert f'rapidata-sdk-version: "{__version__}"' in front_matter assert installed_version(stamped) == __version__ + assert body.startswith(f"> **Installed copy, written by rapidata {__version__}.**") + + +def test_stamp_tells_the_agent_how_to_update_the_copy(): + body = stamp_skill(SKILL, agent="codex", user_level=True).split("---\n")[2] + assert "rapidata.__version__" in body + assert "python -m rapidata skill --install --agent codex --dir ~" in body + + +def test_a_stamp_in_the_body_is_not_read_as_the_version(): + assert installed_version(SKILL + ' rapidata-sdk-version: "1.0"\n') is None def test_hint_needs_no_network(monkeypatch: pytest.MonkeyPatch, sandbox: Path): diff --git a/tests/test_main.py b/tests/test_main.py index e495fe564b..b375b6edaf 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -63,11 +63,8 @@ def test_skill_install_writes_a_version_stamped_copy_to_the_claude_path( assert cli.main(["skill", "--install", "--dir", str(tmp_path)]) == 0 installed = (tmp_path / ".claude/skills/rapidata/SKILL.md").read_text() assert installed.startswith("---\nname: rapidata\n") - assert f"" in installed - assert ( - installed.replace(f"\n", "") - == cli.bundled_skill() - ) + assert _agent_hint.installed_version(installed) == __version__ + assert installed.endswith(cli.bundled_skill().split("\n---\n", 1)[1]) def test_skill_install_honours_agent(tmp_path: Path): From dc7a8ee1c8dcc8e69d5e95c60fe065a8402d1c29 Mon Sep 17 00:00:00 2001 From: RapidPoseidon Date: Mon, 28 Sep 2026 15:46:46 +0000 Subject: [PATCH 4/7] docs(agents): drop the remote skill install from the SDK docs The skill ships with the SDK, so the agent docs now start from `pip install rapidata` and `rapidata skill`. The import hint no longer checks for the Claude Code plugin: the plugin only tells the agent to run `python -m rapidata skill`, which marks the session as read anyway. Co-Authored-By: Claude Opus 5.5 Co-Authored-By: lino@rapidata.ai <68745352+LinoGiger@users.noreply.github.com> --- docs/ai_agents.md | 28 ++++++++-------------------- src/rapidata/_agent_hint.py | 24 +++++------------------- tests/conftest.py | 1 - tests/test_agent_hint.py | 13 ------------- 4 files changed, 13 insertions(+), 53 deletions(-) diff --git a/docs/ai_agents.md b/docs/ai_agents.md index 5b9f346b7b..1a6330e817 100644 --- a/docs/ai_agents.md +++ b/docs/ai_agents.md @@ -4,22 +4,13 @@ Let your coding agent write the Rapidata integration for you. The official Rapid ## Install -Pick your agent. One command. Done. +The skill ships inside the SDK, so installing the SDK is all it takes: -| Agent | Install | -|-------|---------| -| **Claude Code** | `claude plugin marketplace add RapidataAI/skills && claude plugin install rapidata-sdk-plugin@rapidata-sdk-marketplace` | -| **Cursor** | `npx skills add RapidataAI/skills -a cursor` | -| **Windsurf** | `npx skills add RapidataAI/skills -a windsurf` | -| **Copilot** | `npx skills add RapidataAI/skills -a github-copilot` | -| **Cline** | `npx skills add RapidataAI/skills -a cline` | -| **Codex** | `npx skills add RapidataAI/skills -a codex` | -| **Gemini CLI** | `npx skills add RapidataAI/skills -a gemini-cli` | -| **Any other** | `npx skills add RapidataAI/skills` | - -Install once. Works in every session after that. That's it. +```bash +pip install -U rapidata # or: uv add rapidata +``` -The installed skill is a short pointer: it tells the agent to install the SDK and read the full guide that ships with it. Already have the SDK? You can skip the install and read the guide directly: +When a coding agent imports `rapidata`, the SDK points it at the guide. To read it yourself, or to keep a copy in the project so your agent loads it in every session: ```bash rapidata skill # print the guide (same as python -m rapidata skill) @@ -30,7 +21,6 @@ rapidata skill --install --agent cursor # or codex, generic (AGENTS.md) The guide is versioned with the SDK, so it always describes the version you have installed. - ## Logging in The first `RapidataClient()` on a machine opens a browser login and waits for you. An agent can check and trigger it on its own: @@ -46,7 +36,7 @@ If the browser doesn't open, the agent shows you the printed URL. You log in onc ### Automatic -The agent loads the skill when it sees Rapidata-related work. Just ask naturally: +With the skill installed into the project, the agent loads it when it sees Rapidata-related work. Just ask naturally: ``` Create a comparison job that evaluates image quality between two models @@ -58,7 +48,7 @@ Set up a custom audience with 3 qualification examples for prompt adherence ### Manual -On Claude Code, invoke the skill directly: +On Claude Code, invoke an installed skill directly: ``` /rapidata @@ -75,11 +65,9 @@ Other agents follow their own conventions — Cursor rules, Copilot instructions The full guide ships inside the SDK, so upgrading the SDK upgrades the guide: ```bash -pip install -U rapidata # or: uv add -U rapidata +pip install -U rapidata # or: uv lock --upgrade-package rapidata ``` -The installed skill from the table above only points the agent at `python -m rapidata skill`, so it rarely needs an update. Pull one anyway with `claude plugin marketplace update` (Claude Code) or `npx skills update rapidata` (everything else). - Copies written by `rapidata skill --install` record the SDK version that wrote them in their front matter (`metadata.rapidata-sdk-version`) and open with a short check. That check tells the agent to compare the recorded version with `rapidata.__version__` and reinstall on a mismatch. After an SDK upgrade, the next import by a coding agent also prints the reinstall command. Both checks are local and need no network. ## Editing the skill diff --git a/src/rapidata/_agent_hint.py b/src/rapidata/_agent_hint.py index 52bce668d9..b1e54bc4ee 100644 --- a/src/rapidata/_agent_hint.py +++ b/src/rapidata/_agent_hint.py @@ -8,10 +8,10 @@ ``python -m rapidata skill``. Sessions are told apart by the id the runtime exports (:data:`_SESSION_ENV_VARS`); runtimes without one get a read that expires after :data:`ANON_READ_TTL`. -- Never while the Claude Code plugin is installed, or while a copy written by - ``python -m rapidata skill --install`` (project or user level) carries this - SDK's version in its front matter. A copy from another version asks for a - reinstall instead. Nothing here touches the network. +- Never while a copy written by ``python -m rapidata skill --install`` + (project or user level) carries this SDK's version in its front matter. A + copy from another version asks for a reinstall instead. Nothing here touches + the network. ``RAPIDATA_AGENT_HINT=0`` switches it off for processes an agent merely started. State lives in :data:`STATE_FILE`, falling back to the temp dir when a sandbox @@ -30,7 +30,6 @@ from pathlib import Path AGENT_DOCS_URL = "https://docs.rapidata.ai/ai_agents/" -PLUGIN_NAME = "rapidata-sdk-plugin" # Where each agent picks up a project-local skill file, relative to the project root. SKILL_INSTALL_PATHS: dict[str, str] = { @@ -201,19 +200,6 @@ def mark_skill_read() -> None: _save_state(state) -def _plugin_installed() -> bool: - config_dir = Path(os.environ.get("CLAUDE_CONFIG_DIR") or Path.home() / ".claude") - try: - plugins = json.loads( - (config_dir / "plugins" / "installed_plugins.json").read_text( - encoding="utf-8" - ) - ).get("plugins", {}) - except (OSError, ValueError, AttributeError): - return False - return any(name.split("@", 1)[0] == PLUGIN_NAME for name in plugins) - - def installed_copies(root: Path | None = None) -> list[tuple[str, Path, Path, str]]: """Return ``(agent, install_root, path, version)`` for each stamped skill file in the project ``root`` or the home directory.""" root = root or Path.cwd() @@ -260,7 +246,7 @@ def agent_hint() -> str | None: for agent, base, path, version in copies: if version != current: return _stale_hint(agent, base, path, version) - if copies or _plugin_installed() or _read_this_session(_load_state()): + if copies or _read_this_session(_load_state()): return None return AGENT_HINT except Exception: diff --git a/tests/conftest.py b/tests/conftest.py index 1178d6e225..f84d878101 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -31,7 +31,6 @@ *_agent_hint._AGENT_ENV_VARS, *_agent_hint._SESSION_ENV_VARS, "RAPIDATA_AGENT_HINT", - "CLAUDE_CONFIG_DIR", ) diff --git a/tests/test_agent_hint.py b/tests/test_agent_hint.py index 8375bb8ef6..79b86a0e62 100644 --- a/tests/test_agent_hint.py +++ b/tests/test_agent_hint.py @@ -151,19 +151,6 @@ def test_silent_while_running_the_skill_cli(monkeypatch: pytest.MonkeyPatch): assert agent_hint() is None -def test_silent_when_the_plugin_is_installed( - monkeypatch: pytest.MonkeyPatch, tmp_path: Path -): - _agent(monkeypatch) - config = tmp_path / "claude-config" - (config / "plugins").mkdir(parents=True) - (config / "plugins/installed_plugins.json").write_text( - json.dumps({"plugins": {"rapidata-sdk-plugin@rapidata-sdk-marketplace": []}}) - ) - monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(config)) - assert agent_hint() is None - - def test_an_unrelated_agents_md_does_not_count_as_installed( monkeypatch: pytest.MonkeyPatch, sandbox: Path ): From 4191b2e917e56d22c69712c4addd0c60145bd768 Mon Sep 17 00:00:00 2001 From: RapidPoseidon Date: Tue, 29 Sep 2026 07:53:13 +0000 Subject: [PATCH 5/7] ci(release): stop dispatching sdk-release to RapidataAI/skills The skills repo now carries a static pointer to `python -m rapidata skill` and no longer syncs on SDK releases. Co-Authored-By: Claude Opus 5.5 Co-Authored-By: lino@rapidata.ai <68745352+LinoGiger@users.noreply.github.com> --- .github/workflows/release_and_publish.yml | 16 +++------------- 1 file changed, 3 insertions(+), 13 deletions(-) diff --git a/.github/workflows/release_and_publish.yml b/.github/workflows/release_and_publish.yml index 9514284eb0..9b35edf15e 100644 --- a/.github/workflows/release_and_publish.yml +++ b/.github/workflows/release_and_publish.yml @@ -172,9 +172,8 @@ jobs: dist/*.whl dist/*.tar.gz - # A repository_dispatch needs Contents:write on the *target* repo, so a single - # token has to cover every repo the release notifies. The GitHub App grants that - # per run; a repo-scoped PAT silently covers only the repos it was cut for. + # A repository_dispatch needs Contents:write on the *target* repo; the GitHub App + # grants that per run, where a repo-scoped PAT silently covers only its own repos. - name: Create cross-repo dispatch token id: dispatch_token if: steps.release_type.outputs.type == 'stable' @@ -183,18 +182,9 @@ jobs: app-id: ${{ vars.RAPIDATA_OPENAPI_GENERATOR_APP_ID }} private-key: ${{ secrets.RAPIDATA_OPENAPI_GENERATOR_PRIVATE_KEY }} owner: RapidataAI - repositories: "skills,rapidata-mcp" + repositories: "rapidata-mcp" permission-contents: write - - name: Trigger skills plugin version sync - if: steps.release_type.outputs.type == 'stable' - uses: peter-evans/repository-dispatch@v3 - with: - token: ${{ steps.dispatch_token.outputs.token }} - repository: RapidataAI/skills - event-type: sdk-release - client-payload: '{"version": "${{ steps.update_version.outputs.new_version }}"}' - # The hosted MCP server wraps this SDK but resolves it at image build time, so # without a rebuild it keeps serving whatever version was current when that repo # last changed. Hand it the version just published so it pins exactly this release. From 417e9b9e915137374ca8af78da18d17f5f99e2a3 Mon Sep 17 00:00:00 2001 From: RapidPoseidon Date: Tue, 29 Sep 2026 08:16:46 +0000 Subject: [PATCH 6/7] docs(agents): ask contributors' agents to review the skill on every change The repo's agent instructions move from CLAUDE.md to AGENTS.md so Codex and Cursor read them too; CLAUDE.md imports AGENTS.md. The new "Agent skill" section asks for a skill review with every change, matching the Agent Skill PR check. Co-Authored-By: Claude Opus 5.5 Co-Authored-By: lino@rapidata.ai <68745352+LinoGiger@users.noreply.github.com> --- AGENTS.md | 37 +++++++++++++++++++++++++++++++++++++ CLAUDE.md | 34 +--------------------------------- 2 files changed, 38 insertions(+), 33 deletions(-) create mode 100644 AGENTS.md diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000000..2019d6fc70 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,37 @@ +This Project is a Python SDK for the Rapidata API. + +It is built around the RapidataClient class which is the main entry point for interacting with the Rapidata API. + +As a customer you can use the RapidataClient class to access the following: +- JobDefinitionCreation +- AudienceCreation (including filtered audiences via `.filter()` for country/language/demographic targeting) +- ValidationSetCreation +- FlowCreation +- BenchmarkCreation / MRI Creation (including participant metadata such as `participant.rename()` and `participant.set_price()` for the score-vs-cost chart) + +Orders were removed from the SDK entirely (v3.21.0) — do not add, document, or reference order creation. + +The whole authentication and backend communication is handled by the OpenAPIService class. It works in combination with the AUTO GENERATED API CLIENT that is used to make the actual API calls. + +Note that if there are any changes that have to be made in those files you MUST also update the mustache files under openapi/templates. + +please note that there is the RapidataApiClient that wraps every api call to handel backend tracing and error handling. + +The backend errors follow a specific format that you can see in the RapidataError class. + +when doing type annotations use the "from __future__ import annotations" statement and TYPE_CHECKING to check the types - that way you can eliminate the quotation marks around the types. + +## Documentation +When building the docs make sure you use 'uv run --group docs mkdocs build' - otherwise check out the pyproject.toml file for the dependencies. + +When writing documentation, make sure to keep focused, easy to understand, and not repeat information. it should highlight the capabilities while not overexaggerating or falling into hyperbole. + + +## General rules +at the end of your edits make sure to run 'pyright src/rapidata/rapidata_client' and make sure there are no errors. +when updating any interfaces make sure you update the docs and examples. +before every commit make sure to format everything under src/rapidata/rapidata_client with black ('uv run black src/rapidata/rapidata_client'). + +## Agent skill +`src/rapidata/_skill/` is the agent skill that ships in every release and that coding agents read through `python -m rapidata skill`. It is edited only here. +With every change, review whether `SKILL.md` or its companions (`reference.md`, `examples.md`, `flows-for-preference-data.md`) need updating: a new or renamed method, a changed parameter, default or result field, or a new pitfall all do. Update them in the same PR. If nothing the skill documents changed, say so in the PR body, so the reviewer can apply the `skill-unchanged-approved` label that the `Agent Skill` check waits for. diff --git a/CLAUDE.md b/CLAUDE.md index 447f3d2730..43c994c2d3 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1,33 +1 @@ -This Project is a Python SDK for the Rapidata API. - -It is built around the RapidataClient class which is the main entry point for interacting with the Rapidata API. - -As a customer you can use the RapidataClient class to access the following: -- JobDefinitionCreation -- AudienceCreation (including filtered audiences via `.filter()` for country/language/demographic targeting) -- ValidationSetCreation -- FlowCreation -- BenchmarkCreation / MRI Creation (including participant metadata such as `participant.rename()` and `participant.set_price()` for the score-vs-cost chart) - -Orders were removed from the SDK entirely (v3.21.0) — do not add, document, or reference order creation. - -The whole authentication and backend communication is handled by the OpenAPIService class. It works in combination with the AUTO GENERATED API CLIENT that is used to make the actual API calls. - -Note that if there are any changes that have to be made in those files you MUST also update the mustache files under openapi/templates. - -please note that there is the RapidataApiClient that wraps every api call to handel backend tracing and error handling. - -The backend errors follow a specific format that you can see in the RapidataError class. - -when doing type annotations use the "from __future__ import annotations" statement and TYPE_CHECKING to check the types - that way you can eliminate the quotation marks around the types. - -## Documentation -When building the docs make sure you use 'uv run --group docs mkdocs build' - otherwise check out the pyproject.toml file for the dependencies. - -When writing documentation, make sure to keep focused, easy to understand, and not repeat information. it should highlight the capabilities while not overexaggerating or falling into hyperbole. - - -## General rules -at the end of your edits make sure to run 'pyright src/rapidata/rapidata_client' and make sure there are no errors. -when updating any interfaces make sure you update the docs and examples. -before every commit make sure to format everything under src/rapidata/rapidata_client with black ('uv run black src/rapidata/rapidata_client'). +@AGENTS.md From 4efcd502e1828640c3c14431144902633f1bf875 Mon Sep 17 00:00:00 2001 From: RapidPoseidon Date: Tue, 29 Sep 2026 08:31:10 +0000 Subject: [PATCH 7/7] docs(agents): use python -m rapidata skill everywhere One spelling for agents and humans: `python -m rapidata skill` runs against the interpreter the caller uses, while the `rapidata` console script is only on PATH inside an activated environment. Co-Authored-By: Claude Opus 5.5 Co-Authored-By: lino@rapidata.ai <68745352+LinoGiger@users.noreply.github.com> --- README.md | 4 ++-- docs/ai_agents.md | 10 +++++----- src/rapidata/AGENTS.md | 2 +- src/rapidata/_skill/SKILL.md | 2 +- 4 files changed, 9 insertions(+), 9 deletions(-) diff --git a/README.md b/README.md index 5b0f2793d3..ac3164a0bf 100644 --- a/README.md +++ b/README.md @@ -9,8 +9,8 @@ Docs: https://docs.rapidata.ai/ Point it at the guide that ships with the SDK instead of letting it read the installed source: ```bash -rapidata skill # print the guide (same as python -m rapidata skill) -rapidata skill --install # install it into the current project +python -m rapidata skill # print the guide +python -m rapidata skill --install # install it into the current project ``` The guide is edited in [`src/rapidata/_skill/`](https://github.com/RapidataAI/rapidata-python-sdk/tree/main/src/rapidata/_skill), so it always matches the installed version. diff --git a/docs/ai_agents.md b/docs/ai_agents.md index 1a6330e817..8fd610a89d 100644 --- a/docs/ai_agents.md +++ b/docs/ai_agents.md @@ -13,10 +13,10 @@ pip install -U rapidata # or: uv add rapidata When a coding agent imports `rapidata`, the SDK points it at the guide. To read it yourself, or to keep a copy in the project so your agent loads it in every session: ```bash -rapidata skill # print the guide (same as python -m rapidata skill) -rapidata skill reference # companion guides: reference, examples, flows-for-preference-data -rapidata skill --install # write it to .claude/skills/rapidata/SKILL.md -rapidata skill --install --agent cursor # or codex, generic (AGENTS.md) +python -m rapidata skill # print the guide +python -m rapidata skill reference # companion guides: reference, examples, flows-for-preference-data +python -m rapidata skill --install # write it to .claude/skills/rapidata/SKILL.md +python -m rapidata skill --install --agent cursor # or codex, generic (AGENTS.md) ``` The guide is versioned with the SDK, so it always describes the version you have installed. @@ -68,7 +68,7 @@ The full guide ships inside the SDK, so upgrading the SDK upgrades the guide: pip install -U rapidata # or: uv lock --upgrade-package rapidata ``` -Copies written by `rapidata skill --install` record the SDK version that wrote them in their front matter (`metadata.rapidata-sdk-version`) and open with a short check. That check tells the agent to compare the recorded version with `rapidata.__version__` and reinstall on a mismatch. After an SDK upgrade, the next import by a coding agent also prints the reinstall command. Both checks are local and need no network. +Copies written by `python -m rapidata skill --install` record the SDK version that wrote them in their front matter (`metadata.rapidata-sdk-version`) and open with a short check. That check tells the agent to compare the recorded version with `rapidata.__version__` and reinstall on a mismatch. After an SDK upgrade, the next import by a coding agent also prints the reinstall command. Both checks are local and need no network. ## Editing the skill diff --git a/src/rapidata/AGENTS.md b/src/rapidata/AGENTS.md index f5233d480f..e6143b55c1 100644 --- a/src/rapidata/AGENTS.md +++ b/src/rapidata/AGENTS.md @@ -7,7 +7,7 @@ sets, result fields and the mistakes agents make most often. Read it first: ```bash -python -m rapidata skill # print the guide (also: rapidata skill) +python -m rapidata skill # print the guide python -m rapidata skill --install # install it into the current project ``` diff --git a/src/rapidata/_skill/SKILL.md b/src/rapidata/_skill/SKILL.md index f378035b4f..3547fc38bc 100644 --- a/src/rapidata/_skill/SKILL.md +++ b/src/rapidata/_skill/SKILL.md @@ -9,7 +9,7 @@ Rapidata connects you with distributed human labelers worldwide for fast, high-q ## This guide ships with the SDK -`python -m rapidata skill` (or `rapidata skill`) prints the copy bundled with the installed `rapidata` package, so it describes exactly that version. After `pip install -U rapidata`, run it again: an upgrade can change what is documented here. +`python -m rapidata skill` prints the copy bundled with the installed `rapidata` package, so it describes exactly that version. After `pip install -U rapidata`, run it again: an upgrade can change what is documented here. The three companion guides linked at the end print the same way: `python -m rapidata skill reference`, `python -m rapidata skill examples`, `python -m rapidata skill flows-for-preference-data`.