Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions .claude-plugin/marketplace.json
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,16 @@
"name": "TransluceAI"
},
"tags": ["analysis", "ai", "mcp", "ingestion"]
},
{
"name": "fxtr",
"source": "./plugins/fxtr",
"description": "Skills for writing, running, and viewing fxtr experiments, and for calling models from them with behaviors",
"version": "0.1.0",
"author": {
"name": "TransluceAI"
},
"tags": ["experiments", "skills"]
}
]
}
166 changes: 166 additions & 0 deletions .github/scripts/plugin_sanity.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,166 @@
"""Sanity checks for the marketplace and every plugin in it.

Run from the repository root: ``python .github/scripts/plugin_sanity.py``. Every plugin gets the
generic checks (its marketplace entry, manifest, and skill frontmatter); a plugin with checks of
its own below gets those too.
"""

import json
import re
import sys
from pathlib import Path
from typing import Any, Callable

ROOT = Path.cwd()
VERSION = re.compile(r"\d+\.\d+\.\d+")
LINK = re.compile(r"\]\(([^)\s]+)\)")


def fail(message: str) -> None:
raise SystemExit(message)


def load_json(path: Path) -> Any:
try:
return json.loads(path.read_text(encoding="utf-8"))
except Exception as exc:
fail(f"{path} is not valid JSON: {exc}")


def frontmatter(path: Path) -> dict[str, str]:
"""The top-level keys of a SKILL.md's YAML frontmatter, as raw strings."""
text = path.read_text(encoding="utf-8")
match = re.match(r"---\n(.*?)\n---\n", text, re.DOTALL)
if match is None:
fail(f"{path} has no frontmatter")
keys = {}
for line in match.group(1).splitlines():
key = re.match(r"([A-Za-z][\w-]*):\s*(.*)$", line)
if key:
keys[key.group(1)] = key.group(2).strip()
return keys


def check_plugin(entry: dict[str, Any]) -> Path:
"""The checks every plugin gets; returns its directory."""
name = entry.get("name")
plugin_dir = ROOT / entry.get("source", "")
if not plugin_dir.is_dir():
fail(f"marketplace source of {name} does not exist: {plugin_dir}")
manifest = load_json(plugin_dir / ".claude-plugin" / "plugin.json")
if manifest.get("name") != name:
fail(f"plugin manifest name must be {name}")
version = manifest.get("version")
if not isinstance(version, str) or not VERSION.fullmatch(version):
fail(f"{name} plugin manifest version must be plain major.minor.patch")
if entry.get("version") != version:
fail(f"marketplace {name} version must match its plugin manifest version")
for skill in sorted((plugin_dir / "skills").glob("*/SKILL.md")):
keys = frontmatter(skill)
if keys.get("name") != skill.parent.name:
fail(f"{skill} frontmatter name must be {skill.parent.name}")
if not keys.get("description"):
fail(f"{skill} frontmatter needs a description")
return plugin_dir


def check_docent(plugin_dir: Path) -> None:
required_files = [
".claude-plugin/plugin.json",
".mcp.json",
"skills/docent/SKILL.md",
"skills/docent/analysis.md",
"skills/docent/dql-reference.md",
"skills/docent/ingestion-reference.md",
"skills/docent/ingestion.md",
"skills/docent/readings-reference.md",
"skills/docent/report.md",
]
for rel_path in required_files:
path = plugin_dir / rel_path
if not path.is_file():
fail(f"required plugin file is missing: {rel_path}")
if path.suffix == ".md" and not path.read_text(encoding="utf-8").strip():
fail(f"markdown file is empty: {rel_path}")

mcp = load_json(plugin_dir / ".mcp.json")
server = mcp.get("mcpServers", {}).get("docent")
if not isinstance(server, dict):
fail(".mcp.json must define mcpServers.docent")
if server.get("type") != "stdio" or server.get("command") != "uv":
fail("docent MCP server must run as uv stdio")
args = server.get("args")
if not isinstance(args, list) or "--from" not in args:
fail("docent MCP server args must include --from")

forbidden_names = {".mcp.local.json", "docent.env"}
for path in plugin_dir.rglob("*"):
if path.name in forbidden_names or path.name.startswith("docent.env."):
fail(f"local credential/config file must not be published: {path}")


def check_fxtr(plugin_dir: Path) -> None:
"""The fxtr and behaviors skills, as fxtr3's `pnpm sync:plugin` prepares them: both, from
one clean revision of fxtr3, each with its references and links that resolve."""
skills = sorted(path.name for path in (plugin_dir / "skills").iterdir() if path.is_dir())
if skills != ["behaviors", "fxtr"]:
fail(f"fxtr plugin must hold exactly the behaviors and fxtr skills, not {skills}")
revisions = set()
for name in skills:
skill_dir = plugin_dir / "skills" / name
skill_md = skill_dir / "SKILL.md"
if not skill_md.is_file() or not skill_md.read_text(encoding="utf-8").strip():
fail(f"{name} skill needs a non-empty SKILL.md")
references = [p for p in (skill_dir / "references").rglob("*") if p.is_file()]
if not references:
fail(f"{name} skill has no references")
for page in references:
if page.suffix in {".md", ".mdx"} and not page.read_text(encoding="utf-8").strip():
fail(f"reference page is empty: {page.relative_to(plugin_dir)}")

provenance = load_json(skill_dir / "provenance.json")
if provenance.get("skill") != name:
fail(f"{name} provenance names the skill {provenance.get('skill')!r}")
revision = provenance.get("revision")
if not isinstance(revision, str) or not re.fullmatch(r"[0-9a-f]{40}", revision):
fail(f"{name} provenance needs the full fxtr3 revision it was prepared from")
if provenance.get("modified") is not False:
fail(f"{name} was prepared from a modified fxtr3 checkout; sync from a clean one")
revisions.add(revision)

for target in LINK.findall(skill_md.read_text(encoding="utf-8")):
if re.match(r"[a-z][a-z0-9+.-]*:", target) or target.startswith("#"):
continue # a URL or an anchor in the page itself
path = (skill_dir / target.split("#", 1)[0]).resolve()
if not path.is_file():
fail(f"{name}/SKILL.md links to a missing file: {target}")
if len(revisions) != 1:
fail("the fxtr and behaviors skills must be prepared from the same fxtr3 revision")


PLUGIN_CHECKS: dict[str, Callable[[Path], None]] = {
"docent": check_docent,
"fxtr": check_fxtr,
}


def main() -> None:
marketplace = load_json(ROOT / ".claude-plugin" / "marketplace.json")
entries = marketplace.get("plugins")
if not isinstance(entries, list):
fail("marketplace plugins must be a list")
names = [entry.get("name") for entry in entries]
if len(set(names)) != len(names):
fail(f"marketplace plugin names must be unique: {names}")
for name in PLUGIN_CHECKS:
if names.count(name) != 1:
fail(f"marketplace must contain exactly one {name} plugin entry")
for entry in entries:
plugin_dir = check_plugin(entry)
if entry["name"] in PLUGIN_CHECKS:
PLUGIN_CHECKS[entry["name"]](plugin_dir)
print(f"Claude Code plugin sanity checks passed: {', '.join(names)}")


if __name__ == "__main__":
sys.exit(main())
79 changes: 2 additions & 77 deletions .github/workflows/plugin-sanity.yml
Original file line number Diff line number Diff line change
Expand Up @@ -18,80 +18,5 @@ jobs:
with:
python-version: "3.12"

- name: Validate Claude Code plugin package
run: |
python - <<'PY'
import json
import re
from pathlib import Path

root = Path.cwd()

def fail(message: str) -> None:
raise SystemExit(message)

def load_json(path: Path) -> dict:
try:
return json.loads(path.read_text(encoding="utf-8"))
except Exception as exc:
fail(f"{path} is not valid JSON: {exc}")

marketplace = load_json(root / ".claude-plugin" / "marketplace.json")
entries = marketplace.get("plugins")
if not isinstance(entries, list):
fail("marketplace plugins must be a list")

docent_entries = [entry for entry in entries if entry.get("name") == "docent"]
if len(docent_entries) != 1:
fail("marketplace must contain exactly one docent plugin entry")

entry = docent_entries[0]
plugin_dir = root / entry.get("source", "")
if not plugin_dir.is_dir():
fail(f"marketplace source does not exist: {plugin_dir}")

manifest = load_json(plugin_dir / ".claude-plugin" / "plugin.json")
if manifest.get("name") != "docent":
fail("plugin manifest name must be docent")

version = manifest.get("version")
if not isinstance(version, str) or not re.fullmatch(r"\d+\.\d+\.\d+", version):
fail("plugin manifest version must be plain major.minor.patch")
if entry.get("version") != version:
fail("marketplace docent version must match plugin manifest version")

required_files = [
".claude-plugin/plugin.json",
".mcp.json",
"skills/docent/SKILL.md",
"skills/docent/analysis.md",
"skills/docent/dql-reference.md",
"skills/docent/ingestion-reference.md",
"skills/docent/ingestion.md",
"skills/docent/readings-reference.md",
"skills/docent/report.md",
]
for rel_path in required_files:
path = plugin_dir / rel_path
if not path.is_file():
fail(f"required plugin file is missing: {rel_path}")
if path.suffix == ".md" and not path.read_text(encoding="utf-8").strip():
fail(f"markdown file is empty: {rel_path}")

mcp = load_json(plugin_dir / ".mcp.json")
server = mcp.get("mcpServers", {}).get("docent")
if not isinstance(server, dict):
fail(".mcp.json must define mcpServers.docent")
if server.get("type") != "stdio" or server.get("command") != "uv":
fail("docent MCP server must run as uv stdio")
args = server.get("args")
if not isinstance(args, list) or "--from" not in args:
fail("docent MCP server args must include --from")

forbidden_names = {".mcp.local.json", "docent.env"}
for path in plugin_dir.rglob("*"):
if path.name in forbidden_names or path.name.startswith("docent.env."):
fail(f"local credential/config file must not be published: {path}")

print("Claude Code plugin sanity checks passed")
PY
- name: Validate the marketplace and its plugins
run: python .github/scripts/plugin_sanity.py
9 changes: 8 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,17 @@ A collection of plugins for Claude Code.
| Plugin | Description |
|--------|-------------|
| [docent](./plugins/docent) | Docent AI analysis tools |
| [fxtr](./plugins/fxtr) | Skills for writing, running, and viewing fxtr experiments, and for calling models from them with behaviors |

The docent plugin includes two skills:
- **analysis** - Analyzing agent behavior with Docent
- **ingestion** - Structured workflow for ingesting agent run data into Docent

The fxtr plugin includes two skills, each with a copy of the documentation it links from
[docs.transluce.ai](https://docs.transluce.ai):
- **fxtr** - Setting up, writing, running, and viewing fxtr experiments
- **behaviors** - Calling language models from fxtr experiments with the behaviors library

## Installation

Add this marketplace to Claude Code:
Expand All @@ -20,8 +26,9 @@ Add this marketplace to Claude Code:
/plugin marketplace add TransluceAI/claude-code-plugins
```

Then install the plugin:
Then install a plugin:

```shell
/plugin install docent@transluce-plugins
/plugin install fxtr@transluce-plugins
```
8 changes: 8 additions & 0 deletions plugins/fxtr/.claude-plugin/plugin.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,8 @@
{
"name": "fxtr",
"description": "Write, run, and view fxtr experiments",
"author": {
"name": "TransluceAI"
},
"version": "0.1.0"
}
38 changes: 38 additions & 0 deletions plugins/fxtr/skills/behaviors/SKILL.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,38 @@
---
name: behaviors
description: Call language models from fxtr experiments with the behaviors Python library, covering one-shot requests and judge steps, multi-turn chat rollouts, scripted or custom user and tool policies, retries and failures, streaming, checkpoints, and storing conversations. Use whenever a fxtr experiment calls a model or holds a conversation, or when extending behaviors' model adapters or policies.
---

# behaviors

behaviors is a Python library for calling language models from fxtr experiments: provider
adapters, conversation sessions, context policies that play users and tools, retries and failure
recording, and conversation records stored as fxtr entities. Use these pieces when they fit the
experiment, rather than building another chat loop or transcript format.

The pages below are its documentation. Read the ones a task needs before writing code.

## Where to read

| Task | Read |
| ------------------------------------------------------------------------------------------------------------------------------ | ---------------------------------------------------------------------------------- |
| Add behaviors to a project; hold a conversation, judge outputs, and summarize verdicts in fxtr steps; test without credentials | [Calling language models](references/guides/language-models.mdx) |
| Pick a provider, an adapter, and an exact model ID | [Choosing models](references/guides/choosing-models.mdx) |
| Script users, offer tools, or write a custom conversation flow | [Policies and tools](references/guides/policies-and-tools.mdx) |
| Rollout statuses and `max_turns`, failures, retries, streaming | [Rollout outcomes and retries](references/guides/rollout-outcomes.mdx) |
| Messages and content, stored records, transcripts, conversations in the viewer, checkpoints | [Conversation records and checkpoints](references/guides/conversation-records.mdx) |
| Run Docent readings, or convert records for Docent | [Docent readings](references/guides/docent-readings.mdx) |
| Choose a layer; imports, adapters, sampling settings, requests and responses | [API overview](references/reference/api.mdx) |
| Steps, workflows, arrays, launching jobs, and the step cache | the adjacent [fxtr skill](../fxtr/SKILL.md) |

## Best practices

* **Judge with `gpt-6-luna` by default.** For an LLM judge, use `gpt-6-luna` through the
`OpenAIResponsesAPI` adapter, unless the user asks for another model.
* **Keep the model the user names.** Verify its exact ID from the provider's listing, as Choosing
models shows, rather than substituting another. Without credentials to check, take candidate
IDs from the provider's published catalog, and tell the user that account access has not been
verified.
* **Shape model experiments as the guide does**: one step per conversation, with every setting
that affects the request in a configuration entity; fxtr replicas for repeated samples; and
separate steps for judging and for aggregating verdicts.
11 changes: 11 additions & 0 deletions plugins/fxtr/skills/behaviors/provenance.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
{
"skill": "behaviors",
"documentation": "https://docs.transluce.ai/behaviors",
"packages": {
"fxtr": "0.0.1a0",
"behaviors": "0.0.1a0"
},
"revision": "8d820d77c1292b4792602cbbe5a9876693de3d99",
"modified": false,
"content_sha256": "1c5471096fa21b00e363ac7f69f33bb54d19ade3998658a937328b980645485b"
}
19 changes: 19 additions & 0 deletions plugins/fxtr/skills/behaviors/references/INDEX.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
# The behaviors documentation

The pages of https://docs.transluce.ai/behaviors, copied with this skill: `provenance.json`,
beside its `SKILL.md`, says from which version. Each is listed with what it covers,
in the order of the site's navigation.

## Get started

- [Introduction](index.mdx): Primitives for evaluating AI model behavior.

## Other pages

- [Choosing models](guides/choosing-models.mdx): Find the exact model IDs a provider offers your account, pick the adapter for its endpoint, and check a model before a sweep.
- [Conversation records and checkpoints](guides/conversation-records.mdx): What a rollout records, how conversations are stored as fxtr entities and flattened into transcripts, and how to checkpoint a session.
- [Docent readings](guides/docent-readings.mdx): Convert conversations to Docent's data models, and run a Docent reading against a transcript or agent run with a model you choose.
- [Calling language models](guides/language-models.mdx): Use the behaviors library to hold conversations and judge outputs in fxtr steps.
- [Policies and tools](guides/policies-and-tools.mdx): Decide what the model sees each turn: scripted users, tool policies, their combinations, and custom conversation flows.
- [Rollout outcomes and retries](guides/rollout-outcomes.mdx): How a rollout ends, how model call failures are retried, recorded, or raised, and how to watch a rollout as it runs.
- [API overview](reference/api.mdx): The layers of behaviors, where each is imported from, the provider adapters, and the request, response, and sampling types.
Loading
Loading