Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
42 changes: 41 additions & 1 deletion apps/control-api/src/config.ts
Original file line number Diff line number Diff line change
@@ -1,8 +1,35 @@
import { resolve } from "node:path";
import { parsePresetSources, type PresetSourceV1 } from "@harbor-hf/control-core";
import type { ParentHardware } from "@harbor-hf/hf-adapters";
import type { ParentHardware, ReviewedVerifierGrant } from "@harbor-hf/hf-adapters";
import { z } from "zod";

const verifierGrantSchema = z.strictObject({
ref: z.string().regex(/^INFERENCE_API_KEY_[A-Z0-9_]{1,48}$/),
worker_image: z.string().regex(/@sha256:[a-f0-9]{64}$/),
benchmark: z.strictObject({
name: z.string().min(1).max(120),
preset: z.string().min(1).max(120),
}),
dataset_repo: z
.string()
.regex(/^https:\/\/huggingface\.co\/datasets\/[A-Za-z0-9_./-]+\.git@[a-f0-9]{40}$/),
dataset_path: z.string().min(1).max(240),
model: z.string().regex(/^[A-Za-z0-9_./-]{1,160}$/),
base_url: z
.string()
.url()
.refine((value) => {
const url = new URL(value);
return (
url.protocol === "https:" &&
!url.username &&
!url.password &&
!url.search &&
!url.hash
);
}),
});

const schema = z.object({
NODE_ENV: z.enum(["development", "test", "production"]).default("production"),
PORT: z.coerce.number().int().min(1).max(65535).default(7860),
Expand All @@ -14,6 +41,7 @@ const schema = z.object({
HARBOR_HF_AUTH_PATH: z.string().min(1).default("/tmp/harbor-hf/auth.sqlite"),
HARBOR_HF_PRESETS_ROOT: z.string().min(1).default("./presets"),
HARBOR_HF_PRESET_SOURCES: z.string().default("[]"),
HARBOR_HF_VERIFIER_GRANTS: z.string().default("[]"),
HARBOR_HF_LAUNCH_PYTHON: z.string().min(1).optional(),
HARBOR_HF_APPROVED_AGENT_SOURCES: z.string().default("[]"),
HARBOR_HF_MAX_ACTIVE_JOBS: z.coerce.number().int().min(1).max(1024).default(16),
Expand Down Expand Up @@ -80,6 +108,7 @@ export interface AppConfig {
* presets for very popular benchmarks; anything else comes from a source.
*/
preset_sources: PresetSourceV1[];
verifier_grants?: ReviewedVerifierGrant[];
launch_python?: string;
approved_agent_sources?: Record<string, unknown>[];
max_active_jobs: number;
Expand Down Expand Up @@ -183,6 +212,16 @@ export function loadConfig(environment: NodeJS.ProcessEnv = process.env): AppCon
operator_org_subject: parsed.HARBOR_HF_OPERATOR_ORG_SUBJECT ?? null,
};
}
const verifierGrants = z
.array(verifierGrantSchema)
.max(8)
.parse(JSON.parse(parsed.HARBOR_HF_VERIFIER_GRANTS));
const subjects = new Set<string>();
for (const grant of verifierGrants) {
const subject = `${grant.benchmark.name}\u0000${grant.benchmark.preset}`;
if (subjects.has(subject)) throw new Error("duplicate verifier credential grant");
subjects.add(subject);
}
return {
node_env: parsed.NODE_ENV,
port: parsed.PORT,
Expand All @@ -194,6 +233,7 @@ export function loadConfig(environment: NodeJS.ProcessEnv = process.env): AppCon
auth_path: resolve(parsed.HARBOR_HF_AUTH_PATH),
presets_root: resolve(parsed.HARBOR_HF_PRESETS_ROOT),
preset_sources: parsePresetSources(JSON.parse(parsed.HARBOR_HF_PRESET_SOURCES)),
verifier_grants: verifierGrants,
...(parsed.HARBOR_HF_LAUNCH_PYTHON
? { launch_python: parsed.HARBOR_HF_LAUNCH_PYTHON }
: {}),
Expand Down
1 change: 1 addition & 0 deletions apps/control-api/src/runtime.ts
Original file line number Diff line number Diff line change
Expand Up @@ -141,6 +141,7 @@ async function composeRuntime(
parentImage: config.parent_image ?? "",
hardware: config.parent_hardware,
timeoutSeconds: config.parent_timeout_seconds,
verifierGrants: config.verifier_grants ?? [],
})
: new ReadOnlyHuggingFaceJobs({
namespace: config.namespace,
Expand Down
41 changes: 41 additions & 0 deletions apps/control-api/test/config.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,47 @@ describe("control API configuration", () => {
).toThrow();
});

it("loads only pinned, HTTPS verifier credential grants", () => {
const grant = {
ref: "INFERENCE_API_KEY_JUDGE",
worker_image: `example/parent@sha256:${"a".repeat(64)}`,
benchmark: { name: "example-benchmark", preset: "one-task" },
dataset_repo: `https://huggingface.co/datasets/example-org/tasks.git@${"b".repeat(40)}`,
dataset_path: "tasks",
model: "judge-model",
base_url: "https://judge.invalid/v1",
};
expect(loadConfig(environment).verifier_grants).toEqual([]);
expect(
loadConfig({
...environment,
HARBOR_HF_VERIFIER_GRANTS: JSON.stringify([grant]),
}).verifier_grants,
).toEqual([grant]);
expect(() =>
loadConfig({
...environment,
HARBOR_HF_VERIFIER_GRANTS: JSON.stringify([grant, grant]),
}),
).toThrow("duplicate verifier credential grant");
for (const bad of [
{
...grant,
dataset_repo: "https://huggingface.co/datasets/example-org/tasks.git@main",
},
{ ...grant, base_url: "http://judge.invalid/v1" },
{ ...grant, base_url: "https://user:pass@judge.invalid/v1" },
{ ...grant, ref: "HF_TOKEN" },
{ ...grant, extra: "not reviewed" },
])
expect(() =>
loadConfig({
...environment,
HARBOR_HF_VERIFIER_GRANTS: JSON.stringify([bad]),
}),
).toThrow();
});

it("rejects development authentication in production", () => {
expect(() =>
loadConfig({
Expand Down
22 changes: 21 additions & 1 deletion docs/CONTROL_SERVICE.md
Original file line number Diff line number Diff line change
Expand Up @@ -127,7 +127,7 @@ Do not create a resource per run. SQLite files and HF Jobs are temporary. The
three-table projection rebuilds from run records, Harbor results, attempt cost
receipts, and current Job observations.

The Space has two secrets:
The Space requires two core secrets. Reviewed agent and verifier connections can use additional, separately registered secrets:

- `HF_TOKEN` is a purpose-scoped control credential with access to the Bucket,
HF Jobs, and approved private Hugging Face Dataset Git repositories when they
Expand Down Expand Up @@ -155,6 +155,7 @@ The service reads these Space variables:
| `HARBOR_HF_BUCKET_ROOT` | no | `/data` | local filesystem store root |
| `HARBOR_HF_PRESETS_ROOT` | no | `./presets` | reviewed presets for very popular benchmarks |
| `HARBOR_HF_PRESET_SOURCES` | no | `[]` | pinned external preset sources as JSON; see [preset sources](preset-sources.md) |
| `HARBOR_HF_VERIFIER_GRANTS` | no | `[]` | reviewed, benchmark-scoped verifier credential grants as JSON; no secret values |
| `HARBOR_HF_LAUNCH_PYTHON` | no | worker package `.venv/bin/python`; `/opt/harbor-launch/bin/python` in the control image | pinned native launch inspector |
| `HARBOR_HF_APPROVED_AGENT_SOURCES` | no | `[]` | reviewed native ACP source objects as JSON; never credentials |
| `HARBOR_HF_WRITE_MODE` | no | `disabled` | permit Job lifecycle changes |
Expand Down Expand Up @@ -207,6 +208,25 @@ credential files, run configuration, Bucket objects, projections, browser
responses, or logs. It must not enter trial agent environments or agent source
installation. The separate inference credential path remains unchanged.

### Verifier judge credential

Some task verifiers need a different model and key from the agent. Set
`HARBOR_HF_VERIFIER_GRANTS` only after reviewing the benchmark source and the
registered credential reference. Each JSON object names `ref`, `worker_image`,
`benchmark` (`name` and `preset`), `dataset_repo`, `dataset_path`, `model`, and
`base_url`. Pin the Dataset Git URL to a full commit and the worker image to a
SHA256 digest. Keep the source credential in the existing Space secret store;
never put its value in this variable.

At parent Job start, the service checks the exact benchmark, dataset, image,
operator, and active credential reference again. It sends the selected key as
an ephemeral `AGENT_JUDGE_API_KEY` Job secret. It sends the reviewed URL and
model as `AGENT_JUDGE_API_URL` and `AGENT_JUDGE_MODEL`. Harbor then resolves
the task's `${AGENT_JUDGE_API_KEY:-}` verifier environment entry from the
parent Job environment. The run record and trial configuration contain no
judge key. A different dataset or a verifier environment override blocks the
credential. This grant does not replace the separate reviewed agent connection.

Admission and repository inspection occur before model inference. A missing
token, denied repository, missing commit, malformed source, or native Harbor
resolution failure stops the launch. The service does not fall back to another
Expand Down
30 changes: 16 additions & 14 deletions docs/provider-credential-references.md
Original file line number Diff line number Diff line change
Expand Up @@ -57,20 +57,22 @@ style without a connection is refused before any run record is written. A
submission that names no connection keeps the ordinary router route, so approving
a credential never changes where an otherwise identical submission runs.

`model_api` is Harbor's own agent argument: at a custom endpoint Harbor requires
it and `pi` writes it into its `models.json` `api` value. The grant stores the
value it reviewed, and admission requires the built record to carry exactly that
value, so a record whose wire API style changed after the build is denied. Only
the submission names the connection and its wire API style; no preset field, no
alias table and no submission-supplied URL is added.

The run record for such a connection is the ordinary `openai/<model>` route with
the reviewed base URL and an opaque key reference. Admission and restart recheck
that record against the approved grant. A model outside the grant, another preset
version, another worker image, another operator, a changed base URL, a changed
host list or a changed wire API style is denied. A connection the submission
names but the registry does not hold for this exact subject is denied as well,
and the run never falls back to the router.
For agents that accept Harbor's `model_api` option, the built record carries the
approved value. For example, `pi` writes it into the `api` field of its
`models.json`. Admission checks that value again at each start.

Harbor's ACP agent does not accept `model_api`. A reviewed ACP source configures
its own client. The submission must still name the API style approved by the
grant, but the built record does not add an unsupported ACP option. At each start,
admission requires exactly one matching grant for the agent import path and
version, model, operator, worker image and credential reference. It refuses an
ambiguous grant or an ACP record with an injected `model_api` option. The
separate source allowlist checks the pinned executable source.

The run record uses `openai/<model>`, the reviewed base URL and an opaque key
reference. Admission checks the URL and hosts again at each start. A connection
without an approved grant for the exact subject is denied; the run cannot fall
back to the router.

## Deployment caveat

Expand Down
19 changes: 17 additions & 2 deletions packages/control-core/src/inference-bindings.ts
Original file line number Diff line number Diff line change
Expand Up @@ -203,6 +203,17 @@ export class InferenceBindings {
return this.#manifest.bindings.map(({ ref, source_env }) => ({ ref, source_env }));
}

/** Select an active source owned by this operator for a separately scoped verifier grant. */
reviewedSource(ref: string, actor: string): string | null {
const binding = this.#manifest.bindings.find(
(entry) =>
entry.ref === ref &&
entry.enabled &&
entry.uses.some((use) => use.operator_subjects.includes(actor)),
);
return binding?.source_env ?? null;
}

assertSourceTransitionFrom(previous: InferenceBindings): void {
for (const binding of this.sourceIdentities()) {
const prior = previous
Expand Down Expand Up @@ -379,12 +390,16 @@ export class InferenceBindings {
kwargs && typeof kwargs === "object" && !Array.isArray(kwargs)
? (kwargs as Record<string, unknown>).model_api
: undefined;
// ACP has no native model_api option. Its pinned source configures the client;
// admission still requires one unambiguous reviewed grant for the native
// agent identity, model, actor and image at every start or restart.
const isAcp = subject.import_path === "harbor.agents.installed.acp:AcpAgent";
const matches =
modelId === null
modelId === null || (isAcp && modelApi !== undefined)
? undefined
: binding?.uses.filter(
(use) =>
use.model_api === modelApi &&
(isAcp || use.model_api === modelApi) &&
this.presetMatches(use, subject, modelId, actor, image),
);
const grant = matches?.[0];
Expand Down
5 changes: 4 additions & 1 deletion packages/control-core/src/presets.ts
Original file line number Diff line number Diff line change
Expand Up @@ -112,7 +112,10 @@ export function presetEndpointAgent(
);
if (preset.reasoning_option !== null && reasoningEffort !== "default")
kwargs[preset.reasoning_option] = reasoningEffort;
kwargs.model_api = modelApi;
// Harbor's ACP runner takes a pinned source, not a model_api argument. The
// reviewed grant selects the wire API style; the source configures its client.
if (fragment.import_path !== "harbor.agents.installed.acp:AcpAgent")
kwargs.model_api = modelApi;
return {
...base,
...(fragment.name ? { name: fragment.name } : {}),
Expand Down
2 changes: 1 addition & 1 deletion packages/control-core/src/service.ts
Original file line number Diff line number Diff line change
Expand Up @@ -605,7 +605,7 @@ export class ControlService {
return { created: true, run: record };
}),
);
await this.refresh();
await this.projection.rebuild(this.store, await this.jobs.list(), record.run_id);
return result;
}

Expand Down
21 changes: 19 additions & 2 deletions packages/control-core/test/control.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -154,6 +154,20 @@ describe("run submission", () => {
expect(projection.run(result.run.run_id)?.status).toBe("queued");
});

it("updates only the submitted run projection", async () => {
const first = await submit("first-run");
const rebuild = vi.spyOn(projection, "rebuild");
const second = await submit("second-run");
expect(rebuild).toHaveBeenCalledExactlyOnceWith(
store,
expect.any(Array),
second.run.run_id,
);
expect(projection.run(first.run.run_id)?.status).toBe("queued");
expect(projection.run(second.run.run_id)?.status).toBe("queued");
rebuild.mockRestore();
});

it("loads the nine-trial diagnostic preset unchanged through both launch paths", async () => {
const benchmark = { name: "terminal-bench-2-1", preset: "three-tasks-3-trials" };
const canary = presets.benchmark(benchmark.name, benchmark.preset);
Expand Down Expand Up @@ -1536,10 +1550,13 @@ describe("reconciliation", () => {
namespace: "example",
accessToken: "synthetic",
fetch: async (input) => {
if (String(input).includes("cursor=second"))
const url = new URL(String(input));
if (url.searchParams.has("cursor"))
return fails
? new Response("unavailable", { status: 503 })
: new Response(JSON.stringify([raw("parent", "existing-parent")]));
: new Response("[]");
if (url.searchParams.get("label") === "harbor-hf-role=parent")
return new Response(JSON.stringify([raw("parent", "existing-parent")]));
return new Response(
JSON.stringify(
Array.from({ length: 100 }, (_, i) => raw("trial", `child-${i}`)),
Expand Down
62 changes: 62 additions & 0 deletions packages/control-core/test/preset-endpoint.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -198,6 +198,68 @@ describe("reviewed endpoint connections for native presets", () => {
expect(admit(agent)?.source).toBe(source);
});

it("admits a pinned ACP source without adding an unsupported model_api option", () => {
const acpPath = "harbor.agents.installed.acp:AcpAgent";
const acpVersion = "1.2.3";
const sourceConfig = {
repo_url: "https://github.com/example/agent.git",
ref: "a".repeat(40),
source_dir: "agent",
manifest_path: "harbor-agent.json",
};
const preset = {
...catalog.agent("pi", "0.84.4"),
harbor_agent: {
import_path: acpPath,
kwargs: { version: acpVersion, source: sourceConfig },
},
reasoning_option: null,
};
const grant = grantOf({ agent_import_path: acpPath, agent_version: acpVersion });
const policy = policyOf([grant]);
const subject = { import_path: acpPath, version: acpVersion };
expect(
policy.presetConnection(ref, subject, modelId, actor, image, modelApi),
).toMatchObject(connection());
expect(
policy.presetConnection(ref, subject, modelId, actor, image, "openai-responses"),
).toBeNull();
const agent = presetEndpointAgent(
undefined,
preset,
modelId,
connection(),
modelApi,
"default",
);
expect(agent.kwargs).toEqual({ version: acpVersion, source: sourceConfig });
expect(admit(agent, [grant])).toEqual({ ref, source });
expect(() => admit(agent, [grantOf({ ...grant, agent_version: "other" })])).toThrow(
"not reviewed",
);
expect(() =>
admit(
{ ...agent, kwargs: { ...(agent.kwargs as object), model_api: modelApi } },
[grant],
),
).toThrow("not reviewed");
expect(() =>
admit(agent, [grant, grantOf({ ...grant, model_api: "openai-responses" })]),
).toThrow("not reviewed");
expect(() =>
admit(
{
...agent,
env: {
OPENAI_BASE_URL: "https://other.invalid/v1",
OPENAI_API_KEY: `\${${ref}}`,
},
},
[grant],
),
).toThrow("not reviewed");
});

it("keeps the router record when the submission names no connection", () => {
const router = catalog.buildJobConfig("run-1", routerSubmission, "/data");
expect(router.agents?.[0]?.model_name).toBe("huggingface/example/model:endpoint");
Expand Down
Loading
Loading