Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
25 commits
Select commit Hold shift + click to select a range
b0ef4c6
feat: add CreatorBench v1 governance and routing
HomenShum Jul 22, 2026
2097353
fix: hash only declared capability pack manifests
HomenShum Jul 22, 2026
d50acf5
feat: harden CreatorBench acquisition and sealed proof
HomenShum Jul 22, 2026
46b0a0d
test: record CreatorBench UI QA
HomenShum Jul 22, 2026
bb42f21
test: freeze CreatorBench preflight release
HomenShum Jul 22, 2026
6b60321
fix: render absent CreatorBench metrics safely
HomenShum Jul 22, 2026
477c177
feat: publish standalone CreatorBench claim artifact
HomenShum Jul 22, 2026
a546c19
feat(creatorbench): add resumable NASA SVS source provider
HomenShum Jul 22, 2026
bb9c7ca
fix(creatorbench): stratify heldout creator groups
HomenShum Jul 22, 2026
13e8c4d
feat(creatorbench): publish real multi-format render pilot
HomenShum Jul 22, 2026
8ab44a2
chore(creatorbench): bind render pilot to source commit
HomenShum Jul 22, 2026
a42851e
feat(creatorbench): gate workflows by admissible corpus tiers
HomenShum Jul 22, 2026
9eea47f
test(creatorbench): freeze 111-source motion benchmark
HomenShum Jul 22, 2026
70dd312
fix(creatorbench): normalize subgroup evidence in public UI
HomenShum Jul 22, 2026
de03a3d
test(creatorbench): await review image decoding
HomenShum Jul 22, 2026
3856157
feat(creatorbench): harden review and shadow evaluation
HomenShum Jul 22, 2026
55219c7
test(creatorbench): freeze v1.1 evidence
HomenShum Jul 22, 2026
857bfca
chore(convex): refresh CreatorBench API bindings
HomenShum Jul 22, 2026
2387a33
feat(creatorbench): expand v1.2 corpus and workflow evidence
HomenShum Jul 22, 2026
ea308ce
fix(creatorbench): re-freeze consistent v1.3 manifests
HomenShum Jul 22, 2026
95850b6
fix(creatorbench): gate usability claims on human review
HomenShum Jul 22, 2026
86f8a6c
chore(creatorbench): publish frozen v1.4 evidence
HomenShum Jul 22, 2026
804ec17
test(creatorbench): bind e2e to v1.4 evidence
HomenShum Jul 22, 2026
c2baf1e
docs(qa): record CreatorBench v1.4 production proof
HomenShum Jul 22, 2026
5f749a5
docs(testing): repair CreatorBench prompt library
HomenShum Jul 22, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
4 changes: 4 additions & 0 deletions .qa/memory/findings.jsonl
Original file line number Diff line number Diff line change
Expand Up @@ -53,3 +53,7 @@
{"id":"nodevideo-ui-topology-20260721","rootCause":"onboarding-operation-provider-proof-export-composed-as-one-card-dashboard","sev":"P0","symptom":"proof-dashboard-as-primary-product-instead-of-artifact-workspace","status":"fixed","area":"creator-product-topology","evidence":"harness/references/anti-references/nodevideo-proof-dashboard-as-product","fix":"guided-start+artifact-workspace+run-inspector+mobile-modes+topology-gate","ts":"2026-07-22T01:14:09.081Z","fp":"95a1b08805f3"}
{"id":"NV-SR-001","sev":"P1","area":"creator responsive topology","symptom":"Artifact stage falls below majority width at tablet and xl boundary viewports","rootCause":"Three-column workspace stayed active below the width where both side rails could coexist with an artifact-dominant canvas","evidence":"GitHub Actions run 29885587329 and focused xl-boundary/tablet Playwright rerun","fix":"Switch tablet to mode-based surfaces at 1023px and compact rails from 1024px through 1350px","status":"fixed","ts":"2026-07-22T02:30:07.066Z","fp":"09e51987fd46"}
{"id":"NV-ATLAS-001","sev":"P1","area":"tracking-atlas-responsive-topology","symptom":"The 834px tablet viewport retained the desktop navigation rail and overflowed horizontally","rootCause":"The Atlas single-column breakpoint was set to 820px, below the tested tablet width","evidence":".qa/runs/2026-07-21-tracking-atlas/pixels/tablet-dark.png and tablet-light.png before/after pixel rerun","fix":"Move the deliberate single-column navigation transition to 900px and assert document scrollWidth against innerWidth","status":"fixed","ts":"2026-07-22T03:24:00.000Z","fp":"nv-atlas-tablet-900"}
{"id":"NV-CB-001","sev":"P1","area":"creatorbench-responsive-topology","symptom":"mobile-navigation-overflow","rootCause":"desktop-links-remained-unwrapped-at-mobile-breakpoint","evidence":"creatorbench-focused-desktop-mobile-playwright","fix":"compact-wrap-plus-overflow-assertions","status":"fixed","ts":"2026-07-22T04:51:57.719Z","fp":"7fa8eba8dce1"}
{"id":"NV-CB-002","sev":"P0","area":"creatorbench-production-evidence-rendering","symptom":"real-null-cost-metric-crashed-public-dashboard","rootCause":"normalizer-preserved-null-as-numeric-value","evidence":"signed-in-vercel-preview-console-plus-12-pass-regression","fix":"normalize-all-optional-measured-numbers-through-finite-guard","status":"fixed","ts":"2026-07-22T04:59:24.159Z","fp":"c1fa451a2311"}
{"id":"creatorbench-claim-honesty-v14","sev":"P0","area":"CreatorBench public claim honesty","symptom":"The v1.3 report described a 0.0% usable first-pass result even though zero private held-out instances had human editing-quality review.","rootCause":"derivePublicClaim emitted automatic and assisted usability language unconditionally from machine routing dispositions.","evidence":"v1.3 report output contrasted with v1.4 public report and src/lib/creatorbench-contracts.test.ts","fix":"Gate usability and silent-failure claims on complete human review coverage; otherwise report only review-required and safe-abstention routing.","status":"fixed","ts":"2026-07-22T10:05:11.681Z","fp":"e22af4c1239a"}
{"id":"creatorbench-e2e-count-drift-v14","sev":"P1","area":"CreatorBench production UI evidence binding","symptom":"The focused UI proof expected stale pilot corpus counts after the benchmark expanded to the v1.4 corpus.","rootCause":"CreatorBench E2E assertions remained pinned to earlier 112-clip pilot values instead of frozen report values.","evidence":"tests/e2e/creatorbench.spec.ts; 40/40 local and 40/40 production five-viewport reruns","fix":"Bind assertions to the v1.4 evidence counts: 250 clips, 205 creator-disjoint sources, 6,128 instances, and 1,392 private held-out instances.","status":"fixed","ts":"2026-07-22T10:05:33.613Z","fp":"cbfe5598592d"}
3 changes: 3 additions & 0 deletions .qa/memory/runs.jsonl
Original file line number Diff line number Diff line change
Expand Up @@ -19,3 +19,6 @@
{"evidenceDir":".qa/evidence/creator-topology","executor":"codex","gates":{"fullCheck":"pass","topology":"pass","creatorDesktopMobile":"12-pass-4-intentional-skip","build":"pass","typecheck":"pass"},"bar":{"durability":5,"functional":5,"responsive":5,"visual":4,"safety":5,"accessibility":5},"journeys":["guided-first-arrival","artifact-workspace","proposal-review","paid-executor-gate","durable-two-session","export-reopen","mobile-surface-modes"],"ts":"2026-07-22T01:13:56.489Z"}
{"executor":"Codex","journeys":["Smart Reframe local journey"],"bar":{"functional":4,"coherence":4,"responsive":4,"accessibility":4,"proof":4},"gates":{"check":"pass","creatorProof":"pass"},"evidenceDir":".qa/evidence/smart-reframe","ts":"2026-07-22T02:19:26.333Z"}
{"executor":"Codex","mode":"AUTHORIZED PRODUCTION","journeys":{"A0":"PASS","A1":"PASS(local catalog)","A2":"N/A(no external model; local guide)","A3":"PASS(hash-bound receipts)","A4":"PASS(downloadable 42.6s compilation)","A5":"PASS(desktop/tablet/mobile light+dark)","A6":"PASS(fail-closed license/hash validators)","A7":"N/A"},"bar":{"B1":2,"B2":2,"B3":2,"B4":2,"B5":2,"B6":2,"B7":1,"B8":2,"B9":2,"B10":2,"B11":2},"gates":{"atlasVerify":"8-pass","atlasE2E":"4-pass","typecheck":"pass","build":"pass","pixels":"6-pass-no-overflow"},"evidenceDir":".qa/runs/2026-07-21-tracking-atlas","notes":"Eight CC Attribution fixtures; six automatic YOLO11n routes and two explicit first-frame seed plus OpenCV routes. Fixture-bound proof only.","ts":"2026-07-22T03:24:00.000Z"}
{"executor":"Codex","mode":"AUTHORIZED_PRODUCTION","journeys":{"A0":"PASS","A1":"PASS","A2":"PASS","A3":"PASS","A4":"PASS","A5":"PASS","A6":"PASS","A7":"PARTIAL_5_of_250"},"bar":{"B1":2,"B2":2,"B3":2,"B4":2,"B5":2,"B6":2,"B7":2,"B8":2,"B9":2,"B10":2,"B11":2},"gates":{"lint":"pass","typecheck":"pass","unit":"247-pass","build":"pass","creatorbenchE2E":"10-pass","acquisitionTarget":"fail-honest-5-of-250"},"evidenceDir":".qa/runs/2026-07-22-creatorbench-v1","notes":"dataset-and-human-review-incomplete-no-universal-claim","ts":"2026-07-22T04:51:57.778Z"}
{"executor":"codex","journeys":{"A0":"PASS","A1":"PASS","A2":"SKIPPED","A3":"PASS","A4":"PASS","A5":"PASS","A6":"PASS","A7":"NA"},"bar":{"B1":2,"B2":1,"B3":2,"B4":2,"B5":2,"B6":1,"B7":1,"B8":2,"B9":2,"B10":1,"B11":2},"gates":["lint","typecheck","build","playwright14","proof","bloat"],"evidenceDir":".qa/runs/2026-07-21-creatorbench-render-pilot","ts":"2026-07-22T05:40:57.668Z"}
{"executor":"Codex CreatorBench v1.4 production proof","journeys":{"A0":"PASS production shell and report route","A1":"N/A read-only benchmark surface","A2":"N/A no live AI execution on public report","A3":"PASS freeze receipt, provenance, and report binding","A4":"PASS JSON and CSV downloads plus export/reopen evidence","A5":"PASS five viewport projects, both themes, accessibility, and no overflow","A6":"PASS missing-report, infrastructure-only, null-metric, and blind-review states","A7":"N/A public evidence surface is not the durable agent execution route"},"bar":{"B1":2,"B2":2,"B3":2,"B4":2,"B5":2,"B6":2,"B7":1,"B8":2,"B9":2,"B10":2,"B11":2},"gates":{"repository":"PASS npm run check","proof":"PASS 16 CreatorBench proof gates","localE2E":"PASS 40/40 across five viewports","productionE2E":"PASS 40/40 across five viewports","localPixels":"PASS 6/6 theme and viewport captures","productionPixels":"PASS 6/6 theme and viewport captures","deploymentIdentity":"PASS Vercel build receipt, frozen report SHA, and source commit match"},"evidenceDir":"C:\\Users\\hshum\\.codex\\visualizations\\2026\\07\\20\\019f7d76-4ac0-7730-b4f0-9a8b64863b4e\\creatorbench-v1.4-qa","nextImprovement":"Complete authenticated durable human review labels and correction-time measurements before making usability or silent-failure claims.","ts":"2026-07-22T10:05:57.744Z"}
57 changes: 57 additions & 0 deletions .ui/contract.json
Original file line number Diff line number Diff line change
Expand Up @@ -485,6 +485,63 @@
"forbidTexts": ["/ 100"]
}
]
},
{
"id": "creatorbench",
"route": "/creatorbench",
"landmark": {
"testid": "creatorbench-overview"
},
"description": "Public measured-evidence surface for CreatorBench. It separates frozen private routing outcomes, workflow-admissible corpus coverage, public deterministic render proof, known gaps, route evidence, and bounded blinded review from the creator editing workspace.",
"controls": [
{
"id": "creatorbench-overview",
"role": "button",
"name": "Overview"
},
{
"id": "creatorbench-coverage",
"role": "button",
"name": "Coverage"
},
{
"id": "creatorbench-weaknesses",
"role": "button",
"name": "Weaknesses"
},
{
"id": "creatorbench-routes",
"role": "button",
"name": "Route evidence"
},
{
"id": "creatorbench-freeze",
"role": "button",
"name": "Freeze receipt"
},
{
"id": "creatorbench-review",
"role": "button",
"name": "Review lab",
"meaning": "Blinded reviewer surface. Machine findings and route identity remain hidden until judgment. Durable submission requires explicit review-data consent and a reachable Convex backend."
},
{
"id": "creatorbench-guide-input",
"role": "textbox",
"name": "Ask CreatorBench"
}
],
"states": [
{
"id": "creatorbench-evaluated",
"meaning": "Exact counts and confidence intervals are loaded from the public report; zero human labels are disclosed rather than interpreted as editing quality.",
"requireTexts": [
"See what works, what needs help, and what fails.",
"Human editing quality and silent-failure incidence are unverified"
]
}
],
"agentGuidance": "Use this surface for measured evidence, not editing. Verify every rate against its numerator, denominator, interval, scope, and freeze receipt. Do not infer human usability from routing-only or decode-only results."
}
]
}
30 changes: 30 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -111,6 +111,36 @@ See [the generalized architecture](docs/architecture/GENERALIZED_CREATOR_PIPELIN
integration contract](docs/architecture/EXECUTOR_INTEGRATION.md), [Higgsfield runbook](docs/operations/HIGGSFIELD_RUNBOOK.md),
and [copy-ready end-to-end prompt suite](docs/testing/CREATOR_PIPELINE_TEST_PROMPTS.md).

## CreatorBench

`/creatorbench` is the evidence surface for measured generalization. It is separate from the creator
workspace and the fixture-bound Artifact Atlas. Every benchmark request uses the same
`nodevideo.creator-request/v1` contract, and every outcome is classified as automatic usable,
assisted usable, review required, safe abstention, unsupported, technical failure, or silent
failure. A rendered file is never counted as usable without a blinded human judgment.

```powershell
npm run creatorbench:acquire
npm run creatorbench:manifests
npm run creatorbench:dedupe
npm run creatorbench:validate
npm run creatorbench:evaluate
npm run creatorbench:freeze
$env:NODEVIDEO_CREATORBENCH_EVALUATOR_TOKEN='<post-freeze evaluator credential>'
npm run creatorbench:evaluate:sealed
npm run creatorbench:report
npm run proof:creatorbench
```

Acquisition is fail-honest: the target remains 250 rights-cleared clips, 75 creator-disjoint sources,
15 domains, 8 workflows, and 2,000 instances. If a source host or rights review prevents that target,
the receipt publishes the achieved population and gap; it does not prefill the claim. See the
[governance contract](docs/benchmarks/CREATORBENCH_V1_GOVERNANCE.md), [sealed evaluation
boundary](docs/benchmarks/SEALED_EVALUATION.md), and [end-to-end benchmark prompts](docs/testing/CREATORBENCH_E2E_TEST_PROMPTS.md).
The report command also emits the standalone machine-readable
`benchmarks/creatorbench-v1/results/public-claim.json`; it is derived from the frozen held-out
results and is scanned alongside the public report for private locators and hidden targets.

### Music rights modes

| Music input | NodeVideo behavior |
Expand Down
3 changes: 3 additions & 0 deletions apps/atlas/src/atlas.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -459,6 +459,9 @@ function AtlasApp() {
<a href="/creator">
<ArrowLeft /> Creator workspace
</a>
<a href="/creatorbench">
<Gauge /> CreatorBench
</a>
<p>Explore</p>
{MODES.map(({ id, label, icon: Icon }) => (
<button
Expand Down
106 changes: 106 additions & 0 deletions apps/creatorbench/src/creatorbench-review-client.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,106 @@
import { describe, expect, it, vi } from 'vitest';
import {
CreatorBenchReviewClient,
type DurableReviewInput,
type ReviewBackend,
} from './creatorbench-review-client';

class MemoryStorage implements Storage {
private readonly values = new Map<string, string>();
get length() {
return this.values.size;
}
clear() {
this.values.clear();
}
getItem(key: string) {
return this.values.get(key) ?? null;
}
key(index: number) {
return [...this.values.keys()][index] ?? null;
}
removeItem(key: string) {
this.values.delete(key);
}
setItem(key: string, value: string) {
this.values.set(key, value);
}
}

const input: DurableReviewInput = {
benchmarkVersion: 'creatorbench-v1.4',
instanceId: 'instance:public:1',
resultId: 'result:public:1',
split: 'public-test',
blindedVariantIds: ['variant:a', 'variant:b'],
usability: 'usable_after_minor_correction',
correctionTimeSeconds: 18,
reasonCodes: ['wrong_subject'],
explicitOptIn: true,
};

function backend(): ReviewBackend {
const records: Array<Record<string, unknown>> = [];
return {
claimAssignment: vi.fn(async (args) => {
records.push({ ...args, status: 'assigned', blind: true });
}),
submitReview: vi.fn(async (args) => {
const record = records.find((item) => item.assignmentId === args.assignmentId);
if (record) Object.assign(record, args, { status: 'completed' });
}),
listReviewerHistory: vi.fn(async ({ reviewerRef }) =>
records.filter((item) => item.reviewerRef === reviewerRef),
) as ReviewBackend['listReviewerHistory'],
deleteReviewerData: vi.fn(async ({ reviewerRef }) => {
records.splice(0, records.length);
return { deletedCount: 1, deletedAt: Date.now(), reviewerRef };
}),
};
}

describe('CreatorBench review client', () => {
it('requires explicit opt-in before creating an identity or contacting the backend', async () => {
const transport = backend();
const storage = new MemoryStorage();
const client = new CreatorBenchReviewClient(transport, storage);
await expect(client.submit({ ...input, explicitOptIn: false })).rejects.toThrow(/opt-in/u);
expect(transport.claimAssignment).not.toHaveBeenCalled();
expect(storage.length).toBe(0);
});

it('persists a pseudonymous blinded review and verifies the completed record', async () => {
const transport = backend();
const client = new CreatorBenchReviewClient(transport, new MemoryStorage());
const result = await client.submit(input);
expect(result.reviewerRef).toMatch(/^reviewer:[a-f\d]{32}$/u);
expect(result.confirmed).toMatchObject({ blind: true, status: 'completed' });
expect(transport.claimAssignment).toHaveBeenCalledWith(
expect.objectContaining({ consentVersion: 'creatorbench-review-consent/v1' }),
);
});

it('fails closed when the backend is unavailable', async () => {
const client = new CreatorBenchReviewClient(null, new MemoryStorage());
await expect(client.submit(input)).rejects.toThrow(/backend unavailable/u);
});

it('exports pseudonymous history and verifies deletion before clearing local identity', async () => {
const storage = new MemoryStorage();
const client = new CreatorBenchReviewClient(backend(), storage);
await client.submit(input);
const exported = JSON.parse(await client.exportHistory());
expect(exported.reviewerRef).toMatch(/^reviewer:/u);
expect(exported.records).toHaveLength(1);
const receipt = await client.deleteAll();
expect(receipt.deletedCount).toBe(1);
expect(storage.length).toBe(0);
});

it('rejects hidden target and private locator hints in blind variant identifiers', async () => {
const client = new CreatorBenchReviewClient(backend(), new MemoryStorage());
await expect(
client.submit({ ...input, blindedVariantIds: ['private-heldout-target-hint'] }),
).rejects.toThrow(/prohibited/u);
});
});
Loading
Loading