From 6d707e42f6e8dc9d4049a9432c862faadff286e4 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Fri, 18 Sep 2026 22:50:41 +0800 Subject: [PATCH 1/3] docs(blog): publish paired English editions Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../blog/agent-facing-kanban/index.html | 247 ++++++++++++++++++ .../blog/application-scenarios/index.html | 201 ++++++++++++++ .../application-scenarios/presentation.js | 28 ++ .../agent-facing-kanban/authority-en.svg | 44 ++++ .../images/agent-facing-kanban/board-en.svg | 49 ++++ .../images/agent-facing-kanban/lease-en.svg | 39 +++ .../images/agent-facing-kanban/objects-en.svg | 46 ++++ apps/presentation/site/public/blog/index.html | 8 + .../blog/zh/agent-facing-kanban/index.html | 5 +- .../blog/zh/application-scenarios/index.html | 5 +- 10 files changed, 670 insertions(+), 2 deletions(-) create mode 100644 apps/presentation/site/public/blog/agent-facing-kanban/index.html create mode 100644 apps/presentation/site/public/blog/application-scenarios/index.html create mode 100644 apps/presentation/site/public/blog/application-scenarios/presentation.js create mode 100644 apps/presentation/site/public/blog/images/agent-facing-kanban/authority-en.svg create mode 100644 apps/presentation/site/public/blog/images/agent-facing-kanban/board-en.svg create mode 100644 apps/presentation/site/public/blog/images/agent-facing-kanban/lease-en.svg create mode 100644 apps/presentation/site/public/blog/images/agent-facing-kanban/objects-en.svg diff --git a/apps/presentation/site/public/blog/agent-facing-kanban/index.html b/apps/presentation/site/public/blog/agent-facing-kanban/index.html new file mode 100644 index 0000000000..2419b19fef --- /dev/null +++ b/apps/presentation/site/public/blog/agent-facing-kanban/index.html @@ -0,0 +1,247 @@ + + + + + + + LoopX: Native Kanban for long-running agents · LoopX + + + + + + + + + + + + + + + + + + +
+
+ ← All posts +

Agent-native Kanban · Long-running collaboration

+

LoopX: Native Kanban for long-running agents

+

See how agents claim work, honor gates, submit evidence, and collaborate over shared authority state—from Goals and Todos to claims, leases, and lanes.

+ +
+
+ +
+ +
+

Start with a delivery scenario

+

Suppose you tell an agent: “Deliver a bulk export feature. It must handle large datasets without exposing another user’s data. Development, testing, and documentation may continue, but ask me before the production release.”

+

After a while, you usually care less about how many tools the model called than about five questions: What remains before the goal is met? Who is working on it? What can continue? What must wait? Why should anyone believe it is done? The LoopX board organizes information around those questions.

+

This article uses that synthetic scenario to explain the full foundation. We will read one card, follow it into a collaboration that can be handed off and recovered, then return to four advanced questions for agent-facing Kanban. Here, Kanban means a board-shaped collaboration interface, not a redefinition of the full Kanban method.

+
The board is an entry point for understanding and operating work

Goals, tasks, ownership, gates, and evidence each have their own identity and rules. The board arranges them into columns, lanes, and details; the control plane receives and validates legal operations. Screen layout does not become a new permission or source of truth.

+
+ +
+

The core objects in one diagram

+

These concepts are not a form that must be filled all at once. Start with the goal and work items. Express collaboration, risk, concurrency, or recovery only when they arise, through the corresponding contract. A local single-agent Goal can remain simple; a multi-agent Goal reuses the same core objects.

+
Diagram of LoopX board objects: Goal, Todo, Claim, Lease, Gate, Evidence, Lane, and shared authority state
Figure 1 · A Goal defines the outcome and Todos organize work. Ownership, constraints, and evidence answer different questions; a lane is one reading of those facts.Scroll horizontally on narrow screens; select the image to open it full size. All diagrams use synthetic examples.
+
+ + + + + + + + + +
When reading a card, first ask which question each field answers
ConceptQuestion it answersBulk export example
GoalWhat must ultimately be achieved, and which constraints must survive?Deliver a usable, secure, verifiable bulk export.
AcceptanceWhat must be observed before the outcome is accepted?Large-volume behavior, permission isolation, and the usage path are validated.
TodoWhat is the next deliverable piece of work or explicit wait?Implement export, review independently, validate integration, release.
ClaimWhich agent is responsible now?A development agent claims implementation; a reviewer claims review.
LeaseWhen the mode is enabled, which execution is valid, for how long, and within what scope?The owner, expiry, version, and write scope of one controlled execution.
GateWhich decision or authority is missing, and exactly what does it block?Release confirmation blocks release, not independent testing by default.
EvidenceWhat supports a judgment, and which version does it apply to?Test results, review conclusions, and artifact references for candidate revision A.
LaneHow is work grouped and selected from the current agent’s or operator’s perspective?Runnable, monitor due, awaiting a decision, owned by someone else.
Shared authority stateUnder concurrency, which rule-protected state counts?The agreed Todos, claims, leases, and their commit receipts.
+

The goal sets direction; work items carry action; claims express responsibility; leases constrain execution occupancy; gates constrain legality; evidence supports judgment. They connect to one another, but none substitutes for another.

+
+ +
+

Goals and Todos: separate the outcome from the current plan

+

A Goal is the control plane’s outcome boundary. It has a stable goal_id and connects current state, work items, constraints, and history. One Git repository may serve several Goals, and one Goal may span several repositories. Repository location should not replace goal identity.

+

A Todo is an identified unit of work inside the Goal and has a todo_id. A title such as “fix export” is not enough: whoever takes over also needs the concrete action, acceptance, dependencies, owner, and permitted scope. Editing a title should not turn the Todo into a different task; when direction changes, supersession relationships should preserve the lineage.

+
Goal: Deliver bulk export
+Acceptance: permission isolation is correct; large datasets work;
+            a user can complete an export by following the documentation
+Constraints: development and validation are allowed;
+             production release requires confirmation
+
+Todo A: Implement export and basic tests
+Todo B: Review the candidate independently; the author may not self-review
+Todo C: Validate large-volume behavior and permission boundaries
+Todo D: Release the accepted candidate after the release decision
+

This is an explanatory work record, not an executable CLI payload. A through D are the current plan, not the Goal itself. If review reveals a simpler approach, some Todos can be replaced while the outcome and confirmed constraints remain.

+

With multiple agents, a per-Agent Vision adds the direction each peer currently owns, its acceptance, and the triggers for replanning. It is a bounded execution-direction record—not another Goal and not a grant of global administration to one agent. See Work graph, authority, and peer collaboration for the foundation.

+
+ +
+

Cards, states, and lanes: one body of work, several useful readings

+
Synthetic bulk-export board with tasks, owners, a lease, a release gate, and validation evidence
Figure 2 · A board for explanation. Columns and lanes are read models, not a new persisted enum. “Running” requires execution evidence and cannot be inferred from claimed_by alone.Scroll horizontally on narrow screens; select the image to open it full size. All diagrams use synthetic examples.
+

The front of a card is a good place for the work name, priority, owner, key waits, latest evidence, and next step. Full dependencies, history, lease details, and failure reasons belong in the detail view. People can scan quickly while agents can expand the facts through stable identities.

+

Do not collapse three kinds of grouping into one. Lifecycle state says whether a Todo is open, blocked, deferred, or done. Work type says whether it advances, monitors, gates, or reminds. A lane is a grouping, candidate set, or scheduling path for a particular actor. “In development” and “awaiting review” can be domain display columns without becoming universal Kernel states. Supersession is recorded by the supersede operation and relationships such as superseded_by, not by inventing another generic state.

+
+ + + + + +
Common task classes: route by type, not by guessing from the title
Work typePurposeKey distinction
advancement_taskImplementation, research, validation, documentation, or repair.Must produce a verifiable result; it is not limited to code.
continuous_monitorObserve an external change by condition or cadence.Stay quiet without material change; polling count is not delivery.
user_gateWait for a decision that blocks related work.The scope must be explicit; “wait for the user” is insufficient.
user_actionRemind a person to do something.The reminder neither grants authority nor blocks work automatically.
blockerRecord a missing execution condition and recovery path.A person may not be needed—for example, while a test environment recovers.
+

The same “independent review” card can be visible but excluded for the development agent, yet claimable by the review agent. Work claimed by someone else may still appear as context, but the current agent cannot take it over merely because it is visible. claimed_by, excluded_agents, dependencies, capabilities, gates, and the current budget all affect execution eligibility.

+

Therefore, open does not mean runnable; high priority does not bypass a Gate; appearing on the board does not mean the current agent may execute. Quota and scheduler logic derive current routing from these facts. The returned contract distinguishes executable obligations from suggestions. Diagrams and the Planning Horizon help with understanding; neither is a second scheduler.

+
+ +
+

Claims and leases: responsibility is not execution occupancy

+

Claim: who owns the card

+

A claim usually appears as a Todo’s claimed_by. It tells other agents that someone owns the work and lets state writes check the actor. It is soft ownership: it does not prove the process is alive or a Host is bound, and it is not a pass around authority checks.

+

Ordinary lifecycle operations must honor current owner and authority rules. Cross-owner completion, reassignment, or supersession requires explicit delegation already allowed by the contract. Calling an agent a “steward” or “coordinator” does not automatically grant permission to modify every Todo.

+

Lease: which execution is currently valid

+

When a Goal selects the corresponding hard-lease collaboration mode, a lease can also carry execution-instance identity, TTL, version, and write scope. Renewal must prove a valid holder and matching version. Expired, invalid, or stale versions cannot count as current execution authority. Supported operations also depend on the selected provider and completed migration stage.

+

When reading a lease, TTL is its validity window, version identifies its current revision, and epoch can be understood as a generation of execution occupancy that separates old and new actors. Fencing is the pre-write check: an expired or ineligible actor cannot use an old identity to commit a protected write even if its process is still running.

+
Claim and Lease timeline covering ownership, a valid lease, renewal, expiry, and rejection of a stale actor
Figure 3 · A lease path after the mode is enabled and recovery eligibility is met. Exact version and epoch fields follow the selected contract; reassignment must pass current admission, grace, and recovery checks. Lease expiry does not itself kill the old process.Scroll horizontally on narrow screens; select the image to open it full size. All diagrams use synthetic examples.
+

Current quota does not consume hard leases automatically. Whether a lease is used depends on the actual Host and execution-path integration. A claim can exist without a hard lease; a valid lease can exist while a release Gate, missing capability, or budget boundary still prevents execution. A lease constrains only paths integrated with its fencing and cannot provide “exactly once” effects for every external system. Stopping the old executor and starting the new one still require the corresponding Runtime supervision and recovery.

+

A common mistake is to read “replay succeeded” as “authority is still valid.” An idempotent replay returns the historical result of the original operation; it does not renew the lease. Read the current lease before continuing to write. Existing canonical lease renewal explicitly supports renewal for promoted local File and SQLite providers. That does not mean transfer, release, reclaim, and the full actor lifecycle have qualified for every provider.

+
+ +
+

Gates and dependencies: wait precisely, and continue precisely

+

A dependency answers whether a fact has become true. A Gate answers whether the relevant decision or authority exists. A test environment that has not recovered and an upstream artifact that has not arrived do not necessarily require human approval. A release that needs confirmation cannot substitute more monitor cycles for authority.

+

A Gate must have scope. Use the corresponding lane scope when a decision blocks only one agent. Use a concrete Todo link or typed decision scope when it affects one action. Use an explicit global Gate only when the whole Goal truly pauses. A continuation binding such as goal-bound does not imply a global block.

+
How does “ask me before release” appear on the board?

The release Todo references the production decision. Integration validation and documentation can continue when they are independent, authorized, and otherwise eligible. Moving the release card to “ready” does not consume the Gate, and an ordinary user reply cannot be interpreted arbitrarily as release approval.

+

Dependency, successor, and supersede must also remain distinct. A dependency is a prerequisite. A successor identifies who continues the work; it does not automatically mean “wait until the previous item completes.” Supersede means a new route replaces an old one and must not disguise invalidated work as success. A wait condition may bind to Todo completion, a material Monitor change, or another supported event.

+

Recovery rechecks current conditions: the revision may have changed, authority may have been revoked, and an artifact may have expired. See Decision Scope, the Todo Contract, and the task-graph projection.

+
+ +
+

Evidence: make “done” a judgment that can be checked

+

“Done” on a board must open into a reason. First separate three things: an Artifact is the deliverable, Evidence supports a judgment, and a Receipt records that an operation was accepted or observed. An exported file is an artifact; permission tests and sample inspection are evidence; a “release request submitted” receipt proves only that the request occurred.

+
+ + + + + +
Useful evidence lets the next actor answer at least these questions
QuestionExample
What was validated?Candidate revision A, its Todo, and this execution.
Which check was performed?A negative permission-isolation test, large-volume integration test, independent review.
What is the result and boundary?Record passed, failed, and untested separately, including applicable inputs and environment.
How can someone inspect and reproduce it?An authorized artifact reference, run record, exact version, or digest.
What change would make it stale?A change to code, input, dependency, or authority that affects the conclusion.
+

“The summary contains a hash,” “the test command exited 0,” and “an agent sent a message” each prove only a limited fact. Whether evidence satisfies acceptance requires domain validation and, where needed, independent judgment. The control plane preserves identity, scope, result, and relationships; it does not invent correctness on behalf of a person or validator.

+

Put concise, authorized references on the shared board. Keep raw logs, user data, and private documents in authorized storage. Possessing a reference is not reading the content; receiving material is not adopting it; being able to read evidence is not authority to act. The agent-scoped evidence ledger provides a bounded timeline and compressed frontiers from other agents—not a second task database.

+
+ +
+

Shared authority state: let every participant know which facts count

+

If two agents both read “unclaimed,” or one agent returns from another Host with stale state, sharing a table is not enough. The system must know who may commit, which version the write is based on, who loses a conflict, and how to recover when an operation committed but its response was lost.

+

Shared authority state is the authoritative state and write boundary everyone agrees to follow. It does not require putting everything into one giant database. The corresponding Todo, claim, and lease aggregates have their own owners; Goal intent, request and response, and external effects retain their own transaction boundaries. Explicit relationships and receipts reconcile across boundaries instead of pretending one transaction encloses every system.

+
Layered shared-authority diagram with entry points, typed lifecycle owner, selected state source, receipts, and read-only projections
Figure 4 · Several entry points read facts governed by the same rules. The actual source follows the Goal’s current selection and promotion state. The three providers shown are alternative configurations, not three simultaneous primary databases.Scroll horizontally on narrow screens; select the image to open it full size. All diagrams use synthetic examples.
+

Shared facts, committed by version

+

On supported canonical transaction paths, a commit carries an expected revision. Compare-and-swap (CAS) asks whether someone changed the state after it was read. The commit result binds the event to the original operation receipt. A competitor cannot overwrite a new owner merely because it also saw the stale card. A lost response is recovered through the identity of the original operation rather than by blindly executing it twice.

+

Read local defaults, provider selection, and promotion separately

+

LoopX retains both legacy paths and gradually migrated provider paths. Promotion formally moves a Goal’s authority source to a selected canonical provider. An unpromoted Goal still follows the existing Markdown or local-writer authority rules; a provider-first call without a selector chooses the File profile. Selecting a File, SQLite, or PostgreSQL provider neither promotes a Goal nor automatically grants cross-host write authority.

+

File and SQLite have concrete transactions and real-backend validation. Full continuous-operation qualification, a default switch, and owner promotion for SQLite remain separate boundaries. PostgreSQL has provider and service integration contracts plus a test foundation; authentication, tenant isolation, network failure, and cross-host operation must qualify under the relevant profile. “An implementation exists” cannot be presented as “every deployment is ready.”

+

After promotion, an empty read from the selected canonical provider means empty, and a failure must be reported explicitly. The system cannot silently fall back to legacy Markdown or a legacy lease file and keep writing. Markdown may remain as a readable projection; UI, cache, and rows synchronized to external boards remain projections too. For more detail, see Local Authority Provider Selection and the Shared Authority RFC.

+
+ +
+

Put the objects together into one collaboration

+
    +
  1. Define the outcome. The owner submits the bulk-export Goal, acceptance, and release constraints. Work becomes identified Todos, with explicit responsibility boundaries between implementation and review.
  2. +
  3. Select and claim. A development agent reads the current eligible candidates and claims implementation. If the selected mode requires a lease, it also obtains valid execution occupancy. A successful claim does not remove the need to check environment, capability, and authority.
  4. +
  5. Execute and write back. Implementation produces a candidate revision and test evidence. Failures and partial success are recorded honestly. The end of a Turn does not complete a Todo automatically.
  6. +
  7. Take over and validate. A successor review binds the candidate identity and acceptance. The reviewer takes over through its own admission and claim path, reads the artifact, and reaches a conclusion. Registering a receiver, delivering a message, and actual takeover are three different facts.
  8. +
  9. Handle waits. Integration validation can continue while the release Gate constrains only matching work. If the revision changes, an old review does not qualify the new candidate; the relevant validation must be confirmed again.
  10. +
  11. Accept and return. Once agreed acceptance and required release authority are satisfied, deliver through the appropriate domain process and read back the result. If gaps remain, link an existing successor or replan. Return the completion conclusion to the original audience.
  12. +
+

This path shows how several capabilities cooperate. It does not imply a universal command that performs the whole collaboration in one step. Frontend, Lark, CLI, and real Runtime entry points, takeover, and result return must each be validated. A count of registered workers is not evidence of execution or acceptance.

+
+ +
+

How to read a LoopX board in practice

+

For routine inspection, this order is enough. Start with the Goal’s acceptance gaps, then inspect the current agent’s work lane. Open the selected Todo and confirm its owner, dependencies, and Gates. Inspect the lease when concurrent execution needs it. Finally expand evidence and the next step. When information is incomplete, follow stable identities into detail; do not infer “does not exist” from “is not displayed.”

+

These are the query entry points for a connected Goal. Replace placeholders with real, registered identities. For source development, run uv run --extra test loopx … from the corresponding worktree; for an installed build, use loopx directly.

+
# Current goal and expandable work graph
+loopx --format json status --goal-id <goal-id> --include-task-graph
+
+# Work from the current agent's perspective; identity is not display order
+loopx --format json todo list --goal-id <goal-id> --role agent --agent-id <agent-id>
+
+# Current lease facts for one work item
+loopx --format json task-lease inspect --goal-id <goal-id> --todo-id <todo-id>
+
+# Bounded evidence timeline visible to the current agent
+loopx --format json evidence-log --goal-id <goal-id> --agent-id <agent-id> --thin --limit 20
+

Write operations use lifecycle entry points exposed by the current Todo, Gate, claim or lease, and runtime contract—not by editing a display column or constructing status prose. Read current command help and the returned action contract to learn required parameters, whether an execution identity is needed, and whether work can continue.

+

The CLI and local frontend expose operations and read models at different levels. Lark has corresponding messages, goal channels, and a Base board adapter, but that does not establish full equivalence across every field and action. The Lark Kanban adapter still identifies itself as a prototype contract: synchronization and triggers require configuration, and an external board does not create another task identity system.

+
+ +
+

Advanced I: make state transitions executable contracts

+

Once the basic objects exist, “in development → awaiting review” is more than moving a card. It must identify the delivered revision, evidence, review work, and subsequent modification boundary. The model proposes a next step; the control plane checks identity, preconditions, and relevant authority; then it records acceptance or rejection.

+

LoopX organizes Todo completion, successor binding, and next-step updates as lifecycle operations; see Todo Next Action. Domain capabilities organize business stages such as development, review, and release. The Kernel need not hard-code every project’s business columns.

+

Request submission, successful execution, state commit, and user-visible result remain distinct stages. At the end of an operation, the system should answer “what is now a shared fact, and how does the next participant take over?”—not merely append an activity row.

+
+ +
+

Advanced II: give agents bounded, expandable decision context

+

Sending the entire board and full history to a model can bury the current constraints. Returning only “do the next item” loses dependencies and alternate routes. A Planning Horizon provides nearby work, relationships, waits, acceptance gaps, and a small number of comparable options.

+

LoopX’s Planning Horizon is a bounded read model. It must disclose coverage and truncation and allow expansion by identity. It must not quietly change authority through summary order or mistake a local view for global state.

+

A reviewer needs: “Review revision A. Current tests pass, but large-volume evidence is missing. The developer is adding validation. Release is not authorized.” That is closer to the decision than a complete chat transcript. Suggestions remain suggestions; machine-enforced execution obligations must be explicit in the contract.

+
+ +
+

Advanced III: reassess the Goal after a Todo completes

+

A merged pull request may complete implementation without making bulk export usable to a user. When local work completes, link an explicit successor, create genuinely necessary work, or state why no successor is needed. Preserve supersession when direction changes.

+

If the task chain is exhausted while acceptance remains unmet, replan. If acceptance is established, stop. LoopX’s replan settlement binds writeback to a specific completed identity or obligation. It does not count a task created only to close the process as progress.

+

No-follow-up settles the current continuation question; it does not automatically accept the entire Goal. Adoption of a message, release of a lease, and marking a Todo done each prove a change only at their own layer.

+
+ +
+

Advanced IV: preserve facts and boundaries through cross-layer recovery

+

Runtime and Harness provide execution and sessions. The control plane maintains identity, state, and legal transitions. Domain capabilities and models organize routes and acceptance. The board presents the result to people and agents. Good interfaces let them cooperate without allowing one layer to take over another layer’s authority.

+

If a remote action happened but the local response was lost, a “pending” state does not prove the action did not occur. Recover through existing idempotency identities, receipts, and result queries. When evidence is insufficient, preserve uncertainty. Lease expiry does not undo external actions that already happened, and retry is not universally safe.

+

Shared authority can protect commits inside the boundary it owns. Workflows across code repositories, release systems, and messaging channels still need their own effect and reconciliation contracts. See From one-shot agents to long-horizon control for the complete layering and Effect Programs.

+
+ +
+

How to tell whether the board actually improves collaboration

+
    +
  • Two actors claim at once: can the system determine who succeeded and who must reread, instead of leaving both believing they may execute?
  • +
  • A stale actor returns: can an expired lease, version, or source be misused? Are external actions that lack fencing identified explicitly?
  • +
  • The reviewed revision changes: can old evidence be mistaken for validation of the new version?
  • +
  • Release awaits confirmation: does the Gate block release while independent, authorized validation continues?
  • +
  • A card completes or is superseded: can someone read back the artifact, remaining acceptance gaps, successor, and reason for the change?
  • +
  • The authority source is unavailable: can the system silently fall back to an old file and create two writers?
  • +
  • The board or history is truncated: can readers discover undisplayed work instead of incorrectly declaring the Goal complete?
  • +
+

This is also how to read capability status: separate published foundation contracts, optional modes, operations validated for a particular provider, and product paths that still need end-to-end qualification. This article is pinned to a public revision. Command readback and qualification evidence for the deployed version determine what a deployment can do.

+

A useful LoopX board should let a person and the next agent answer together: Where are we going? Who owns the work now? What can continue? What must wait? What is the evidence? How will the next step be validated? The core objects supply a shared language; long-running collaboration keeps it intact across sessions, interruptions, and changes in direction.

+

Implementation and documentation are pinned to a96c9aa91. The four diagrams and the bulk-export scenario are synthetic explanations, not production screenshots or performance data. This article does not claim equivalent qualification across every Host, provider, and product entry point.

+
+
+
+
+ + + diff --git a/apps/presentation/site/public/blog/application-scenarios/index.html b/apps/presentation/site/public/blog/application-scenarios/index.html new file mode 100644 index 0000000000..b7cd356482 --- /dev/null +++ b/apps/presentation/site/public/blog/application-scenarios/index.html @@ -0,0 +1,201 @@ + + + + + + + Where LoopX fits: complex tasks, open-ended exploration, and continuous delivery · LoopX + + + + + + + + + + + + + + + + + + + +
+
+

Scenarios and practice · Long-running agents

+

Where does LoopX fit?
Complex tasks, open-ended exploration, and continuous delivery

+

Some work is about finishing a defined task. Some is about finding the next promising direction. Some requires sustained responsibility for an outcome. Each calls for different domain capabilities—and for goals, evidence, and constraints that survive across sessions.

+
Ruiteng HuangThree scenarios · One long-horizon control plane
+
+
+ +
+
+

Start with the work, then choose the abstraction

+
+
01 / COMPLETION

Complex tasks with clear acceptance

Refactors, migrations, and protocol implementations. The destination is fairly clear even when the execution path is not.

Preserve: acceptance gaps, versions, repair evidence
Measure: completion rate and cost
+
02 / DISCOVERY

Open-ended exploration

Research, algorithm experiments, and system optimization. New evidence continuously changes the next route.

Preserve: hypotheses, counterevidence, candidate directions
Measure: useful discoveries and validation
+
03 / DELIVERY

Digital workers for continuous delivery

Take an issue, fix it, follow review, handle new feedback, and remain accountable for the result.

Preserve: responsibility, external state, waiting conditions
Measure: delivery quality and human effort
+
+

These categories can nest. An agent maintaining a repository over time may first explore a performance problem, then complete a fix with clear acceptance, and finally follow the pull request through review. The first two categories mainly describe how work is solved; the third adds an ongoing responsibility and a stream of newly arriving work.

+

What should persist is the work—and the reasoning behind its decisions.

+
+ +
+

01 / Finish a complex task

+

“Implement a decoder that conforms to the specification” sounds definite. Yet after the visible examples pass, edge inputs, memory safety, or compatibility may still be missing. Start a new session, and previously confirmed failure conditions may disappear from context.

+
+
Define acceptanceSpecification, baseline, budget
Implement and validateBind artifacts to a specific version
Find the gapsWhat remains beyond the tests?
Continue or stopClose gaps, or close out explicitly
+
+

LoopX keeps the goal and its acceptance gaps continuous across turns, then carries new evidence into the next decision. The model implements and judges; domain validators check the result; the control plane records which results were accepted and which work remains.

+

The best-fitting tasks usually span repeated validation, waiting, or handoffs. Existing agents may already be enough for small work that can be accepted in one session; an added control plane must justify its overhead.

+
+ +
+

Benchmarks: three studies, three signals worth investigating

+

The studies below observe continuation, delivery, and validation behavior under different tasks, model settings, and measurement rules. Each needs to be read on its own terms. The current evidence does not establish a universal LoopX gain.

+ +
+

SWE-Marathon: continuation must close real gaps

+

GPT-5.6 Sol / high · 15 matched tasks · 3 retained modes, one run per cell · timeout factor 0.3

+

Observation: native Goal completed 4/15; Heartbeat completed 5/15. Total cost for the two groups rose from $533 to $830, about 56%.

+

Insight: in the zstd case, continuation addressed acceptance beyond the visible tests; in the Excel case, repeated continuations still did not finish. The questions worth pursuing are which gap a next run closes and whether that gain justifies the added cost.

+

Boundary: each cell has one run and several mechanisms change together. There are both useful cases and ineffective continuation; stable gains or cost advantages have not been established.

+ +
+ +
+

DeepSWE × Sol: decompose the completion difference

+

Script default GPT-5.6 Sol / xhigh · 113 tasks · historical best-valid aggregation

+

Observation: bare Codex, native Goal, and Heartbeat completed 54, 60, and 70 tasks respectively. Heartbeat completed ten more than Goal, but its default time window was also longer.

+

Insight: the source separates “continue execution” from “deliver a valid result”: even when a worktree contains code, the collector may still receive an empty patch. Recovery, patch delivery, and independent acceptance should be examined separately before attributing their contribution to the completion difference.

+

Boundary: this is the best valid result per task, not a single-run pass rate; budgets differ, and attempts and cost are not fully disclosed. Unreverified merged scores cannot be treated as a net gain at equal budget.

+ +
+ +
+

DeepSWE × V4 Flash max: let counterexamples change the implementation

+

DeepSeek V4 Flash / max + Codex · frozen 113 tasks · Goal and LoopX compared under the same hint condition

+

Observation: LoopX had about two percentage points higher feature coverage per task on average. In a post-hoc long-duration slice where both groups had hints, completions rose from 12/29 to 14/29 while cumulative duration fell 16.4%.

+

Insight: the selected cases differ in whether failure counterexamples are retained and whether external contracts can overturn implementation assumptions. Validation matters when it changes the fix and triggers revalidation—not merely when it adds more checks.

+

Boundary: coverage is not success rate; the long-duration slice is post-hoc, and elapsed time includes successful and failed runs. Local findings cannot be generalized into an overall gain or equal-quality speedup.

+ +
+ +

The current signal is that useful continuation, reliable delivery, and counterexample-driven repair deserve further study.

+

The studies offer local positive observations while exposing cost, failure, and attribution problems. Matched budgets and repeated experiments are needed to determine which gains reproduce reliably.

+

Withdrawn SSH Goal and Codex CLI scores and conclusions from SWE-Marathon and DeepSWE × Sol are not used in these comparisons.

+
+ +
+

02 / Find a path through open-ended exploration

+

“Find a better algorithm” does not come with a fully specified task chain. Real progress may validate a hypothesis—or eliminate a promising-looking route. If only successful conclusions survive, the next run can easily repeat ideas that were already disproved.

+
+
Frame the questionConstraints and hypotheses to test
Try several routesIsolate experiments and bound cost
Record positive and negative evidenceSupport, refute, or raise a new question
Update the directionContinue, combine, retire, or pause
+
+

Explore provides an optional evidence graph and bounded branch planning: nodes represent questions and findings, while relationships express supports, refutes, and leads_to. Planning suggestions still pass through normal execution boundaries; the graph neither launches workers nor grants spending authority.

+

Auto Research organizes this pattern as research work: select a topic, propose hypotheses, run experiments, evaluate independently, and produce a report. Existing protocols and commands provide a foundation, while the public showcase still contains blueprints and items awaiting validation. Real effectiveness must be demonstrated on specific tasks.

+ +

Place RSI here, but keep the claim precise

+

When the object of improvement is the agent’s own tools, strategies, or harness, the work becomes self-improvement research. Changing its own code only creates a candidate; sustained improvement still requires independent evaluation, cross-task generalization, regression checks, and rollback.

+

Applying past experience to a later decision, automatically generating experiments, and modifying the harness are different levels. We can discuss an experimental path toward recursive self-improvement (RSI) here; we cannot equate “automatic iteration” with open-ended self-improvement already achieved.

+
+ +
+

03 / Own continuous delivery

+

For a pull-request or issue-fix digital worker, the job begins by deciding whether the issue is worth fixing. After delivery, CI, review feedback, branch changes, and the merge result still remain. New tasks keep arriving, and external facts keep changing.

+

Responsibility does not end when the patch is generated.

+

This synthetic scenario starts after a fix has been submitted while CI and review are still running. Choose an external state to see how the next step changes.

+ +
+
External fact
GitHub reports that checks are pending for the current revision.
+
Domain state
Record the pull request, revision, and check state; an unchanged observation does not create progress.
+
Capability proposal
Preserve the monitor and recovery condition, then wait for the result.
+
Control-plane check
Admit work by authority, budget, and eligibility; work outside the wait scope can still proceed.
+
Meaning for people
Stay quiet when nothing material changed, so polling does not become nagging.
+
+ +

This interactive diagram is a design explanation. It is not connected to a real repository and performs no GitHub operations.

+

Here, a “digital worker” means continuing responsibility, boundaries, memory, and a feedback loop. Its value should be judged by accepted fixes, reopenings and regressions, handling latency, delivery cost, and how often a person must repeatedly supervise it. Pull-request count and uptime are only process signals.

+
+ +
+

The richer the domain, the clearer the division of responsibility must be

+
+
Kernel
Authority and common lifecycle
Which goals and work items are valid, who may act, which authority and budgets apply, and how work is handed off, paused, and committed.
+
Domain State
Domain continuity
Whether this issue is actionable, which revision the pull request names, where CI and review stand, and which conclusions remain valid.
+
Capability
Outcome contract and judgment
Translate domain facts into a verifiable next step: repair, wait, ask a person, or close out—and define acceptance for the scenario.
+
Provider / Runtime
External I/O and execution
Call repositories, tests, and tools; perform authorized actions; and read back the result. GitHub still owns the external facts about code, CI, and review.
+
+

For checks failing, a domain capability can propose a successor task to repair CI. The fact that a repair is needed does not itself grant repository write or merge authority. In an experiment, check state becomes metrics and held-out conclusions; the common claim, authority, and recovery rules should remain consistent.

+

That is the value of a capability: it captures recurring judgments and outcome contracts for a scenario without forcing every business stage into the Kernel.

+
+ +
+

Evolution: add one verifiable capability at a time

+
    +
  1. One-shot execution → recoverable long-running taskRecover the goal, evidence, and next step across sessions and process interruptions. First prove one thing can be finished reliably.
  2. +
  3. Generic task → domain delivery and explorationUse issue-fix and research work for real outcomes, preserve domain state, and define what makes a result valuable.
  4. +
  5. One agent → small-team collaborationFirst have two or three workers complete two rounds of dependent delivery: receive artifacts, accept independently, absorb corrections, and return results.
  6. +
  7. Local collaboration → continuous work across hostsValidate shared state, authority revocation, stale-worker isolation, network failure, and shared budgets. Successful registration is not proof of collaboration.
  8. +
  9. Sustainable operation → measurable improvementUse outcome feedback to improve selection, memory, and strategy; prove gains through controlled experiments, held-out validation, and rollback. Qualify scaling separately.
  10. +
+

These are a build order and validation targets, not a completion checklist. Foundations, partial implementations, blueprints, and end-to-end qualification should be read separately.

+

I want what we hand to an agent to evolve from a single instruction into a work agreement it can keep fulfilling.

+

That agreement should remain legible and editable to people, actionable to agents, and auditable through evidence after failure.

+
+ +
+

Public sources and further reading

+

The technical content is derived only from public repository material. Source and data links are pinned to the revisions read for this article; historical experiments retain their own versions. This page introduces no new experimental results.

+
    +
  1. SWE-Marathon: continuous self-verification; setup, positive and negative cases, and limitations; public aggregate data. See the study for contributors and case provenance.
  2. +
  3. DeepSWE × Sol: from continued execution to valid delivery (Chinese). The standalone brief covers historical results over 113 tasks, mechanism diagrams, and pinned primary sources; research archive contribution: @gwh6669999, #4502.
  4. +
  5. DeepSWE × V4 Flash max: from hints to behavior; disclosure boundary; charts, cases, and metrics at the pinned revision.
  6. +
  7. Explore: evidence graphs, planning, and authority boundaries.
  8. +
  9. Auto Research: public blueprint and command path.
  10. +
  11. PR / Issue Fix: how State Kernel and domain state work together (Chinese).
  12. +
  13. Overall roadmap: product goals and independent acceptance milestones.
  14. +
  15. From one-shot agents to long-horizon control; LoopX: Native Kanban for long-running agents.
  16. +
+
+
+
+
+ + + + diff --git a/apps/presentation/site/public/blog/application-scenarios/presentation.js b/apps/presentation/site/public/blog/application-scenarios/presentation.js new file mode 100644 index 0000000000..b5aa60eee6 --- /dev/null +++ b/apps/presentation/site/public/blog/application-scenarios/presentation.js @@ -0,0 +1,28 @@ +const presentation = document.querySelector('#presentation'); +presentation.hidden = false; +document.querySelector('.stage-buttons').hidden = false; +presentation.addEventListener('click', () => { + const enabled = document.body.classList.toggle('presenting'); + presentation.setAttribute('aria-pressed', String(enabled)); + presentation.textContent = enabled ? 'Reading mode' : 'Presentation mode'; +}); +const states = { + pending: ['GitHub reports that checks are pending for the current revision.','Record the pull request, revision, and check state; an unchanged observation does not create progress.','Preserve the monitor and recovery condition, then wait for the result.','Admit work by authority, budget, and eligibility; work outside the wait scope can still proceed.','Stay quiet when nothing material changed, so polling does not become nagging.'], + failed: ['CI failed on the current revision.','Distinguish a code regression, an environment failure, and an unknown cause; preserve the evidence.','Propose a bounded repair successor with a clear acceptance method.','Recheck the actor, write scope, and budget; execute only after eligibility is established.','Turn the failure into work someone can take over, not merely an error message.'], + changed: ['The pull request changed from revision A to revision B.','The original review and tests are bound to A and cannot establish that B qualifies.','Reconfirm the change and validation scope; repeat review when needed.','Preserve version lineage and current authority; an old receipt cannot stand in for a new result.','Keep every “pass” attached to an unambiguous deliverable.'], + merged: ['GitHub confirms that the pull request was merged.','Record the terminal state and result; a merge does not automatically complete the whole goal.','Check goal acceptance: add integration validation, continue with a successor, or explain why no successor is needed.','Settle the current monitor and successor; publication and other actions still require their own authority.','Give every piece of work an explicit destination, and do not keep searching for work by inertia after completion.'] +}; +document.querySelectorAll('[data-state]').forEach(button => button.addEventListener('click', () => { + document.querySelectorAll('[data-state]').forEach(b => b.setAttribute('aria-pressed',String(b === button))); + ['fact','domain','proposal','kernel','meaning'].forEach((id,i) => document.getElementById(id).textContent = states[button.dataset.state][i]); +})); +document.addEventListener('keydown', event => { + if (!document.body.classList.contains('presenting') || event.altKey || event.ctrlKey || event.metaKey) return; + if (event.key === 'Escape') { presentation.click(); return; } + if (event.target.closest('input,textarea,select,[contenteditable=true]')) return; + if (!['ArrowRight','ArrowLeft'].includes(event.key)) return; + const sections = [...document.querySelectorAll('article > section')]; + let index = sections.findIndex(s => s.getBoundingClientRect().bottom > 100); + index = Math.max(0, Math.min(sections.length - 1, index + (event.key === 'ArrowRight' ? 1 : -1))); + event.preventDefault(); sections[index].scrollIntoView({behavior: 'instant'}); history.replaceState(null,'','#'+sections[index].id); +}); diff --git a/apps/presentation/site/public/blog/images/agent-facing-kanban/authority-en.svg b/apps/presentation/site/public/blog/images/agent-facing-kanban/authority-en.svg new file mode 100644 index 0000000000..f5472a158d --- /dev/null +++ b/apps/presentation/site/public/blog/images/agent-facing-kanban/authority-en.svg @@ -0,0 +1,44 @@ + +04 Shared authority state: which facts count? +CLI, frontend, and Lark use existing entry points. A typed lifecycle owner validates operations against the selected authority source. Unpromoted and canonical state remain distinct. File, SQLite, and PostgreSQL are alternatives that qualify separately. Receipts and read-only projections sit downstream. + + + + + + + +04 Shared authority state: which facts count? +Entry points may differ; every write must know which rules and current state it follows. + +CLI +Commands and structured readback + +Frontend +Board and details + +Lark +Messages / configured adapters + + +Typed lifecycle owner · Receive and validate legal operations +Identity + authority → current version + preconditions → commit / reject → receipt + + +The Goal's provider selection and promotion state determine authority +Unpromoted: existing Markdown / local-writer rules +Canonical path: the selected provider owns its transactions + +File · qualify separately + +SQLite · qualify separately + +PostgreSQL · qualify separately + + +Receipts + read-only projections · state / board / bounded evidence timeline +Projections retain source and identity; provider failure does not silently create another writer. +Selecting a provider does not promote a Goal; these are alternatives, not parallel primary stores. +LOOPX / KANBAN FOUNDATIONS · SYNTHETIC EXAMPLE + + diff --git a/apps/presentation/site/public/blog/images/agent-facing-kanban/board-en.svg b/apps/presentation/site/public/blog/images/agent-facing-kanban/board-en.svg new file mode 100644 index 0000000000..b5b3844cb4 --- /dev/null +++ b/apps/presentation/site/public/blog/images/agent-facing-kanban/board-en.svg @@ -0,0 +1,49 @@ + +02 Read the current situation on one board +A synthetic board: integration validation is claimed and has run evidence; release waits on a scoped Gate; implementation and review have results. The Goal is not complete. Columns are explanatory projections, not persisted state enums. + + +02 Read the current situation on one board +Goal · Ship bulk export safely / Missing: large-volume validation, release confirmation + + + +Can proceed +Awaiting decision +Results available + +C · Integration validation +Owner: validation agent +Claim: held +Lease: valid (enabled here) +Run: tests in progress +Next: return result + evidence + +Monitor · validation result +Observe when due; +quiet if unchanged + +D · Release candidate +Release Gate: unconfirmed +Scope: release decision +Prerequisite: acceptance passed +Next: release when eligible + +Blocks only matching work +Independent validation continues + +A · Export implementation +State: done · revision A +Evidence: isolation test passed +Successor: B independent review + +B · Independent review +Owner: review agent +Evidence: revision A approved +Successor: C integration test +One Todo is presented by each agent's eligibility and responsibility. +Column names explain; they are not persisted states. Claimed is not running; execution needs evidence. +One completed card does not mean the Goal has been accepted. +LOOPX / KANBAN FOUNDATIONS · SYNTHETIC EXAMPLE + + diff --git a/apps/presentation/site/public/blog/images/agent-facing-kanban/lease-en.svg b/apps/presentation/site/public/blog/images/agent-facing-kanban/lease-en.svg new file mode 100644 index 0000000000..cead7f6a39 --- /dev/null +++ b/apps/presentation/site/public/blog/images/agent-facing-kanban/lease-en.svg @@ -0,0 +1,39 @@ + +03 Claim and Lease: from responsibility to valid execution +A five-step timeline: claim, acquire a lease, renew it, reassess after expiry, and take over when eligible. Renewal does not change ownership or scope. Takeover is not automatic at expiry, and the lease protects only paths integrated with fencing. + + +03 Claim and Lease: from responsibility to valid execution +A claim expresses responsibility; a lease further constrains execution. This path assumes lease qualification. + + + +1 +Claim the work +Agent A owns the Todo; a claim alone does not prove its process is alive. + + +2 +Acquire a valid lease +Execution A holds the current expiry, version, epoch, and bounded write scope. + + +3 +Renew before expiry +Verify holder and expected version; advance version and expiry while preserving scope. + + +4 +Reassess after expiry +Expiry does not kill a process; check grace, admission, and recovery before takeover. + + +5 +Execute only after an eligible takeover +The new run uses a new lease identity; protected writes from the old epoch are rejected. + +Historical receipt ≠ current lease · Valid lease ≠ release authority · Fencing ≠ exactly-once effects +Read current facts before acting; the corresponding Runtime and system still govern processes and effects. +LOOPX / KANBAN FOUNDATIONS · SYNTHETIC EXAMPLE + + diff --git a/apps/presentation/site/public/blog/images/agent-facing-kanban/objects-en.svg b/apps/presentation/site/public/blog/images/agent-facing-kanban/objects-en.svg new file mode 100644 index 0000000000..37dbdd53e3 --- /dev/null +++ b/apps/presentation/site/public/blog/images/agent-facing-kanban/objects-en.svg @@ -0,0 +1,46 @@ + +01 One board, several different questions +A Goal defines the outcome and acceptance, while Todos organize work. Claims express ownership, leases constrain execution, gates capture decisions, dependencies capture prerequisite facts, evidence supports judgment, lanes organize views, and authority state determines which facts count. + + + + + + + +01 One board, several different questions +Read from outcomes, work, and constraints through to shared facts. + +Goal · Deliver a usable bulk export +Acceptance · Works at scale, isolates permissions, and ships only after confirmation + + +Claim · Who owns it? +Dev agent claims the work +Lease · Which run is valid? +Optional: expiry, version, scope + +Todo · Deliverable work +Build export + permission test +Stable id · state · next step +Recheck Goal after completion + +Gate · Which decision? +Owner confirms release +Dependency · Which fact? +Example: review passed + + + + +Evidence · Why believe it? +Tests, review, and artifacts for revision A + +Lane · How do I read it? +Runnable / waiting / monitor / owned elsewhere + +Shared authority state · The accepted facts and write rules +The board reads and expands these facts; display columns do not grant authority. +LOOPX / KANBAN FOUNDATIONS · SYNTHETIC EXAMPLE + + diff --git a/apps/presentation/site/public/blog/index.html b/apps/presentation/site/public/blog/index.html index 9e0bd43dc3..bd1765877f 100644 --- a/apps/presentation/site/public/blog/index.html +++ b/apps/presentation/site/public/blog/index.html @@ -32,6 +32,14 @@

LoopX · Blog

Engineering the long run.

Design, engineering, and field notes on long-horizon agents.

+
+

Scenarios · Practice and research

September 18, 2026

+

Where LoopX fits: complex tasks, open-ended exploration, and continuous delivery

See how goals, evidence, and delivery responsibility work across three scenarios, alongside signals from SWE-Marathon, DeepSWE × Sol, and V4 Flash max.

Read the article
+
+
+

Agent-native Kanban · Long-running collaboration

September 15, 2026

+

LoopX: Native Kanban for long-running agents

Four diagrams explain Goals, Todos, claims, leases, gates, evidence, and shared authority state—and how they support long-running collaboration.

Read the article
+

Architecture · Design notes

September 2026

From one-shot agents to long-horizon control

How LoopX uses goals, a state kernel, and Effect Programs to carry work across sessions, preserve human judgment, and recover from interruptions.

Read the article
diff --git a/apps/presentation/site/public/blog/zh/agent-facing-kanban/index.html b/apps/presentation/site/public/blog/zh/agent-facing-kanban/index.html index 75ccaa193c..c70a2ab4b1 100644 --- a/apps/presentation/site/public/blog/zh/agent-facing-kanban/index.html +++ b/apps/presentation/site/public/blog/zh/agent-facing-kanban/index.html @@ -7,6 +7,9 @@ LoopX:长程 Agent 的原生 Kanban · LoopX + + + @@ -22,7 +25,7 @@
diff --git a/apps/presentation/site/public/blog/zh/application-scenarios/index.html b/apps/presentation/site/public/blog/zh/application-scenarios/index.html index 86dc536acb..307bfa74fa 100644 --- a/apps/presentation/site/public/blog/zh/application-scenarios/index.html +++ b/apps/presentation/site/public/blog/zh/application-scenarios/index.html @@ -7,6 +7,9 @@ LoopX 用在哪里:复杂任务、开放探索与持续交付 + + + @@ -41,7 +44,7 @@
From adf476f768cf6698b3e2bbc3191e94b5d038e819 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Fri, 18 Sep 2026 22:50:47 +0800 Subject: [PATCH 2/3] test(site): enforce bilingual blog completeness Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- apps/presentation/site/README.md | 4 ++ examples/frontstage-share-bundle-smoke.mjs | 49 +++++++++++++++++++--- 2 files changed, 48 insertions(+), 5 deletions(-) diff --git a/apps/presentation/site/README.md b/apps/presentation/site/README.md index b4a27ec8ad..f7f6b79b53 100644 --- a/apps/presentation/site/README.md +++ b/apps/presentation/site/README.md @@ -19,6 +19,10 @@ alternate-language metadata, and the complete article in HTML. Reading and navigation work without JavaScript. Vite copies these pages into both the local build and the existing Pages export; no separate hosting or content service is needed. Relative navigation supports both root and repository base paths. +Both locale indexes list every article. The +frontstage share-bundle smoke discovers article directories rather than relying +on a hand-maintained allowlist, and fails when a translation, index entry, or +language alternate is missing or when paired article sections drift. The Chinese DeepSWE × Sol research brief is a static page at `public/benchmarks/deepswe-sol/`, linked from the homepage research collection. diff --git a/examples/frontstage-share-bundle-smoke.mjs b/examples/frontstage-share-bundle-smoke.mjs index 58dd6a04f1..962df8c35a 100644 --- a/examples/frontstage-share-bundle-smoke.mjs +++ b/examples/frontstage-share-bundle-smoke.mjs @@ -133,9 +133,44 @@ for (const route of ["benchmarks/deepswe-sol/"]) { } // Editorial pages must ship their text and locale navigation without an SPA // fallback or client-side execution, including on repository-base hosting. -const blogArticle = "from-one-shot-agents-to-long-horizon-control/"; +const blogDir = resolve(siteDir, "blog"); +async function collectBlogArticleSlugs(locale) { + const localeDir = resolve(blogDir, locale); + const entries = await readdir(localeDir, { withFileTypes: true }); + return entries + .filter( + (entry) => + entry.isDirectory() && + entry.name !== "zh" && + existsSync(resolve(localeDir, entry.name, "index.html")), + ) + .map((entry) => entry.name) + .sort(); +} +const englishBlogArticles = await collectBlogArticleSlugs(""); +const chineseBlogArticles = await collectBlogArticleSlugs("zh"); +if (JSON.stringify(englishBlogArticles) !== JSON.stringify(chineseBlogArticles)) { + throw new Error( + `Every Blog article must ship paired English and Chinese editions: en=${englishBlogArticles.join(",")} zh=${chineseBlogArticles.join(",")}`, + ); +} +for (const slug of englishBlogArticles) { + const englishHtml = await readFile(resolve(blogDir, slug, "index.html"), "utf8"); + const chineseHtml = await readFile(resolve(blogDir, "zh", slug, "index.html"), "utf8"); + const sectionIds = (html) => [...html.matchAll(/ match[1]); + if (JSON.stringify(sectionIds(englishHtml)) !== JSON.stringify(sectionIds(chineseHtml))) { + throw new Error(`Paired Blog article sections must match: ${slug}`); + } +} for (const locale of ["", "zh/"]) { - const articles = locale ? ["", blogArticle, "agent-facing-kanban/", "application-scenarios/"] : ["", blogArticle]; + const articles = ["", ...englishBlogArticles.map((slug) => `${slug}/`)]; + const indexPath = resolve(blogDir, locale, "index.html"); + const indexHtml = await readFile(indexPath, "utf8"); + for (const slug of englishBlogArticles) { + if (!indexHtml.includes(`href="${slug}/"`)) { + throw new Error(`Blog index must link every ${locale || "English "}article: ${slug}`); + } + } for (const article of articles) { const pagePath = resolve(siteDir, "blog", locale, article, "index.html"); assertExists(pagePath); @@ -152,11 +187,15 @@ for (const locale of ["", "zh/"]) { } assertExists(resolve(dirname(pagePath), "presentation.js")); } - // A single-language article must not advertise a nonexistent translation. - const alternates = ["agent-facing-kanban/", "application-scenarios/"].includes(article) ? [] : ["en", "zh-CN", "x-default"]; - for (const hreflang of alternates) { + for (const hreflang of ["en", "zh-CN", "x-default"]) { if (!html.includes(`hreflang="${hreflang}"`)) throw new Error(`Missing Blog language alternate: ${hreflang}`); } + if (article) { + const counterpart = locale ? `../../../blog/${article}` : `../../blog/zh/${article}`; + if (!html.includes(`href="${counterpart}"`)) { + throw new Error(`Blog article must link its paired edition: ${pagePath}`); + } + } const stylesheet = html.match(/ Date: Fri, 18 Sep 2026 23:02:32 +0800 Subject: [PATCH 3/3] test(site): add fast bilingual blog catalog check Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .github/workflows/frontstage-pages.yml | 5 ++ apps/presentation/site/README.md | 9 ++- examples/blog-bilingual-index-smoke.mjs | 87 ++++++++++++++++++++++ examples/frontstage-share-bundle-smoke.mjs | 46 +----------- 4 files changed, 99 insertions(+), 48 deletions(-) create mode 100644 examples/blog-bilingual-index-smoke.mjs diff --git a/.github/workflows/frontstage-pages.yml b/.github/workflows/frontstage-pages.yml index 9c8bf78ca9..4fe7140b1d 100644 --- a/.github/workflows/frontstage-pages.yml +++ b/.github/workflows/frontstage-pages.yml @@ -25,6 +25,7 @@ on: - "docs/assets/long-running-loop-openviking-trajectory.png" - "docs/assets/long-running-loop-ml-experiment-trajectory.png" - "examples/export-frontstage-share-bundle.mjs" + - "examples/blog-bilingual-index-smoke.mjs" - "examples/frontstage-share-bundle-smoke.mjs" - "examples/dev-book-browser-smoke.mjs" - "examples/readme-star-history-smoke.py" @@ -60,6 +61,7 @@ on: - "docs/assets/long-running-loop-openviking-trajectory.png" - "docs/assets/long-running-loop-ml-experiment-trajectory.png" - "examples/export-frontstage-share-bundle.mjs" + - "examples/blog-bilingual-index-smoke.mjs" - "examples/frontstage-share-bundle-smoke.mjs" - "examples/dev-book-browser-smoke.mjs" - "examples/readme-star-history-smoke.py" @@ -148,6 +150,9 @@ jobs: - name: Validate Developer Book publication run: python3 examples/dev-book-publication-smoke.py + - name: Validate bilingual Blog catalog + run: node examples/blog-bilingual-index-smoke.mjs + - name: Validate public-safe frontstage bundle working-directory: apps/presentation/dashboard run: npm run smoke:frontstage-share-bundle diff --git a/apps/presentation/site/README.md b/apps/presentation/site/README.md index f7f6b79b53..c06f682feb 100644 --- a/apps/presentation/site/README.md +++ b/apps/presentation/site/README.md @@ -19,10 +19,11 @@ alternate-language metadata, and the complete article in HTML. Reading and navigation work without JavaScript. Vite copies these pages into both the local build and the existing Pages export; no separate hosting or content service is needed. Relative navigation supports both root and repository base paths. -Both locale indexes list every article. The -frontstage share-bundle smoke discovers article directories rather than relying -on a hand-maintained allowlist, and fails when a translation, index entry, or -language alternate is missing or when paired article sections drift. +Both locale indexes list every article. The focused bilingual Blog smoke and +the frontstage share-bundle smoke discover article directories rather than +relying on a hand-maintained allowlist. They fail when a translation, index +entry, language alternate, counterpart link, or paired article section is +missing. The Chinese DeepSWE × Sol research brief is a static page at `public/benchmarks/deepswe-sol/`, linked from the homepage research collection. diff --git a/examples/blog-bilingual-index-smoke.mjs b/examples/blog-bilingual-index-smoke.mjs new file mode 100644 index 0000000000..b9454adebe --- /dev/null +++ b/examples/blog-bilingual-index-smoke.mjs @@ -0,0 +1,87 @@ +#!/usr/bin/env node +// Fast source-level contract test for the bilingual static Blog catalog. + +import { existsSync } from "node:fs"; +import { readFile, readdir } from "node:fs/promises"; +import { dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; + +async function collectArticleSlugs(localeDir, { exclude = [] } = {}) { + const entries = await readdir(localeDir, { withFileTypes: true }); + return entries + .filter( + (entry) => + entry.isDirectory() && + !exclude.includes(entry.name) && + existsSync(resolve(localeDir, entry.name, "index.html")), + ) + .map((entry) => entry.name) + .sort(); +} + +function assertIncludes(html, value, message) { + if (!html.includes(value)) throw new Error(message); +} + +function sectionIds(html) { + return [...html.matchAll(/ match[1]); +} + +export async function validateBilingualBlog(blogDir) { + const englishSlugs = await collectArticleSlugs(blogDir, { exclude: ["zh"] }); + const chineseSlugs = await collectArticleSlugs(resolve(blogDir, "zh")); + if (JSON.stringify(englishSlugs) !== JSON.stringify(chineseSlugs)) { + throw new Error( + `Every Blog article must ship paired English and Chinese editions: en=${englishSlugs.join(",")} zh=${chineseSlugs.join(",")}`, + ); + } + + const locales = [ + { + directory: blogDir, + language: "en", + counterpartHref: (slug) => `../../blog/zh/${slug}/`, + canonicalHref: (slug) => `https://huangruiteng.github.io/loopx/blog/${slug}/`, + }, + { + directory: resolve(blogDir, "zh"), + language: "zh-CN", + counterpartHref: (slug) => `../../../blog/${slug}/`, + canonicalHref: (slug) => `https://huangruiteng.github.io/loopx/blog/zh/${slug}/`, + }, + ]; + + for (const locale of locales) { + const indexHtml = await readFile(resolve(locale.directory, "index.html"), "utf8"); + assertIncludes(indexHtml, ``, `Blog index language drifted: ${locale.directory}`); + for (const slug of englishSlugs) { + assertIncludes(indexHtml, `href="${slug}/"`, `Blog index must link every ${locale.language} article: ${slug}`); + const articleHtml = await readFile(resolve(locale.directory, slug, "index.html"), "utf8"); + assertIncludes(articleHtml, ``, `Blog article language drifted: ${slug}`); + assertIncludes(articleHtml, "

", `Blog article must contain a visible title: ${slug}`); + assertIncludes(articleHtml, `rel="canonical" href="${locale.canonicalHref(slug)}"`, `Blog canonical URL drifted: ${slug}`); + assertIncludes(articleHtml, `href="${locale.counterpartHref(slug)}"`, `Blog article must link its paired edition: ${slug}`); + for (const hreflang of ["en", "zh-CN", "x-default"]) { + assertIncludes(articleHtml, `hreflang="${hreflang}"`, `Blog article is missing ${hreflang}: ${slug}`); + } + } + } + + for (const slug of englishSlugs) { + const englishHtml = await readFile(resolve(blogDir, slug, "index.html"), "utf8"); + const chineseHtml = await readFile(resolve(blogDir, "zh", slug, "index.html"), "utf8"); + if (JSON.stringify(sectionIds(englishHtml)) !== JSON.stringify(sectionIds(chineseHtml))) { + throw new Error(`Paired Blog article sections must match: ${slug}`); + } + } + + return { articleSlugs: englishSlugs }; +} + +const modulePath = fileURLToPath(import.meta.url); +if (process.argv[1] && resolve(process.argv[1]) === modulePath) { + const repoRoot = resolve(dirname(modulePath), ".."); + const blogDir = resolve(repoRoot, "apps/presentation/site/public/blog"); + const { articleSlugs } = await validateBilingualBlog(blogDir); + console.log(`blog-bilingual-index-smoke: ok (${articleSlugs.length} paired articles)`); +} diff --git a/examples/frontstage-share-bundle-smoke.mjs b/examples/frontstage-share-bundle-smoke.mjs index 962df8c35a..2d97dfea62 100644 --- a/examples/frontstage-share-bundle-smoke.mjs +++ b/examples/frontstage-share-bundle-smoke.mjs @@ -6,6 +6,7 @@ import { readFile, readdir, rm, stat } from "node:fs/promises"; import { existsSync } from "node:fs"; import { dirname, resolve } from "node:path"; import { fileURLToPath } from "node:url"; +import { validateBilingualBlog } from "./blog-bilingual-index-smoke.mjs"; const repoRoot = resolve(dirname(fileURLToPath(import.meta.url)), ".."); const outDir = resolve("/tmp", "loopx-frontstage-share-bundle-smoke"); @@ -134,43 +135,9 @@ for (const route of ["benchmarks/deepswe-sol/"]) { // Editorial pages must ship their text and locale navigation without an SPA // fallback or client-side execution, including on repository-base hosting. const blogDir = resolve(siteDir, "blog"); -async function collectBlogArticleSlugs(locale) { - const localeDir = resolve(blogDir, locale); - const entries = await readdir(localeDir, { withFileTypes: true }); - return entries - .filter( - (entry) => - entry.isDirectory() && - entry.name !== "zh" && - existsSync(resolve(localeDir, entry.name, "index.html")), - ) - .map((entry) => entry.name) - .sort(); -} -const englishBlogArticles = await collectBlogArticleSlugs(""); -const chineseBlogArticles = await collectBlogArticleSlugs("zh"); -if (JSON.stringify(englishBlogArticles) !== JSON.stringify(chineseBlogArticles)) { - throw new Error( - `Every Blog article must ship paired English and Chinese editions: en=${englishBlogArticles.join(",")} zh=${chineseBlogArticles.join(",")}`, - ); -} -for (const slug of englishBlogArticles) { - const englishHtml = await readFile(resolve(blogDir, slug, "index.html"), "utf8"); - const chineseHtml = await readFile(resolve(blogDir, "zh", slug, "index.html"), "utf8"); - const sectionIds = (html) => [...html.matchAll(/ match[1]); - if (JSON.stringify(sectionIds(englishHtml)) !== JSON.stringify(sectionIds(chineseHtml))) { - throw new Error(`Paired Blog article sections must match: ${slug}`); - } -} +const { articleSlugs: englishBlogArticles } = await validateBilingualBlog(blogDir); for (const locale of ["", "zh/"]) { const articles = ["", ...englishBlogArticles.map((slug) => `${slug}/`)]; - const indexPath = resolve(blogDir, locale, "index.html"); - const indexHtml = await readFile(indexPath, "utf8"); - for (const slug of englishBlogArticles) { - if (!indexHtml.includes(`href="${slug}/"`)) { - throw new Error(`Blog index must link every ${locale || "English "}article: ${slug}`); - } - } for (const article of articles) { const pagePath = resolve(siteDir, "blog", locale, article, "index.html"); assertExists(pagePath); @@ -187,15 +154,6 @@ for (const locale of ["", "zh/"]) { } assertExists(resolve(dirname(pagePath), "presentation.js")); } - for (const hreflang of ["en", "zh-CN", "x-default"]) { - if (!html.includes(`hreflang="${hreflang}"`)) throw new Error(`Missing Blog language alternate: ${hreflang}`); - } - if (article) { - const counterpart = locale ? `../../../blog/${article}` : `../../blog/zh/${article}`; - if (!html.includes(`href="${counterpart}"`)) { - throw new Error(`Blog article must link its paired edition: ${pagePath}`); - } - } const stylesheet = html.match(/