From 388b1fb7700d073a37826e15dc697bd170816c8b Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Thu, 1 Oct 2026 06:34:59 +0800 Subject: [PATCH 1/7] refactor(aidd-refine): simplify improve instructions --- .../aidd-refine/skills/05-improve/SKILL.md | 17 ++---- .../actions/01-read-conversation.md | 29 ++++------ .../skills/05-improve/actions/02-recommend.md | 57 ++++++++----------- .../05-improve/actions/03-target-edits.md | 28 ++++----- 4 files changed, 54 insertions(+), 77 deletions(-) diff --git a/plugins/aidd-refine/skills/05-improve/SKILL.md b/plugins/aidd-refine/skills/05-improve/SKILL.md index 61f51420e..d2da36217 100644 --- a/plugins/aidd-refine/skills/05-improve/SKILL.md +++ b/plugins/aidd-refine/skills/05-improve/SKILL.md @@ -8,18 +8,14 @@ argument-hint: conversation | export ```mermaid flowchart LR start([conversation ID or export]) --> read-conversation - read-conversation -->|complete| scopes{two or more scopes?} + read-conversation -->|complete| recommend --> target-edits read-conversation -->|unavailable| unavailable([stop]) - scopes -->|no| recommend-local[recommend locally] --> target-edits - scopes -->|yes| isolation{isolated artifact context?} - isolation -->|yes| recommend-parallel[recommend in parallel] --> target-edits - isolation -->|no| recommend-local target-edits --> question([ask next intent]) --> stop([stop]) ``` ## Actions -Run the flow above. Read only the next action file. +Run all three actions without confirmation. Read only the next action file. | Action | Does | | --- | --- | @@ -29,8 +25,7 @@ Run the flow above. Read only the next action file. ## Transversal rules -- After resolving the source, run all three actions without pausing for confirmation. -- Analyze only a complete conversation and the exact skills or documents it names. -- Exclude every current or previous `improve` invocation and report from the analysis evidence. -- Never invent time, tokens, source coverage, or document status. -- Do not modify, stage, or persist any project file. +- Stop if the exact complete transcript is unavailable. +- Assess only named artifacts alongside conversation behavior. +- Never invent metrics, coverage, status, or private reasoning. +- Keep the project read-only; write only the unique temporary report. diff --git a/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md b/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md index 1b146ce3a..a45802b7e 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md +++ b/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md @@ -1,6 +1,6 @@ # 01 - Read conversation -Load complete conversation evidence once and profile its visible cost. +Freeze complete evidence and measure visible cost. ## Input @@ -12,24 +12,19 @@ An evidence boundary, a `## Timing` table with `Activity | Observed time | Share ## Process -1. **Resolve.** Choose the host-specific transcript source from [conversation sources](../assets/conversation-sources.md). - - Stop when no complete transcript can be resolved for the exact conversation. -2. **Bound.** Freeze the evidence at the message before the current invocation, or at the supplied export boundary. -3. **Read.** Load every in-scope message, tool call, tool result, and timestamp from that transcript. -4. **Index.** Record relevant turns and invoked or named skills and knowledge files for downstream analysis. -5. **Measure.** Calculate visible time and usage only from timestamps, elapsed records, or host usage data. - - Group explicit tool activity as `research and diagnosis`, `implementation`, or `validation`; keep mixed or unknown time `unattributed`. - - Include tokens, requests, and monetary cost only when the source exposes them. - - Never label unattributed wait as reasoning time. -6. **Render.** Mark a missing metric `unavailable` and order known activity times from longest to shortest. +1. **Resolve.** Use the host route in [conversation sources](../assets/conversation-sources.md); stop unless it yields the exact complete transcript. +2. **Freeze.** End before this invocation, or at the export boundary. Exclude every `improve` invocation and report. +3. **Read once.** Load all in-scope messages, tool calls, results, and timestamps. Index relevant turns and named or invoked skill and knowledge paths. +4. **Measure.** Use only timestamps, elapsed records, and exposed host usage. + - Group tool time as `research and diagnosis`, `implementation`, `validation`, or `unattributed`. + - Include tokens, requests, and cost only when exposed. +5. **Render.** Order known activity times descending; mark missing metrics `unavailable`. Never call unattributed or unavailable time private reasoning. ## Test | Case | Pass | | --- | --- | -| A conversation is analyzed | its source resolves to one exact complete transcript | -| The same conversation is analyzed again | its evidence excludes every `improve` invocation and report | -| A timing metric is shown | its evidence identifies visible timestamp or elapsed records | -| A metric is unavailable | the report does not estimate or call it reasoning time | -| A usage metric is shown | its evidence identifies the host record that exposes it | -| A scope is indexed | it identifies exact turns or artifact paths, not a generated summary | +| Conversation | exact complete transcript; frozen boundary; no `improve` evidence | +| Metric | cited host record, timestamp, or elapsed record; otherwise `unavailable` | +| Time | visible categories only; no private-reasoning claim | +| Scope | exact turns or artifact paths, never a generated summary | diff --git a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md index 6e8f2681f..02de4fe1b 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md +++ b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md @@ -1,6 +1,6 @@ # 02 - Recommend -Analyze each relevant scope with the smallest grounded change. +Find the smallest grounded improvements. ## Input @@ -12,39 +12,32 @@ A `## Recommendations` table with `ID | Question | Type | Diagnostic | Evidence ## Process -1. **Scope.** Select `behavior`, plus `skill` and `knowledge` only when the scope index names relevant artifacts. -2. **Dispatch.** Analyze locally when one scope exists; otherwise dispatch one read-only analyst per scope in parallel. - - Prefer a lightweight available model and low reasoning effort when the host supports per-agent overrides; otherwise inherit the run defaults. - - Give the behavior analyst the complete frozen transcript; give artifact analysts the same boundary, indexed turns, and exact artifact paths. - - Request isolated or minimal context for artifact analysts when supported; otherwise analyze every scope locally instead of duplicating the transcript. - - Do not dispatch an analyst with no relevant evidence. - - Require table rows only and forbid file writes. -3. **Question.** Make each analyst answer every prompt for its scope. - - How could the next run be faster or better? - - What information should be removed or clarified? - - Where should the change live? - - How could it save time or tokens? - - What work was counterproductive? -4. **Verify.** Read a named skill or knowledge file before assessing its information. -5. **Assess.** Label relevant information `obsolete`, `over-specific-or-time-bound`, `duplicate`, `inconsistent`, `counterproductive`, or `correct`. - - Use `correct` when no evidence supports another label, and never render it as a recommendation. -6. **Merge.** Deduplicate findings across scopes and verify only their cited evidence against the frozen source. -7. **Render.** Order by question then `behavior`, `skill`, `knowledge`, and describe each change with the fewest unambiguous words. - - Use `skill`, `behavior`, `knowledge`, or `tooling` as the target type. - - State `time`, `tokens`, `both`, or `unknown` as its saving. - - Render `no change` when a scope has no evidence-backed recommendation. +1. **Scope.** Always assess `behavior`; add `skill` or `knowledge` only for indexed artifacts. +2. **Dispatch.** With multiple scopes and isolated or minimal context, run one read-only analyst per scope in parallel. Otherwise analyze locally. + - Prefer a cheap model and low reasoning effort when per-agent overrides are supported; otherwise inherit host defaults. + - Give the behavior analyst the frozen transcript; give each artifact analyst the boundary, indexed turns, and paths. + - Give every analyst steps 3–7; require table rows and no writes. +3. **Question.** Answer all for each scope: + - Faster or better next run? What was removable, unclear, or counterproductive? + - Where should change live? Would it save `time`, `tokens`, `both`, or `unknown`? + - Which back-and-forth, bottlenecks, or tool calls could be removed, batched, parallelized, or replaced? +4. **Verify.** Read each named artifact before judging it. +5. **Assess.** Use `obsolete`, `over-specific-or-time-bound`, `duplicate`, `inconsistent`, `counterproductive`, or `correct`. + - For skills, especially knowledge: delete evidence-backed waste or inconsistency first, without quota; then consolidate or clarify. + - Add only for a demonstrated gap the existing content cannot cover. Preserve useful context and requirements. + - `correct` means `no change`, never a recommendation. +6. **Merge.** Deduplicate, then verify cited evidence against the frozen source. +7. **Render.** Order by question, then `behavior`, `skill`, `knowledge`. Use the fewest actionable words. + - Type: `skill`, `behavior`, `knowledge`, or `tooling`. + - Saving: `time`, `tokens`, `both`, or `unknown`. + - Use `no change` when evidence supports none. ## Test | Case | Pass | | --- | --- | -| A question is shown | it is answered from conversation evidence | -| One relevant scope exists | no parallel analyst is dispatched | -| Several relevant scopes exist | their analysts run in parallel and return the same columns | -| An artifact analyst cannot receive isolated context | every scope is analyzed locally instead of duplicating the transcript | -| A type is shown | it is `skill`, `behavior`, `knowledge`, or `tooling` | -| Information is assessed | it has one allowed label backed by evidence | -| Information is correct | its scope says `no change` and no recommendation is rendered | -| A named target is assessed | the target file was read before the verdict | -| A saving is shown | it is categorical and never an invented amount | -| The same evidence is analyzed again | recommendations keep the same order and do not cite an earlier `improve` report | +| Dispatch | parallel only for multiple isolated scopes; otherwise local | +| Analysis | every question answered with exact evidence or `no change` | +| Finding | exact turn, tool-call, or read-artifact evidence | +| No finding | `no change`; no false recommendation | +| Output | allowed labels, types, savings, and stable order; no `improve` evidence | diff --git a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md index 53bc43068..71fffdf66 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md +++ b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md @@ -1,6 +1,6 @@ # 03 - Target edits -Map recommendations to minimal edits and render the report. +Map minimal edits and render the report. ## Input @@ -12,24 +12,18 @@ An HTML report in a temporary directory with local `report.css` and `report.js`, ## Process -1. **Target.** Map each file-targeted recommendation to a real project path. - - Omit behavior-only recommendations from this table. -2. **Summarize.** Build a `Fichier | + Ajout | − Retrait ou clarification` table. - - Consolidate repeated targets and render each diagnostic's smallest change. - - Render one `Aucun fichier recommandé | — | —` row when no recommendation needs a file edit. -3. **Render.** Fill [the report template](../assets/report-template.html) with only measured values and grounded findings, then copy its local CSS and JavaScript beside it. - - Remove every sample value and sample finding from the produced report. - - HTML-escape every injected value; allow only template-owned markup and local asset references. - - Write only to a unique temporary directory, never the project. -4. **Return.** Provide the report path and end with this exact question: `Quel changement d’intention général, même minime, appliquons-nous au prochain run pour rendre notre amélioration cumulative et mesurable ?` +1. **Target.** Map file recommendations to real paths; omit behavior-only rows. Build `Fichier | + Ajout | − Retrait ou clarification`, consolidate targets, and show `Aucun fichier recommandé | — | —` when empty. +2. **Render.** Fill [the report template](../assets/report-template.html) with measured values and grounded findings; copy its local CSS and JavaScript. + - Remove samples. Escape every injected value; allow only template markup and local assets. + - Keep `data-prompt` and editable execution instructions as short as possible without losing targets or actions. + - Write only to the allowed unique temporary directory. +3. **Return.** Provide its path and end exactly: `Quel changement d’intention général, même minime, appliquons-nous au prochain run pour rendre notre amélioration cumulative et mesurable ?` ## Test | Case | Pass | | --- | --- | -| A file is shown | it is the target of an earlier recommendation | -| No file needs editing | the empty-table row is shown | -| The report is rendered | its HTML, CSS, and JavaScript resolve locally with no network dependency | -| Conversation evidence is rendered | it is escaped as text and cannot create markup, scripts, or URLs | -| A value is unavailable | the report says `unavailable` instead of showing sample data | -| The run closes | the returned chat response ends with the exact next-intent question | +| Edit table | prior file targets only; consolidated or empty row | +| Report | no samples; unavailable values named; injected evidence inert | +| Assets | HTML, CSS, and JavaScript resolve locally only | +| Close | path returned; exact final question | From 06347d51f5e28c195320fcc4a9c6c7f346aa1bc0 Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Thu, 1 Oct 2026 06:47:40 +0800 Subject: [PATCH 2/7] fix(aidd-refine): keep improve self-assessment internal --- .../skills/05-improve/actions/02-recommend.md | 13 +++++++------ .../skills/05-improve/actions/03-target-edits.md | 6 +++--- .../skills/05-improve/assets/report-template.html | 4 ++-- 3 files changed, 12 insertions(+), 11 deletions(-) diff --git a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md index 02de4fe1b..e57fea062 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md +++ b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md @@ -8,7 +8,7 @@ The evidence boundary, timing and usage tables, complete conversation, and scope ## Output -A `## Recommendations` table with `ID | Question | Type | Diagnostic | Evidence | Smallest change | Target | Saving`. +A `## Recommendations` table with `ID | Focus | Type | Diagnostic | Evidence | Smallest change | Target | Saving`. ## Process @@ -17,9 +17,10 @@ A `## Recommendations` table with `ID | Question | Type | Diagnostic | Evidence - Prefer a cheap model and low reasoning effort when per-agent overrides are supported; otherwise inherit host defaults. - Give the behavior analyst the frozen transcript; give each artifact analyst the boundary, indexed turns, and paths. - Give every analyst steps 3–7; require table rows and no writes. -3. **Question.** Answer all for each scope: - - Faster or better next run? What was removable, unclear, or counterproductive? - - Where should change live? Would it save `time`, `tokens`, `both`, or `unknown`? +3. **Reflect.** Ask yourself, for each scope; keep this assessment internal: + - How could the next run be faster or better? + - What should be removed or clarified? What was counterproductive? + - Where should change live? How could it save time or tokens? - Which back-and-forth, bottlenecks, or tool calls could be removed, batched, parallelized, or replaced? 4. **Verify.** Read each named artifact before judging it. 5. **Assess.** Use `obsolete`, `over-specific-or-time-bound`, `duplicate`, `inconsistent`, `counterproductive`, or `correct`. @@ -27,7 +28,7 @@ A `## Recommendations` table with `ID | Question | Type | Diagnostic | Evidence - Add only for a demonstrated gap the existing content cannot cover. Preserve useful context and requirements. - `correct` means `no change`, never a recommendation. 6. **Merge.** Deduplicate, then verify cited evidence against the frozen source. -7. **Render.** Order by question, then `behavior`, `skill`, `knowledge`. Use the fewest actionable words. +7. **Render.** Order by focus, then `behavior`, `skill`, `knowledge`. Report findings, not the questionnaire. Use the fewest actionable words. - Type: `skill`, `behavior`, `knowledge`, or `tooling`. - Saving: `time`, `tokens`, `both`, or `unknown`. - Use `no change` when evidence supports none. @@ -37,7 +38,7 @@ A `## Recommendations` table with `ID | Question | Type | Diagnostic | Evidence | Case | Pass | | --- | --- | | Dispatch | parallel only for multiple isolated scopes; otherwise local | -| Analysis | every question answered with exact evidence or `no change` | +| Analysis | internal assessment answers every question with exact evidence or `no change` | | Finding | exact turn, tool-call, or read-artifact evidence | | No finding | `no change`; no false recommendation | | Output | allowed labels, types, savings, and stable order; no `improve` evidence | diff --git a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md index 71fffdf66..cc170257d 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md +++ b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md @@ -8,7 +8,7 @@ The timing, usage, and recommendations tables. ## Output -An HTML report in a temporary directory with local `report.css` and `report.js`, plus its path and the next-intent question. +An HTML report in a temporary directory with local `report.css` and `report.js`, plus its path and a next-intent request. ## Process @@ -17,7 +17,7 @@ An HTML report in a temporary directory with local `report.css` and `report.js`, - Remove samples. Escape every injected value; allow only template markup and local assets. - Keep `data-prompt` and editable execution instructions as short as possible without losing targets or actions. - Write only to the allowed unique temporary directory. -3. **Return.** Provide its path and end exactly: `Quel changement d’intention général, même minime, appliquons-nous au prochain run pour rendre notre amélioration cumulative et mesurable ?` +3. **Return.** Return its path. Ask the user, in their language, which small general change in intent to apply next run for cumulative, measurable improvement. ## Test @@ -26,4 +26,4 @@ An HTML report in a temporary directory with local `report.css` and `report.js`, | Edit table | prior file targets only; consolidated or empty row | | Report | no samples; unavailable values named; injected evidence inert | | Assets | HTML, CSS, and JavaScript resolve locally only | -| Close | path returned; exact final question | +| Close | path returned; next intent requested in the user's language | diff --git a/plugins/aidd-refine/skills/05-improve/assets/report-template.html b/plugins/aidd-refine/skills/05-improve/assets/report-template.html index 0c5300815..292c0078e 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report-template.html +++ b/plugins/aidd-refine/skills/05-improve/assets/report-template.html @@ -3,7 +3,7 @@ Template contract: replace every sample run value, timing, finding and file row with evidence from the current conversation. Duplicate or remove `.finding` blocks as needed, keep their data-signal/data-target values aligned with the - visible tags, derive the compact question tag from the finding question, write + visible tags, derive the compact focus tag from the finding focus, write each accepted action in its data-prompt attribute, and ship report.css/report.js beside this file. Remove this comment from the report. --> @@ -198,7 +198,7 @@

Preuve

Changements acceptés : -Une fois le travail validé, pose-moi cette question : « Quel changement d’intention général, même minime, appliquons-nous au prochain run pour rendre notre amélioration cumulative et mesurable ? » +Après validation, propose un changement général pour améliorer le prochain run et mesurer son effet.
Le texte reste éditable. From 3147994843b027534a054aed58a44bf93ef0a635 Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Thu, 1 Oct 2026 06:50:26 +0800 Subject: [PATCH 3/7] fix(aidd-refine): keep next-run intent change small --- .../aidd-refine/skills/05-improve/assets/report-template.html | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugins/aidd-refine/skills/05-improve/assets/report-template.html b/plugins/aidd-refine/skills/05-improve/assets/report-template.html index 292c0078e..3ac50ad60 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report-template.html +++ b/plugins/aidd-refine/skills/05-improve/assets/report-template.html @@ -198,7 +198,7 @@

Preuve

Changements acceptés : -Après validation, propose un changement général pour améliorer le prochain run et mesurer son effet. +Après validation, propose un petit changement général pour améliorer le prochain run et mesurer son effet.
Le texte reste éditable. From 79660f07580c24532bc65a9607013ea54d1e600e Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Mon, 5 Oct 2026 08:55:50 +0700 Subject: [PATCH 4/7] fix(aidd-refine): make improve recommendations actionable and efficient --- .../aidd-refine/skills/05-improve/SKILL.md | 2 +- .../actions/01-read-conversation.md | 14 +- .../skills/05-improve/actions/02-recommend.md | 21 +- .../05-improve/actions/03-target-edits.md | 12 +- .../05-improve/assets/report-template.html | 214 +++--------- .../skills/05-improve/assets/report.css | 328 ++++-------------- .../skills/05-improve/assets/report.js | 51 ++- scripts/__tests__/improve-report.test.js | 46 +++ 8 files changed, 242 insertions(+), 446 deletions(-) create mode 100644 scripts/__tests__/improve-report.test.js diff --git a/plugins/aidd-refine/skills/05-improve/SKILL.md b/plugins/aidd-refine/skills/05-improve/SKILL.md index d2da36217..509d26ec2 100644 --- a/plugins/aidd-refine/skills/05-improve/SKILL.md +++ b/plugins/aidd-refine/skills/05-improve/SKILL.md @@ -26,6 +26,6 @@ Run all three actions without confirmation. Read only the next action file. ## Transversal rules - Stop if the exact complete transcript is unavailable. -- Assess only named artifacts alongside conversation behavior. +- Assess conversation behavior, invoked skill resources, applicable project instructions, and task-relevant indexed memory. - Never invent metrics, coverage, status, or private reasoning. - Keep the project read-only; write only the unique temporary report. diff --git a/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md b/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md index a45802b7e..d23d1d1e6 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md +++ b/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md @@ -8,17 +8,19 @@ A current conversation, an exact conversation ID, or a complete transcript expor ## Output -An evidence boundary, a `## Timing` table with `Activity | Observed time | Share | Evidence`, a `## Usage` table with `Metric | Value | Evidence`, and a scope index. +An evidence boundary, a `## Timing` table with `Activity | Observed time | Share | Evidence`, a `## Usage` table with `Metric | Value | Evidence`, and a `Scope | Status | Evidence` index. ## Process 1. **Resolve.** Use the host route in [conversation sources](../assets/conversation-sources.md); stop unless it yields the exact complete transcript. 2. **Freeze.** End before this invocation, or at the export boundary. Exclude every `improve` invocation and report. -3. **Read once.** Load all in-scope messages, tool calls, results, and timestamps. Index relevant turns and named or invoked skill and knowledge paths. -4. **Measure.** Use only timestamps, elapsed records, and exposed host usage. +3. **Read once.** Load all in-scope messages, tool calls, results, and timestamps. Index relevant turns, invoked skills with used actions or resources, and applicable project instructions. +4. **Context.** Read indexed skill resources and applicable `AGENTS.md`. Use the project memory index to read only task-relevant memory; never scan the whole memory library. Mark only read files `checked`; mark unresolved or skipped scopes `missing` or `not reviewed`. +5. **Measure.** Use only timestamps, elapsed records, and exposed host usage. - Group tool time as `research and diagnosis`, `implementation`, `validation`, or `unattributed`. - Include tokens, requests, and cost only when exposed. -5. **Render.** Order known activity times descending; mark missing metrics `unavailable`. Never call unattributed or unavailable time private reasoning. + - Separate background-process lifetime from blocking time. +6. **Render.** Order known activity times descending; mark missing metrics `unavailable`. Never call unattributed or unavailable time private reasoning. ## Test @@ -26,5 +28,5 @@ An evidence boundary, a `## Timing` table with `Activity | Observed time | Share | --- | --- | | Conversation | exact complete transcript; frozen boundary; no `improve` evidence | | Metric | cited host record, timestamp, or elapsed record; otherwise `unavailable` | -| Time | visible categories only; no private-reasoning claim | -| Scope | exact turns or artifact paths, never a generated summary | +| Time | visible categories only; background lifetime separated; no private-reasoning claim | +| Scope | exact turns or paths; only read files are `checked` | diff --git a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md index e57fea062..10a5e5319 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md +++ b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md @@ -1,6 +1,6 @@ # 02 - Recommend -Find the smallest grounded improvements. +Reduce avoidable work without reducing reliability. ## Input @@ -8,11 +8,11 @@ The evidence boundary, timing and usage tables, complete conversation, and scope ## Output -A `## Recommendations` table with `ID | Focus | Type | Diagnostic | Evidence | Smallest change | Target | Saving`. +A `## Recommendations` table with `ID | Focus | Type | Diagnostic | Evidence | Smallest change | Source | Saving`, plus non-actionable limitations. ## Process -1. **Scope.** Always assess `behavior`; add `skill` or `knowledge` only for indexed artifacts. +1. **Scope.** Always assess `behavior`; add `skill` or `knowledge` for indexed skill resources, instructions, or memory. 2. **Dispatch.** With multiple scopes and isolated or minimal context, run one read-only analyst per scope in parallel. Otherwise analyze locally. - Prefer a cheap model and low reasoning effort when per-agent overrides are supported; otherwise inherit host defaults. - Give the behavior analyst the frozen transcript; give each artifact analyst the boundary, indexed turns, and paths. @@ -22,13 +22,19 @@ A `## Recommendations` table with `ID | Focus | Type | Diagnostic | Evidence | S - What should be removed or clarified? What was counterproductive? - Where should change live? How could it save time or tokens? - Which back-and-forth, bottlenecks, or tool calls could be removed, batched, parallelized, or replaced? -4. **Verify.** Read each named artifact before judging it. +4. **Verify.** Resolve and read each maintained source before judging it; never target caches or installs. Account for intentional host transforms. Judge only `checked` sources; an installed artifact never makes its source `checked`. 5. **Assess.** Use `obsolete`, `over-specific-or-time-bound`, `duplicate`, `inconsistent`, `counterproductive`, or `correct`. + - Compare observed work with the shortest reliable path to the requested result. Keep checks for mutable state or unresolved uncertainty; reuse still-valid results instead of repeating research. + - Raw call count never proves waste. Ground the smallest general correction in evidence. + - Recommend memory only for concise, durable knowledge that prevents recurring rediscovery, never transient state or one-off implementation details. - For skills, especially knowledge: delete evidence-backed waste or inconsistency first, without quota; then consolidate or clarify. - - Add only for a demonstrated gap the existing content cannot cover. Preserve useful context and requirements. + - Reuse or strengthen existing content; add only for a demonstrated gap it cannot cover. Preserve useful context and requirements; use `correct` when it already suffices. + - Claim time savings from blocking-time or resource-impact evidence, never background-process uptime alone. - `correct` means `no change`, never a recommendation. 6. **Merge.** Deduplicate, then verify cited evidence against the frozen source. 7. **Render.** Order by focus, then `behavior`, `skill`, `knowledge`. Report findings, not the questionnaire. Use the fewest actionable words. + - Every recommendation, including behavior or tooling, needs an exact maintained source path, exact current excerpt, minimal replacement or explicit `+`/`−` edit, and a short target/action prompt. + - If no source resolves or no useful persistent edit exists, render a collapsed limitation, not an accept-toggle recommendation. - Type: `skill`, `behavior`, `knowledge`, or `tooling`. - Saving: `time`, `tokens`, `both`, or `unknown`. - Use `no change` when evidence supports none. @@ -39,6 +45,9 @@ A `## Recommendations` table with `ID | Focus | Type | Diagnostic | Evidence | S | --- | --- | | Dispatch | parallel only for multiple isolated scopes; otherwise local | | Analysis | internal assessment answers every question with exact evidence or `no change` | -| Finding | exact turn, tool-call, or read-artifact evidence | +| Efficiency | shortest reliable path; necessary checks retained; repeated work evidenced | +| Finding | exact evidence, maintained source path, minimal edit, and target/action prompt | | No finding | `no change`; no false recommendation | +| Coverage | verdicts only for `checked` files; other scopes stay explicit | +| Limitation | unresolved or non-persistent work is not actionable | | Output | allowed labels, types, savings, and stable order; no `improve` evidence | diff --git a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md index cc170257d..ddc8cc87a 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md +++ b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md @@ -4,7 +4,7 @@ Map minimal edits and render the report. ## Input -The timing, usage, and recommendations tables. +The timing, usage, recommendations, and scope index tables. ## Output @@ -12,9 +12,12 @@ An HTML report in a temporary directory with local `report.css` and `report.js`, ## Process -1. **Target.** Map file recommendations to real paths; omit behavior-only rows. Build `Fichier | + Ajout | − Retrait ou clarification`, consolidate targets, and show `Aucun fichier recommandé | — | —` when empty. +1. **Target.** Map every recommendation to its maintained source path. Build `Fichier | + Ajout | − Retrait ou clarification`, consolidate targets, and show `Aucun fichier recommandé | — | —` when empty. Recommend only; never apply. 2. **Render.** Fill [the report template](../assets/report-template.html) with measured values and grounded findings; copy its local CSS and JavaScript. - Remove samples. Escape every injected value; allow only template markup and local assets. + - Put findings first. Preserve diagnostic/type tags, collapsible filters, accept controls, target table, and prompt. + - Show each source path and minimal edit. Keep raw evidence, secondary metrics, timing, coverage, source records, and limitations collapsed. + - Remove repeated metadata, focus/action tags, score labels, verbose help, and redundant run metadata; put the run ID in collapsed source records. - Keep `data-prompt` and editable execution instructions as short as possible without losing targets or actions. - Write only to the allowed unique temporary directory. 3. **Return.** Return its path. Ask the user, in their language, which small general change in intent to apply next run for cumulative, measurable improvement. @@ -23,7 +26,8 @@ An HTML report in a temporary directory with local `report.css` and `report.js`, | Case | Pass | | --- | --- | -| Edit table | prior file targets only; consolidated or empty row | -| Report | no samples; unavailable values named; injected evidence inert | +| Edit table | every actionable finding has a maintained source row; consolidated or empty | +| Report | no samples; unavailable values and scope coverage named; injected evidence inert | | Assets | HTML, CSS, and JavaScript resolve locally only | +| Findings | exact source edit and accept control; evidence and limitations collapse | | Close | path returned; next intent requested in the user's language | diff --git a/plugins/aidd-refine/skills/05-improve/assets/report-template.html b/plugins/aidd-refine/skills/05-improve/assets/report-template.html index 3ac50ad60..0d0f26cae 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report-template.html +++ b/plugins/aidd-refine/skills/05-improve/assets/report-template.html @@ -1,11 +1,9 @@ @@ -18,191 +16,89 @@
-

Conversation actuelle · couverture partielle

Rapport d’amélioration

-

- Les changements les plus courts qui rendront le prochain run plus rapide, - plus fiable et moins coûteux. -

-
-

Généré12 septembre 2026 · 10:42

-

SourceConversation + fichiers cités

-

Findings3 recommandations

-
-
- -
-
-

Mesure

Coût de la session

- Données partielles -
-
-
Durée active24 minmesurée
-
Tokens128,4 kentrée + sortie + cache
-
Requêtes42tous outils
-
Coût monétaireInconnunon exposé par le fournisseur
-
-

Source : aidd telemetry report v8 · 2026-09-12T02:00Z → 2026-09-12T03:00Z

-
- -
-
-

Chronologie

Où le temps est parti

- 24 min au total -
-
    -
  1. 01
    Recherche et diagnostic12 min · 50 %
    100 %
  2. -
  3. 02
    Implémentation8 min · 33 %
    66 %
  4. -
  5. 03
    Validation4 min · 17 %
    33 %
  6. -
+

Exemple · conversation actuelle · couverture partielle · 1 recommandation

-
-

Décisions

Changements recommandés

-

3 recommandations

-
+

Changements recommandés

1 recommandation

-
-
- Diagnostic - - - - - - -
-
- Cible - - - - - +
+ Filtres +
+
+ Diagnostic + + + + + + +
+
+ Cible + + + + + +
-
+
-
+
F-001
-
-
Plus rapideObsolèteToolingMettre à jour
-

La version flottante casse le reload

-
-

ÉconomieTemps

- +
IncohérentSkill

Exemple · préciser la preuve d’économie

+
-
-

Pourquoi

-

latest pointe désormais vers une CLI dont la commande framework build n’existe plus.

-

Preuve

-

Commande observée : error: unknown command 'build'

-
-
scripts/dev-sync.sh
-
−AIDD_CLI_VERSION="${AIDD_CLI_VERSION:-latest}"
-+AIDD_CLI_VERSION="${AIDD_CLI_VERSION:-5.2.2}"
+
plugins/aidd-refine/skills/05-improve/actions/02-recommend.md
+
−Claim time savings from blocking-time or resource-impact evidence, never background-process uptime alone.
++Claim time savings only from blocking-time or resource-impact evidence; background uptime alone proves none.
+

Sépare durée observée et impact réel.

+
Preuve

Le processus observé restait actif sans bloquer la suite.

-
-
-
F-002
-
-
Économiser des tokensDoublonSkillFusionner
-

Deux règles décrivent le même contrôle

-
-

ÉconomieTokens

- -
-
-
-

Pourquoi

-

Le routeur et l’action répètent le même invariant. Une seule source suffit.

-

Preuve

-

Même exigence relevée dans SKILL.md:24 et actions/02-recommend.md:31.

-
-
-
skills/05-improve/SKILL.md
-
−Verify every recommendation against evidence.
- The recommend action owns evidence verification.
-
-
-
- -
-
-
F-003
-
-
Supprimer ou clarifierIncohérentKnowledgeClarifier
-

La documentation promet un coût indisponible

-
-

ÉconomieTemps

- -
-
-
-

Pourquoi

-

Le montant exact n’est pas exposé. Le rapport doit distinguer valeur inconnue et zéro.

-

Preuve

-

cost_micro_usd absent des observations de cette session.

-
-
-
aidd_docs/memory/telemetry.md
-
−Le coût de chaque session est affiché.
-+Le coût est affiché lorsqu’il est mesuré ; sinon : Inconnu.
-
-
-
- + + +
Limites non actionnables · 1

Une source citée n’a pas pu être résolue ; aucun changement n’est proposé.

-

Synthèse

Fichiers ciblés

3 fichiers
-
- - - - - - - -
Fichier+ Ajout− Retrait ou clarification
scripts/dev-sync.shÉpingler une version compatibleRetirer latest
skills/05-improve/SKILL.mdConserver un propriétaireRetirer la règle dupliquée
aidd_docs/memory/telemetry.mdExpliciter l’absence de mesureRemplacer la promesse absolue
-
+

Fichiers ciblés

1 fichier
+
+ +
Fichier+ Ajout− Retrait ou clarification
plugins/aidd-refine/skills/05-improve/actions/02-recommend.md« uptime seul = aucune économie prouvée »Formulation ambiguë
+
+ Mesures, chronologie et couverture +

Durée active24 min

Tokens128,4 k

Requêtes42

CoûtInconnu

+
ActivitéTempsPart
Recherche et diagnostic12 min50 %
Implémentation8 min33 %
Validation4 min17 %
+

Couverture

AGENTS.md · checked
aidd_docs/memory/testing.md · not reviewed
skill export · missing

+
Sources brutes

run_2026-09-12_01 · aidd telemetry report v8 · 2026-09-12T02:00Z → 03:00Z

+
+
-
-

Passage à l’action

Prompt d’exécution

- 0 recommandation acceptée -
-

Acceptez les changements à appliquer, ajustez le texte si nécessaire, puis copiez-le dans votre conversation avec l’IA.

- -
- Le texte reste éditable. - -
+
Le texte reste éditable.
diff --git a/plugins/aidd-refine/skills/05-improve/assets/report.css b/plugins/aidd-refine/skills/05-improve/assets/report.css index 1f5a54a28..b7da388d5 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report.css +++ b/plugins/aidd-refine/skills/05-improve/assets/report.css @@ -6,293 +6,117 @@ --green-dark: color-mix(in oklch, var(--green) 62%, black); --green-soft: color-mix(in oklch, var(--green) 15%, white); --ink: oklch(0.16 0.01 25); - --ink-soft: oklch(0.38 0.01 25); - --muted: oklch(0.53 0.008 25); + --soft-ink: oklch(0.4 0.01 25); + --muted: oklch(0.55 0.008 25); --line: oklch(0.87 0.006 25); - --line-strong: oklch(0.72 0.01 25); - --surface: oklch(1 0 0); - --surface-soft: oklch(0.975 0.003 25); - --add-bg: var(--green-soft); - --add-ink: var(--green-dark); - --del-bg: var(--red-soft); - --del-ink: var(--red-dark); + --strong-line: oklch(0.72 0.01 25); + --surface: white; + --soft-surface: oklch(0.975 0.003 25); --focus: oklch(0.64 0.18 250); --mono: "SFMono-Regular", Consolas, "Liberation Mono", Menlo, monospace; } * { box-sizing: border-box; } - -html { scroll-behavior: smooth; } - -body { - margin: 0; - background: var(--surface); - color: var(--ink); - font-family: var(--mono); - font-size: 14px; - line-height: 1.6; -} - -button, a { font: inherit; } - -button:focus-visible, -a:focus-visible { - outline: 3px solid var(--focus); - outline-offset: 3px; -} - +body { margin: 0; background: var(--surface); color: var(--ink); font: 14px/1.6 var(--mono); } +button, textarea { font: inherit; } +button:focus-visible, summary:focus-visible, textarea:focus { outline: 3px solid var(--focus); outline-offset: 3px; } code { font-family: var(--mono); } -.site-header { - min-height: 58px; - padding: 0 32px; - border-bottom: 1px solid var(--line); - display: flex; - align-items: center; - justify-content: space-between; - gap: 24px; - position: sticky; - top: 0; - z-index: 10; - background: color-mix(in oklch, var(--surface) 94%, transparent); - backdrop-filter: blur(12px); -} - -.brand { - color: var(--ink); - text-decoration: none; - display: inline-flex; - align-items: center; - gap: 10px; - font-weight: 700; - letter-spacing: -0.02em; -} - -.brand-mark { - width: 25px; - height: 25px; - display: grid; - place-items: center; - background: var(--red); - color: var(--surface); - font-size: 13px; -} - -.header-actions { display: flex; align-items: center; gap: 16px; } -.run-id { color: var(--muted); font-size: 12px; } - -.button { - border: 1px solid var(--ink); - background: var(--ink); - color: var(--surface); - padding: 7px 12px; - cursor: pointer; -} - -.button:hover { background: var(--red); border-color: var(--red); } - -main { width: min(1120px, calc(100% - 48px)); margin: 0 auto; } - -.intro { padding: 72px 0 44px; border-bottom: 1px solid var(--ink); } -.context, .section-index { margin: 0 0 8px; color: var(--red); font-size: 12px; font-weight: 700; } -.context::before, .section-index::before { content: "# "; } - -h1, h2, h3, h4, p { text-wrap: pretty; } -h1 { max-width: 16ch; margin: 0; font-size: 44px; line-height: 1.08; letter-spacing: -0.04em; } -.lede { max-width: 68ch; margin: 22px 0 34px; color: var(--ink-soft); font-size: 16px; } - -.run-meta { margin: 0; display: flex; flex-wrap: wrap; gap: 20px 48px; } -.run-meta p { margin: 0; display: grid; gap: 2px; } -.run-meta span { color: var(--muted); font-size: 11px; } -.run-meta strong { font-size: 12px; } - -.section { padding: 52px 0 56px; border-bottom: 1px solid var(--line); } -.section-heading { display: flex; align-items: flex-end; justify-content: space-between; gap: 24px; margin-bottom: 24px; } -.section-heading h2 { margin: 0; font-size: 23px; line-height: 1.2; letter-spacing: -0.035em; } -.section-total, .result-count { margin: 0; color: var(--muted); font-size: 12px; } - -.coverage { display: inline-flex; align-items: center; gap: 8px; color: var(--ink-soft); font-size: 12px; } -.coverage::before { content: ""; width: 8px; height: 8px; border-radius: 50%; background: var(--red); } - -.metrics { display: grid; grid-template-columns: repeat(4, 1fr); border-top: 1px solid var(--ink); border-bottom: 1px solid var(--line); } -.metric { min-height: 132px; padding: 20px; border-right: 1px solid var(--line); display: flex; flex-direction: column; } -.metric:last-child { border-right: 0; } -.metric > span { color: var(--muted); font-size: 11px; } -.metric strong { margin-top: auto; font-size: 25px; line-height: 1.2; letter-spacing: -0.04em; font-variant-numeric: tabular-nums; } -.metric small { color: var(--muted); font-size: 10px; } -.metric-unknown strong { color: var(--red-dark); font-size: 18px; } -.source-note { margin: 12px 0 0; color: var(--muted); font-size: 11px; } -.source-note code { color: var(--ink-soft); } - -.timing-list { padding: 0; margin: 0; list-style: none; border-top: 1px solid var(--ink); } -.timing-list li { min-height: 74px; border-bottom: 1px solid var(--line); display: grid; grid-template-columns: 44px minmax(220px, 1fr) minmax(160px, 2fr); align-items: center; gap: 20px; } -.timing-rank { color: var(--red); font-weight: 700; } -.timing-list li > div { display: grid; } -.timing-list li > div span { color: var(--muted); font-size: 11px; } -.timing-bar { - width: 100%; - height: 7px; - border: 1px solid var(--line); - border-radius: 0; - background: var(--surface-soft); - appearance: none; -} -.timing-bar::-webkit-progress-bar { background: var(--surface-soft); } -.timing-bar::-webkit-progress-value { background: var(--red); } -.timing-bar::-moz-progress-bar { background: var(--red); } - -.filters { margin-bottom: 24px; padding: 14px 0; border-top: 1px solid var(--ink); border-bottom: 1px solid var(--line); display: flex; flex-wrap: wrap; gap: 16px 36px; } +.site-header { min-height: 58px; padding: 0 32px; border-bottom: 1px solid var(--line); display: flex; align-items: center; justify-content: space-between; position: sticky; top: 0; z-index: 10; background: color-mix(in oklch, white 94%, transparent); backdrop-filter: blur(12px); } +.brand { color: var(--ink); text-decoration: none; display: flex; align-items: center; gap: 10px; font-weight: 700; } +.brand-mark { width: 25px; height: 25px; display: grid; place-items: center; background: var(--red); color: var(--ink); } +.button { border: 1px solid var(--ink); padding: 7px 12px; background: var(--ink); color: white; cursor: pointer; } +.button:hover { border-color: var(--red); background: var(--red); color: var(--ink); } + +main { width: min(1040px, calc(100% - 48px)); margin: auto; } +.intro { padding: 58px 0 30px; border-bottom: 1px solid var(--ink); } +.intro h1 { margin: 0; font-size: clamp(34px, 6vw, 48px); line-height: 1.05; letter-spacing: -0.045em; } +.intro p { margin: 16px 0 0; color: var(--muted); font-size: 12px; } +.section { padding: 42px 0; border-bottom: 1px solid var(--line); } +.section-heading, .prompt-heading { display: flex; align-items: end; justify-content: space-between; gap: 20px; margin-bottom: 20px; } +.section-heading h2, .prompt-heading h2 { margin: 0; font-size: 22px; letter-spacing: -0.03em; } +.result-count, .section-total, .prompt-heading span, .prompt-actions span { color: var(--muted); font-size: 11px; } + +details > summary { cursor: pointer; color: var(--soft-ink); font-weight: 700; } +.filter-panel { margin-bottom: 18px; border-block: 1px solid var(--line); padding: 11px 0; } +.filters { display: flex; flex-wrap: wrap; gap: 12px 30px; padding-top: 12px; } .filter-group { display: flex; flex-wrap: wrap; align-items: center; gap: 6px; } -.filter-group > span { margin-right: 5px; color: var(--muted); font-size: 11px; } -.filter-group button { - border: 1px solid var(--line-strong); - border-radius: 999px; - padding: 4px 9px; - background: var(--surface); - color: var(--ink-soft); - font-size: 11px; - cursor: pointer; -} -.filter-group button:hover { border-color: var(--ink); color: var(--ink); } -.filter-group button[aria-pressed="true"] { border-color: var(--ink); background: var(--ink); color: var(--surface); } +.filter-group > span { margin-right: 4px; color: var(--muted); font-size: 11px; } +.filter-group button { border: 1px solid var(--strong-line); border-radius: 99px; padding: 4px 8px; background: white; color: var(--soft-ink); font-size: 11px; cursor: pointer; } +.filter-group button[aria-pressed="true"] { border-color: var(--ink); background: var(--ink); color: white; } .findings { border-top: 1px solid var(--ink); } .finding { border-bottom: 1px solid var(--ink); } .finding[hidden] { display: none; } -.finding-header { padding: 26px 0 22px; display: grid; grid-template-columns: 62px minmax(0, 1fr) auto auto; gap: 20px; align-items: start; transition: padding 180ms ease-out; } -.finding-id { color: var(--red); font-weight: 700; font-size: 12px; } -.finding-title h3 { margin: 11px 0 0; font-size: 19px; line-height: 1.3; letter-spacing: -0.025em; } -.tags { display: flex; flex-wrap: wrap; gap: 6px; } -.tag { padding: 3px 7px; border: 1px solid var(--line-strong); font-size: 10px; line-height: 1.4; } -.tag-question { border-color: var(--ink); background: var(--ink); color: var(--surface); } -.tag-signal { border-color: var(--red); background: var(--red); color: var(--surface); } -.tag-action { border-color: var(--ink); color: var(--ink); } - -.finding-score { margin: 0; display: flex; gap: 24px; } -.finding-score p { margin: 0; display: grid; gap: 2px; } -.finding-score span { color: var(--muted); font-size: 9px; } -.finding-score strong { font-size: 11px; } - -.accept-toggle { - min-height: 31px; - padding: 5px 9px; - border: 1px solid var(--ink); - display: inline-flex; - align-items: center; - gap: 7px; - color: var(--ink); - font-size: 10px; - font-weight: 700; - cursor: pointer; -} -.accept-toggle:hover { border-color: var(--red); color: var(--red-dark); } -.accept-toggle:focus-within { outline: 3px solid var(--focus); outline-offset: 3px; } +.finding-header { padding: 22px 0; display: grid; grid-template-columns: 58px minmax(0, 1fr) auto; gap: 18px; align-items: start; } +.finding-id { color: var(--red-dark); font-size: 11px; font-weight: 700; } +.finding-title h3 { margin: 9px 0 0; font-size: 18px; line-height: 1.3; } +.tags { display: flex; gap: 6px; } +.tag { border: 1px solid var(--strong-line); padding: 3px 7px; font-size: 11px; } +.tag-signal { border-color: var(--red); background: var(--red); color: var(--ink); } +.accept-toggle { min-height: 31px; border: 1px solid var(--ink); padding: 5px 9px; display: flex; align-items: center; gap: 7px; font-size: 11px; font-weight: 700; cursor: pointer; } .accept-input { width: 14px; height: 14px; margin: 0; accent-color: var(--red); } -.accept-input:focus-visible { outline: 0; } - -.finding-collapse { display: grid; grid-template-rows: 1fr; transition: grid-template-rows 200ms ease-out; } -.finding-body { padding: 0 0 30px 82px; display: grid; grid-template-columns: minmax(240px, 0.8fr) minmax(360px, 1.45fr); gap: 36px; transition: padding 180ms ease-out; } -.finding-id, .finding-title, .finding-score, .finding-collapse { transition: opacity 160ms ease-out; } -.finding.is-accepted .finding-header { padding-block: 14px; } -.finding.is-accepted .finding-collapse { grid-template-rows: 0fr; } -.finding.is-accepted .finding-body { min-height: 0; padding-bottom: 0; overflow: hidden; } -.finding.is-accepted .finding-id, -.finding.is-accepted .finding-title, -.finding.is-accepted .finding-score, -.finding.is-accepted .finding-collapse { opacity: 0.48; } +.finding-collapse { display: grid; grid-template-rows: 1fr; transition: grid-template-rows 180ms ease-out, opacity 180ms ease-out; } +.finding-body { min-height: 0; padding: 0 0 26px 76px; overflow: hidden; } +.finding.is-accepted .finding-header { padding-block: 12px; opacity: 0.55; } +.finding.is-accepted .finding-collapse { grid-template-rows: 0fr; opacity: 0.35; } .finding.is-accepted .accept-toggle { border-color: var(--red); background: var(--red-soft); color: var(--red-dark); } -.finding-copy h4 { margin: 0 0 5px; color: var(--muted); font-size: 10px; } -.finding-copy p { max-width: 62ch; margin: 0 0 18px; color: var(--ink-soft); font-size: 12px; } -.finding-copy .evidence { color: var(--ink); } -.change { align-self: start; border: 1px solid var(--line-strong); background: var(--surface-soft); overflow: hidden; } -.change-header { min-height: 38px; padding: 8px 11px; border-bottom: 1px solid var(--line); display: flex; justify-content: space-between; align-items: center; gap: 16px; font-size: 10px; } +.change { border: 1px solid var(--strong-line); background: var(--soft-surface); overflow: hidden; } +.change-header { min-height: 38px; padding: 8px 11px; border-bottom: 1px solid var(--line); display: flex; justify-content: space-between; gap: 16px; font-size: 11px; } .change-header span { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } -.copy-path { padding: 2px 0; border: 0; background: transparent; color: var(--red-dark); font-size: 9px; cursor: pointer; } -.copy-path:hover { text-decoration: underline; } -.diff { margin: 0; padding: 10px 0; overflow-x: auto; font: 11px/1.8 var(--mono); } +.copy-path { border: 0; padding: 0; background: transparent; color: var(--red-dark); font-size: 11px; cursor: pointer; } +.diff { margin: 0; padding: 9px 0; overflow-x: auto; font: 11px/1.8 var(--mono); } .diff code { display: block; min-width: max-content; } .diff-line { display: block; padding: 0 13px; } .diff-line b { display: inline-block; width: 18px; user-select: none; } -.diff-del { background: var(--del-bg); color: var(--del-ink); } -.diff-add { background: var(--add-bg); color: var(--add-ink); } -.diff-del b { color: var(--red-dark); } -.diff-add b { color: var(--green-dark); } -.diff-context { color: var(--muted); } -.empty-state { margin: 0; padding: 48px 0; border-bottom: 1px solid var(--ink); color: var(--muted); text-align: center; } +.diff-del { background: var(--red-soft); color: var(--red-dark); } +.diff-add { background: var(--green-soft); color: var(--ink); } +.why { margin: 12px 0; color: var(--soft-ink); font-size: 11px; } +.evidence-details { margin-top: 10px; font-size: 11px; } +.evidence-details p { margin: 8px 0 0; padding: 10px; background: var(--soft-surface); } +.empty-state { padding: 35px 0; color: var(--muted); text-align: center; } +.limitations { margin-top: 18px; padding: 12px 0; border-bottom: 1px solid var(--line); font-size: 11px; } .table-wrap { overflow-x: auto; border-top: 1px solid var(--ink); } -table { width: 100%; min-width: 680px; border-collapse: collapse; font-size: 12px; text-align: left; } -th { padding: 12px 14px; border-bottom: 1px solid var(--ink); color: var(--muted); font-size: 10px; font-weight: 500; } -td { padding: 16px 14px; border-bottom: 1px solid var(--line); vertical-align: top; } +table { width: 100%; min-width: 650px; border-collapse: collapse; font-size: 11px; text-align: left; } +th { padding: 10px 12px; border-bottom: 1px solid var(--ink); color: var(--muted); font-size: 11px; font-weight: 500; } +td { padding: 13px 12px; border-bottom: 1px solid var(--line); vertical-align: top; } th:first-child, td:first-child { padding-left: 0; } th:last-child, td:last-child { padding-right: 0; } td code { color: var(--red-dark); font-weight: 700; } -.prompt-builder { margin: 64px 0; padding: 28px 0 0; border-top: 4px solid var(--red); } -.prompt-heading { display: flex; align-items: end; justify-content: space-between; gap: 24px; } -.prompt-heading p { margin: 0 0 5px; color: var(--red); font-size: 11px; font-weight: 700; } -.prompt-heading h2 { margin: 0; font-size: 22px; line-height: 1.3; letter-spacing: -0.025em; } -.prompt-heading > span, .prompt-actions > span { color: var(--muted); font-size: 11px; } -.prompt-help { max-width: 72ch; margin: 18px 0 12px; color: var(--ink-soft); font-size: 12px; } -#execution-prompt { - width: 100%; - min-height: 230px; - padding: 16px; - border: 1px solid var(--ink); - border-radius: 0; - resize: vertical; - background: var(--surface-soft); - color: var(--ink); - font: 12px/1.7 var(--mono); -} -#execution-prompt:focus { outline: 3px solid var(--focus); outline-offset: 3px; } +.raw-section { padding: 20px 0; } +.raw-section > summary { font-size: 15px; } +.raw-section[open] > summary { margin-bottom: 18px; } +.metrics { display: grid; grid-template-columns: repeat(4, 1fr); border-top: 1px solid var(--ink); } +.metrics p { margin: 0; padding: 14px; border-right: 1px solid var(--line); display: grid; gap: 5px; } +.metrics p:last-child { border-right: 0; } +.metrics span { color: var(--muted); font-size: 11px; } +.metrics strong { font-size: 18px; } +.raw-section h3 { margin: 20px 0 5px; font-size: 12px; } +.raw-section p { color: var(--soft-ink); font-size: 11px; } +.source-records { margin-top: 14px; font-size: 11px; } + +.prompt-builder { margin: 54px 0; padding-top: 24px; border-top: 4px solid var(--red); } +#execution-prompt { width: 100%; min-height: 220px; padding: 15px; border: 1px solid var(--ink); resize: vertical; background: var(--soft-surface); color: var(--ink); font: 12px/1.7 var(--mono); } .prompt-actions { margin-top: 10px; display: flex; align-items: center; justify-content: space-between; gap: 20px; } -@media (max-width: 820px) { - .metrics { grid-template-columns: repeat(2, 1fr); } - .metric:nth-child(2) { border-right: 0; } - .metric:nth-child(-n + 2) { border-bottom: 1px solid var(--line); } - .finding-header { grid-template-columns: 50px minmax(0, 1fr) auto; } - .finding-score { grid-column: 2; } - .accept-toggle { grid-column: 3; grid-row: 1 / span 2; } - .finding-body { padding-left: 70px; grid-template-columns: 1fr; } -} - -@media (max-width: 580px) { +@media (max-width: 680px) { .site-header { padding: 0 16px; } - .run-id { display: none; } - main { width: min(100% - 32px, 1120px); } - .intro { padding-top: 48px; } - h1 { font-size: 36px; } - .section-heading { align-items: start; flex-direction: column; gap: 8px; } - .metrics { grid-template-columns: 1fr; } - .metric { min-height: 105px; border-right: 0; border-bottom: 1px solid var(--line); } - .metric:last-child { border-bottom: 0; } - .timing-list li { padding: 14px 0; grid-template-columns: 30px 1fr; gap: 8px 12px; } - .timing-bar { grid-column: 2; } + main { width: min(100% - 32px, 1040px); } .finding-header { grid-template-columns: 1fr; gap: 8px; } - .finding-score { grid-column: 1; flex-wrap: wrap; } - .accept-toggle { grid-column: 1; grid-row: auto; justify-self: start; } + .accept-toggle { justify-self: start; } .finding-body { padding-left: 0; } - .finding-id { order: -1; } - .prompt-heading { align-items: start; flex-direction: column; gap: 6px; } - .prompt-actions { align-items: stretch; flex-direction: column; } - .prompt-actions .button { align-self: flex-start; } -} - -@media (prefers-reduced-motion: reduce) { - html { scroll-behavior: auto; } - .finding-header, .finding-id, .finding-title, .finding-score, .finding-collapse { transition-duration: 0.001ms; } + .metrics { grid-template-columns: repeat(2, 1fr); } + .metrics p:nth-child(2) { border-right: 0; } + .section-heading, .prompt-heading, .prompt-actions { align-items: start; flex-direction: column; } } +@media (prefers-reduced-motion: reduce) { .finding-collapse { transition-duration: 0.001ms; } } @media print { - .site-header, .filters, .copy-path, .accept-toggle, .prompt-actions { display: none !important; } - body { color: black; font-size: 11px; } + .site-header, .filter-panel, .copy-path, .accept-toggle, .prompt-actions { display: none !important; } main { width: 100%; } - .intro { padding-top: 0; } - .section, .finding { break-inside: avoid; } .finding[hidden] { display: block; } .finding.is-accepted { display: none; } - .change { background: white; } } diff --git a/plugins/aidd-refine/skills/05-improve/assets/report.js b/plugins/aidd-refine/skills/05-improve/assets/report.js index 72770d06f..f36fda79e 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report.js +++ b/plugins/aidd-refine/skills/05-improve/assets/report.js @@ -1,12 +1,36 @@ (() => { + const promptAnchor = "Changements acceptés :"; + + function composePrompt(value, accepted, savedLines = new Map(), knownIds = new Set(accepted.map(({ id }) => id))) { + const generatedPattern = /^- \[([^\]]+)\] /; + const original = value.split("\n"); + + for (const line of original) { + const match = line.match(generatedPattern); + if (match && knownIds.has(match[1])) savedLines.set(match[1], line); + } + + const lines = original.filter((line) => { + const match = line.match(generatedPattern); + return !match || !knownIds.has(match[1]); + }); + const generated = accepted.map(({ id, prompt }) => savedLines.get(id) || `- [${id}] ${prompt}`); + const anchorIndex = lines.findIndex((line) => line.trim() === promptAnchor); + lines.splice(anchorIndex >= 0 ? anchorIndex + 1 : lines.length, 0, ...generated); + return lines.join("\n"); + } + + if (typeof module !== "undefined") module.exports = { composePrompt }; + if (typeof document === "undefined") return; + const state = { signal: "all", target: "all" }; const findings = [...document.querySelectorAll(".finding")]; const count = document.querySelector("#result-count"); const empty = document.querySelector("#empty-state"); const prompt = document.querySelector("#execution-prompt"); const acceptedCount = document.querySelector("#accepted-count"); - const promptAnchor = "Changements acceptés :"; - const closingPrompt = "Une fois le travail validé"; + const savedPromptLines = new Map(); + const findingIds = new Set(findings.map((finding) => finding.querySelector(".finding-id").textContent.trim())); function render() { let visible = 0; @@ -44,22 +68,13 @@ }); }); - function updatePrompt(changedFinding) { + function updatePrompt() { const accepted = findings.filter((finding) => finding.querySelector(".accept-input").checked); - const changedInput = changedFinding.querySelector(".accept-input"); - const id = changedFinding.querySelector(".finding-id").textContent.trim(); - const prefix = `- [${id}] `; - const generatedLine = `${prefix}${changedInput.dataset.prompt}`; - const lines = prompt.value.split("\n").filter((line) => !line.startsWith(prefix)); - const anchorIndex = lines.findIndex((line) => line.trim() === promptAnchor); - - if (changedInput.checked) { - const closingIndex = lines.findIndex((line) => line.startsWith(closingPrompt)); - let insertAt = closingIndex >= 0 ? closingIndex : anchorIndex + 1; - if (lines[insertAt - 1] === "") insertAt -= 1; - lines.splice(Math.max(insertAt, 0), 0, generatedLine); - } - prompt.value = lines.join("\n"); + const entries = accepted.map((finding) => ({ + id: finding.querySelector(".finding-id").textContent.trim(), + prompt: finding.querySelector(".accept-input").dataset.prompt, + })); + prompt.value = composePrompt(prompt.value, entries, savedPromptLines, findingIds); const total = accepted.length; acceptedCount.textContent = `${total} recommandation${total === 1 ? "" : "s"} acceptée${total === 1 ? "" : "s"}`; @@ -70,7 +85,7 @@ const finding = input.closest(".finding"); finding.classList.toggle("is-accepted", input.checked); finding.querySelector(".accept-toggle span").textContent = input.checked ? "Accepté" : "Accepter"; - updatePrompt(finding); + updatePrompt(); }); }); diff --git a/scripts/__tests__/improve-report.test.js b/scripts/__tests__/improve-report.test.js new file mode 100644 index 000000000..374699090 --- /dev/null +++ b/scripts/__tests__/improve-report.test.js @@ -0,0 +1,46 @@ +const assert = require('node:assert/strict'); +const { resolve } = require('node:path'); +const test = require('node:test'); + +const { composePrompt } = require(resolve( + __dirname, + '../../plugins/aidd-refine/skills/05-improve/assets/report.js', +)); + +test('accepted edits keep report order and preserve manual prompt text', () => { + const saved = new Map(); + const known = new Set(['F-001', 'F-002']); + const base = [ + 'User introduction.', + '', + 'Changements acceptés :', + '- [NOTE] Keep this manual note.', + '', + 'User closing.', + ].join('\n'); + + let value = composePrompt(base, [{ id: 'F-002', prompt: 'Second edit.' }], saved, known); + value = value.replace('- [F-002] Second edit.', '- [F-002] Second edit, manually refined.'); + value = composePrompt(value, [ + { id: 'F-001', prompt: 'First edit.' }, + { id: 'F-002', prompt: 'Second edit.' }, + ], saved, known); + + assert.ok(value.indexOf('[F-001]') < value.indexOf('[F-002]')); + assert.match(value, /\[F-002\] Second edit, manually refined\./); + assert.match(value, /^User introduction\./); + assert.match(value, /- \[NOTE\] Keep this manual note\./); + assert.match(value, /User closing\.$/); + + value = composePrompt(value, [{ id: 'F-001', prompt: 'First edit.' }], saved, known); + assert.doesNotMatch(value, /\[F-002\]/); + value = composePrompt(value, [ + { id: 'F-001', prompt: 'First edit.' }, + { id: 'F-002', prompt: 'Second edit.' }, + ], saved, known); + assert.match(value, /\[F-002\] Second edit, manually refined\./); + assert.equal(composePrompt(value, [ + { id: 'F-001', prompt: 'First edit.' }, + { id: 'F-002', prompt: 'Second edit.' }, + ], saved, known), value); +}); From 521bf54b423be902f8054cce8259f93ee69de3ed Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Mon, 5 Oct 2026 10:25:44 +0700 Subject: [PATCH 5/7] fix(test): isolate hook probes from source scanners --- scripts/__tests__/check-written-file-hook.test.js | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/scripts/__tests__/check-written-file-hook.test.js b/scripts/__tests__/check-written-file-hook.test.js index 16b8a9242..260f50cd7 100644 --- a/scripts/__tests__/check-written-file-hook.test.js +++ b/scripts/__tests__/check-written-file-hook.test.js @@ -14,7 +14,8 @@ function runHook(payload) { } function probe(name, content) { - const file = path.join(root, "cli/src", `.hook-probe-${process.pid}-${name}.ts`); + // Keep transient probes outside directories scanned by repository-wide tests. + const file = path.join(root, "cli", `.hook-probe-${process.pid}-${name}.ts`); fs.writeFileSync(file, content); return file; } From 9a9cffdcd4b9cb0feb932794001ea83e6f533233 Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Mon, 5 Oct 2026 10:27:02 +0700 Subject: [PATCH 6/7] refactor(aidd-refine): scope improve actions and report --- .../aidd-refine/skills/05-improve/SKILL.md | 17 ++-- .../actions/01-read-conversation.md | 23 ++--- .../skills/05-improve/actions/02-recommend.md | 40 +++------ .../05-improve/actions/03-target-edits.md | 23 ++--- .../05-improve/assets/conversation-sources.md | 8 -- .../05-improve/assets/report-template.html | 86 +++++++++---------- .../skills/05-improve/assets/report.js | 29 ++++--- scripts/__tests__/improve-report.test.js | 35 +++++++- 8 files changed, 120 insertions(+), 141 deletions(-) diff --git a/plugins/aidd-refine/skills/05-improve/SKILL.md b/plugins/aidd-refine/skills/05-improve/SKILL.md index 509d26ec2..fd24b0e7a 100644 --- a/plugins/aidd-refine/skills/05-improve/SKILL.md +++ b/plugins/aidd-refine/skills/05-improve/SKILL.md @@ -15,17 +15,10 @@ flowchart LR ## Actions -Run all three actions without confirmation. Read only the next action file. +Run all three actions in order without confirmation. Read only the next file in `actions/`. Keep the project read-only; write only the unique temporary report. -| Action | Does | +| Order | Action | | --- | --- | -| read-conversation | freeze complete evidence and measure visible cost | -| recommend | analyze relevant scopes and merge grounded findings | -| target-edits | render minimal edits and an executable prompt | - -## Transversal rules - -- Stop if the exact complete transcript is unavailable. -- Assess conversation behavior, invoked skill resources, applicable project instructions, and task-relevant indexed memory. -- Never invent metrics, coverage, status, or private reasoning. -- Keep the project read-only; write only the unique temporary report. +| 1 | `01-read-conversation.md` | +| 2 | `02-recommend.md` | +| 3 | `03-target-edits.md` | diff --git a/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md b/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md index d23d1d1e6..ede26347b 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md +++ b/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md @@ -8,25 +8,16 @@ A current conversation, an exact conversation ID, or a complete transcript expor ## Output -An evidence boundary, a `## Timing` table with `Activity | Observed time | Share | Evidence`, a `## Usage` table with `Metric | Value | Evidence`, and a `Scope | Status | Evidence` index. +An evidence boundary, a `## Timing` table with `Activity | Observed time | Share | Evidence`, a `## Usage` table with `Metric | Value | Evidence`, and a scope index of encountered resources. ## Process -1. **Resolve.** Use the host route in [conversation sources](../assets/conversation-sources.md); stop unless it yields the exact complete transcript. +1. **Resolve.** Use the host route in [conversation sources](../assets/conversation-sources.md). Prefer its complete export or host reader. Match the exact session ID before direct storage; search only for that ID, never unrelated sessions, configuration, or authentication. Stop unless the exact complete transcript is available. 2. **Freeze.** End before this invocation, or at the export boundary. Exclude every `improve` invocation and report. -3. **Read once.** Load all in-scope messages, tool calls, results, and timestamps. Index relevant turns, invoked skills with used actions or resources, and applicable project instructions. -4. **Context.** Read indexed skill resources and applicable `AGENTS.md`. Use the project memory index to read only task-relevant memory; never scan the whole memory library. Mark only read files `checked`; mark unresolved or skipped scopes `missing` or `not reviewed`. -5. **Measure.** Use only timestamps, elapsed records, and exposed host usage. +3. **Read once.** Load every in-scope message, tool call, result, and timestamp. Index relevant turns; invoked skills and their used actions or resources; applicable `AGENTS.md`; and task-relevant memory references. Do not read maintained sources yet. +4. **Measure.** Use only timestamps, elapsed records, and exposed host usage. - Group tool time as `research and diagnosis`, `implementation`, `validation`, or `unattributed`. - - Include tokens, requests, and cost only when exposed. + - Include tokens, requests, and cost only when exposed; otherwise mark them `unavailable`. - Separate background-process lifetime from blocking time. -6. **Render.** Order known activity times descending; mark missing metrics `unavailable`. Never call unattributed or unavailable time private reasoning. - -## Test - -| Case | Pass | -| --- | --- | -| Conversation | exact complete transcript; frozen boundary; no `improve` evidence | -| Metric | cited host record, timestamp, or elapsed record; otherwise `unavailable` | -| Time | visible categories only; background lifetime separated; no private-reasoning claim | -| Scope | exact turns or paths; only read files are `checked` | + - Treat private reasoning duration as unavailable unless exposed directly; never infer it. +5. **Deliver.** Order known activity times descending and cite every value. diff --git a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md index 10a5e5319..9180a14d0 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md +++ b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md @@ -4,50 +4,34 @@ Reduce avoidable work without reducing reliability. ## Input -The evidence boundary, timing and usage tables, complete conversation, and scope index. +The evidence boundary, timing and usage tables, complete conversation, and resource index. ## Output -A `## Recommendations` table with `ID | Focus | Type | Diagnostic | Evidence | Smallest change | Source | Saving`, plus non-actionable limitations. +An updated `Scope | Status | Evidence` index, a `## Recommendations` table with `ID | Focus | Type | Diagnostic | Evidence | Smallest change | Source | Saving`, and non-actionable limitations. ## Process -1. **Scope.** Always assess `behavior`; add `skill` or `knowledge` for indexed skill resources, instructions, or memory. -2. **Dispatch.** With multiple scopes and isolated or minimal context, run one read-only analyst per scope in parallel. Otherwise analyze locally. +1. **Verify.** Resolve and read indexed maintained skill resources, applicable `AGENTS.md`, and task-relevant memory references once; reuse a still-valid read. Add and read only relevant paths from the project memory index; never scan unrelated memory. Neither judge from nor target caches or installs. Account for host transforms. Mark only read sources `checked`; mark unresolved or skipped ones `missing` or `not reviewed`. +2. **Scope.** Always assess `behavior`; add `skill` or `knowledge` only for checked indexed sources. +3. **Dispatch.** With multiple isolated scopes, run one read-only analyst per scope in parallel. Otherwise analyze locally. - Prefer a cheap model and low reasoning effort when per-agent overrides are supported; otherwise inherit host defaults. - - Give the behavior analyst the frozen transcript; give each artifact analyst the boundary, indexed turns, and paths. - - Give every analyst steps 3–7; require table rows and no writes. -3. **Reflect.** Ask yourself, for each scope; keep this assessment internal: + - Give the behavior analyst the frozen transcript. Give artifact analysts the boundary, indexed turns, and verified excerpts with evidence; allow a targeted read-only refresh only when validity is uncertain. + - Give every analyst steps 4–6 and require table rows. +4. **Reflect.** Ask yourself, for each scope; keep this assessment internal: - How could the next run be faster or better? - What should be removed or clarified? What was counterproductive? - Where should change live? How could it save time or tokens? - Which back-and-forth, bottlenecks, or tool calls could be removed, batched, parallelized, or replaced? -4. **Verify.** Resolve and read each maintained source before judging it; never target caches or installs. Account for intentional host transforms. Judge only `checked` sources; an installed artifact never makes its source `checked`. 5. **Assess.** Use `obsolete`, `over-specific-or-time-bound`, `duplicate`, `inconsistent`, `counterproductive`, or `correct`. - Compare observed work with the shortest reliable path to the requested result. Keep checks for mutable state or unresolved uncertainty; reuse still-valid results instead of repeating research. - Raw call count never proves waste. Ground the smallest general correction in evidence. - Recommend memory only for concise, durable knowledge that prevents recurring rediscovery, never transient state or one-off implementation details. - For skills, especially knowledge: delete evidence-backed waste or inconsistency first, without quota; then consolidate or clarify. - - Reuse or strengthen existing content; add only for a demonstrated gap it cannot cover. Preserve useful context and requirements; use `correct` when it already suffices. + - Reuse or strengthen existing content; add only for a demonstrated gap it cannot cover. Preserve useful context and requirements; classify sufficient content as `correct` => `no change`. - Claim time savings from blocking-time or resource-impact evidence, never background-process uptime alone. - - `correct` means `no change`, never a recommendation. -6. **Merge.** Deduplicate, then verify cited evidence against the frozen source. -7. **Render.** Order by focus, then `behavior`, `skill`, `knowledge`. Report findings, not the questionnaire. Use the fewest actionable words. - - Every recommendation, including behavior or tooling, needs an exact maintained source path, exact current excerpt, minimal replacement or explicit `+`/`−` edit, and a short target/action prompt. - - If no source resolves or no useful persistent edit exists, render a collapsed limitation, not an accept-toggle recommendation. +6. **Deliver.** Deduplicate, verify cited evidence, and order by focus, then `behavior`, `skill`, `knowledge`. Report findings, not the questionnaire, in the fewest actionable words. + - Every recommendation needs an exact maintained source path, current excerpt, minimal correction or explicit `+`/`−` edit, and a short target/action prompt. + - If no source resolves or no useful persistent edit exists, report a non-actionable limitation. - Type: `skill`, `behavior`, `knowledge`, or `tooling`. - Saving: `time`, `tokens`, `both`, or `unknown`. - - Use `no change` when evidence supports none. - -## Test - -| Case | Pass | -| --- | --- | -| Dispatch | parallel only for multiple isolated scopes; otherwise local | -| Analysis | internal assessment answers every question with exact evidence or `no change` | -| Efficiency | shortest reliable path; necessary checks retained; repeated work evidenced | -| Finding | exact evidence, maintained source path, minimal edit, and target/action prompt | -| No finding | `no change`; no false recommendation | -| Coverage | verdicts only for `checked` files; other scopes stay explicit | -| Limitation | unresolved or non-persistent work is not actionable | -| Output | allowed labels, types, savings, and stable order; no `improve` evidence | diff --git a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md index ddc8cc87a..ec8f9c812 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md +++ b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md @@ -12,22 +12,9 @@ An HTML report in a temporary directory with local `report.css` and `report.js`, ## Process -1. **Target.** Map every recommendation to its maintained source path. Build `Fichier | + Ajout | − Retrait ou clarification`, consolidate targets, and show `Aucun fichier recommandé | — | —` when empty. Recommend only; never apply. -2. **Render.** Fill [the report template](../assets/report-template.html) with measured values and grounded findings; copy its local CSS and JavaScript. - - Remove samples. Escape every injected value; allow only template markup and local assets. - - Put findings first. Preserve diagnostic/type tags, collapsible filters, accept controls, target table, and prompt. - - Show each source path and minimal edit. Keep raw evidence, secondary metrics, timing, coverage, source records, and limitations collapsed. - - Remove repeated metadata, focus/action tags, score labels, verbose help, and redundant run metadata; put the run ID in collapsed source records. - - Keep `data-prompt` and editable execution instructions as short as possible without losing targets or actions. - - Write only to the allowed unique temporary directory. +1. **Target.** Consolidate the provided recommendations by source into `File | + Added | − Removed or clarified`; use `No recommended file | — | —` when empty. Never resolve, judge, or apply sources here. +2. **Render.** Fill [the report template](../assets/report-template.html) from the provided data; remove its sample and copy its local CSS and JavaScript into the allowed unique temporary directory. + - Escape injected values; allow only template markup and local assets. Localize the HTML `lang`, visible text, and dynamic `data-label-*` text to the user's language. + - Put findings first with diagnostic/type tags, filters, accept controls, the target table, and an editable global prompt. Copy each recommendation's short target/action prompt into `data-prompt` without redrafting it. + - Show source paths and minimal edits. Collapse raw evidence, secondary metrics, timing, coverage, source records, and limitations; keep the run ID in source records and avoid repeated metadata. 3. **Return.** Return its path. Ask the user, in their language, which small general change in intent to apply next run for cumulative, measurable improvement. - -## Test - -| Case | Pass | -| --- | --- | -| Edit table | every actionable finding has a maintained source row; consolidated or empty | -| Report | no samples; unavailable values and scope coverage named; injected evidence inert | -| Assets | HTML, CSS, and JavaScript resolve locally only | -| Findings | exact source edit and accept control; evidence and limitations collapse | -| Close | path returned; next intent requested in the user's language | diff --git a/plugins/aidd-refine/skills/05-improve/assets/conversation-sources.md b/plugins/aidd-refine/skills/05-improve/assets/conversation-sources.md index 329cd7da7..c4a689fbe 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/conversation-sources.md +++ b/plugins/aidd-refine/skills/05-improve/assets/conversation-sources.md @@ -7,11 +7,3 @@ Use this static routing map to load one complete conversation. Do not fill or co | Codex | host thread reader with the exact thread ID, then TUI `/export` Markdown | `CODEX_HOME/history.jsonl`, normally `~/.codex/history.jsonl` | TUI `/status` estimated thread credits or cost; tool elapsed records when exposed | | Claude Code | `/export` text, or the documented script interface for an exact session ID | `~/.claude/projects/{project}/{session-id}.jsonl` | transcript timestamps and tool records when exposed | | OpenCode | `opencode export {session-id} --sanitize` JSON | `~/.local/share/opencode/project/{project-slug}/storage/` | `opencode stats`; `~/.local/share/opencode/log/` | - -## Rules - -- Prefer the complete export or host reader over direct storage parsing. -- Use direct storage only after matching one exact session ID. -- Search shared history or logs only for the exact session ID and ignore unrelated records. -- Never enumerate unrelated sessions or inspect configuration or authentication files. -- Treat private reasoning duration as unavailable unless the host exposes it directly. diff --git a/plugins/aidd-refine/skills/05-improve/assets/report-template.html b/plugins/aidd-refine/skills/05-improve/assets/report-template.html index 0d0f26cae..bc239f99e 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report-template.html +++ b/plugins/aidd-refine/skills/05-improve/assets/report-template.html @@ -1,11 +1,5 @@ - - + @@ -17,33 +11,33 @@ -
+
-

Rapport d’amélioration

-

Exemple · conversation actuelle · couverture partielle · 1 recommandation

+

Improve report

+

Example only · fictional content · partial coverage · 1 recommendation

-

Changements recommandés

1 recommandation

+

Recommended changes

1 recommendation

- Filtres -
+ Filters +
Diagnostic - - - - - - + + + + + +
-
- Cible - +
+ Target + @@ -53,52 +47,52 @@

Rapport d’amélioration

-
+
F-001
-
IncohérentSkill

Exemple · préciser la preuve d’économie

- +
DuplicateKnowledge

Example · remove a fictional duplicate rule

+
-
plugins/aidd-refine/skills/05-improve/actions/02-recommend.md
-
−Claim time savings from blocking-time or resource-impact evidence, never background-process uptime alone.
-+Claim time savings only from blocking-time or resource-impact evidence; background uptime alone proves none.
+
docs/example-guidance.md
+
 Validate before publishing.
+−Validate before publishing.
-

Sépare durée observée et impact réel.

-
Preuve

Le processus observé restait actif sans bloquer la suite.

+

One rule carries the same requirement.

+
Evidence

The fictional source contains the same rule twice.

- + -
Limites non actionnables · 1

Une source citée n’a pas pu être résolue ; aucun changement n’est proposé.

+
Non-actionable limitations · 1

A cited source could not be resolved, so no change is proposed.

-

Fichiers ciblés

1 fichier
-
- +

Target files

1 file
+
Fichier+ Ajout− Retrait ou clarification
plugins/aidd-refine/skills/05-improve/actions/02-recommend.md« uptime seul = aucune économie prouvée »Formulation ambiguë
+
File+ Added− Removed or clarified
docs/example-guidance.md—Second duplicate rule
- Mesures, chronologie et couverture -

Durée active24 min

Tokens128,4 k

Requêtes42

CoûtInconnu

-
ActivitéTempsPart
Recherche et diagnostic12 min50 %
Implémentation8 min33 %
Validation4 min17 %
-

Couverture

AGENTS.md · checked
aidd_docs/memory/testing.md · not reviewed
skill export · missing

-
Sources brutes

run_2026-09-12_01 · aidd telemetry report v8 · 2026-09-12T02:00Z → 03:00Z

+ Metrics, timeline, and coverage +

Active time24 min

Tokens128.4k

Requests42

CostUnknown

+
ActivityTimeShare
Research and diagnosis12 min50%
Implementation8 min33%
Validation4 min17%
+

Coverage

AGENTS.md · checked
aidd_docs/memory/testing.md · not reviewed
skill export · missing

+
Raw sources

run_2026-09-12_01 · aidd telemetry report v8 · 2026-09-12T02:00Z → 03:00Z

-

Prompt d’exécution

0 recommandation acceptée
- -
Le texte reste éditable.
+--- +
The text remains editable.
diff --git a/plugins/aidd-refine/skills/05-improve/assets/report.js b/plugins/aidd-refine/skills/05-improve/assets/report.js index f36fda79e..0b91b3bec 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report.js +++ b/plugins/aidd-refine/skills/05-improve/assets/report.js @@ -1,5 +1,10 @@ (() => { - const promptAnchor = "Changements acceptés :"; + const promptMarker = "---"; + + function formatCount(count, one, other) { + const template = count === 1 ? one : other; + return template ? template.replace("{count}", String(count)) : String(count); + } function composePrompt(value, accepted, savedLines = new Map(), knownIds = new Set(accepted.map(({ id }) => id))) { const generatedPattern = /^- \[([^\]]+)\] /; @@ -15,15 +20,16 @@ return !match || !knownIds.has(match[1]); }); const generated = accepted.map(({ id, prompt }) => savedLines.get(id) || `- [${id}] ${prompt}`); - const anchorIndex = lines.findIndex((line) => line.trim() === promptAnchor); - lines.splice(anchorIndex >= 0 ? anchorIndex + 1 : lines.length, 0, ...generated); + const markerIndex = lines.findIndex((line) => line.trim() === promptMarker); + lines.splice(markerIndex >= 0 ? markerIndex : lines.length, 0, ...generated); return lines.join("\n"); } - if (typeof module !== "undefined") module.exports = { composePrompt }; + if (typeof module !== "undefined") module.exports = { composePrompt, formatCount }; if (typeof document === "undefined") return; const state = { signal: "all", target: "all" }; + const report = document.querySelector("#report"); const findings = [...document.querySelectorAll(".finding")]; const count = document.querySelector("#result-count"); const empty = document.querySelector("#empty-state"); @@ -40,7 +46,7 @@ finding.hidden = !(matchesSignal && matchesTarget); if (!finding.hidden) visible += 1; } - count.textContent = `${visible} recommandation${visible === 1 ? "" : "s"}`; + count.textContent = formatCount(visible, report.dataset.labelResultOne, report.dataset.labelResultOther); empty.hidden = visible !== 0; } @@ -60,9 +66,9 @@ const original = button.textContent; try { await navigator.clipboard.writeText(button.dataset.copy); - button.textContent = "Copié"; + button.textContent = button.dataset.labelSuccess || original; } catch { - button.textContent = "Copie indisponible"; + button.textContent = button.dataset.labelError || original; } window.setTimeout(() => { button.textContent = original; }, 1400); }); @@ -77,14 +83,15 @@ prompt.value = composePrompt(prompt.value, entries, savedPromptLines, findingIds); const total = accepted.length; - acceptedCount.textContent = `${total} recommandation${total === 1 ? "" : "s"} acceptée${total === 1 ? "" : "s"}`; + acceptedCount.textContent = formatCount(total, report.dataset.labelAcceptedOne, report.dataset.labelAcceptedOther); } document.querySelectorAll(".accept-input").forEach((input) => { input.addEventListener("change", () => { const finding = input.closest(".finding"); finding.classList.toggle("is-accepted", input.checked); - finding.querySelector(".accept-toggle span").textContent = input.checked ? "Accepté" : "Accepter"; + const label = finding.querySelector(".accept-toggle span"); + label.textContent = input.checked ? input.dataset.labelOn : input.dataset.labelOff; updatePrompt(); }); }); @@ -94,11 +101,11 @@ const original = button.textContent; try { await navigator.clipboard.writeText(prompt.value); - button.textContent = "Prompt copié"; + button.textContent = button.dataset.labelSuccess || original; } catch { prompt.focus(); prompt.select(); - button.textContent = "Texte sélectionné"; + button.textContent = button.dataset.labelError || original; } window.setTimeout(() => { button.textContent = original; }, 1600); }); diff --git a/scripts/__tests__/improve-report.test.js b/scripts/__tests__/improve-report.test.js index 374699090..1eb1f41a1 100644 --- a/scripts/__tests__/improve-report.test.js +++ b/scripts/__tests__/improve-report.test.js @@ -2,7 +2,7 @@ const assert = require('node:assert/strict'); const { resolve } = require('node:path'); const test = require('node:test'); -const { composePrompt } = require(resolve( +const { composePrompt, formatCount } = require(resolve( __dirname, '../../plugins/aidd-refine/skills/05-improve/assets/report.js', )); @@ -13,9 +13,11 @@ test('accepted edits keep report order and preserve manual prompt text', () => { const base = [ 'User introduction.', '', - 'Changements acceptés :', + 'Accepted changes:', '- [NOTE] Keep this manual note.', '', + '---', + '', 'User closing.', ].join('\n'); @@ -31,6 +33,7 @@ test('accepted edits keep report order and preserve manual prompt text', () => { assert.match(value, /^User introduction\./); assert.match(value, /- \[NOTE\] Keep this manual note\./); assert.match(value, /User closing\.$/); + assert.ok(value.indexOf('[F-002]') < value.indexOf('---')); value = composePrompt(value, [{ id: 'F-001', prompt: 'First edit.' }], saved, known); assert.doesNotMatch(value, /\[F-002\]/); @@ -44,3 +47,31 @@ test('accepted edits keep report order and preserve manual prompt text', () => { { id: 'F-002', prompt: 'Second edit.' }, ], saved, known), value); }); + +test('prompt insertion is independent of localized headings', () => { + for (const [heading, closing] of [ + ['Accepted changes:', 'After validation, propose one improvement.'], + ['Changements acceptés :', 'Après validation, propose une amélioration.'], + ]) { + const base = ['Introduction.', '', heading, '', '---', '', closing].join('\n'); + const value = composePrompt(base, [{ id: 'F-001', prompt: 'Apply the edit.' }]); + + assert.ok(value.indexOf('[F-001]') < value.indexOf('---')); + assert.ok(value.indexOf('---') < value.indexOf(closing)); + assert.match(value, new RegExp(heading.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'))); + } +}); + +test('missing prompt marker preserves manual text and appends accepted edits safely', () => { + const base = ['Manual introduction.', '', 'Manual closing.'].join('\n'); + const value = composePrompt(base, [{ id: 'F-001', prompt: 'Apply the edit.' }]); + + assert.match(value, /^Manual introduction\./); + assert.match(value, /Manual closing\./); + assert.ok(value.indexOf('Manual closing.') < value.indexOf('[F-001]')); +}); + +test('count labels come from localized HTML data', () => { + assert.equal(formatCount(1, '{count} recommendation', '{count} recommendations'), '1 recommendation'); + assert.equal(formatCount(2, '{count} recommandation', '{count} recommandations'), '2 recommandations'); +}); From e8a3434729ca5e19e37c7bc655fcce18975c6205 Mon Sep 17 00:00:00 2001 From: alexsoyes Date: Wed, 7 Oct 2026 10:07:00 +0700 Subject: [PATCH 7/7] fix(aidd-refine): streamline improve analysis and report Keep optimization analysis focused and reports directly actionable. --- .../actions/01-read-conversation.md | 10 +- .../skills/05-improve/actions/02-recommend.md | 20 ++- .../05-improve/actions/03-target-edits.md | 18 +-- .../05-improve/assets/report-template.html | 130 ++++++++++-------- .../skills/05-improve/assets/report.css | 129 ++++++++--------- .../skills/05-improve/assets/report.js | 45 +----- scripts/__tests__/improve-report.test.js | 99 +++++++++++++ 7 files changed, 259 insertions(+), 192 deletions(-) diff --git a/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md b/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md index ede26347b..02a09fa04 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md +++ b/plugins/aidd-refine/skills/05-improve/actions/01-read-conversation.md @@ -8,16 +8,22 @@ A current conversation, an exact conversation ID, or a complete transcript expor ## Output -An evidence boundary, a `## Timing` table with `Activity | Observed time | Share | Evidence`, a `## Usage` table with `Metric | Value | Evidence`, and a scope index of encountered resources. +- Evidence boundary and resource index. +- Chronological timeline with stable event IDs: `Moment | Actor/tool | Action/source | Observed duration | Result`. +- Timing: `Activity | Observed time | Share | Evidence`. +- Usage: `Metric | Value | Evidence`. ## Process 1. **Resolve.** Use the host route in [conversation sources](../assets/conversation-sources.md). Prefer its complete export or host reader. Match the exact session ID before direct storage; search only for that ID, never unrelated sessions, configuration, or authentication. Stop unless the exact complete transcript is available. 2. **Freeze.** End before this invocation, or at the export boundary. Exclude every `improve` invocation and report. 3. **Read once.** Load every in-scope message, tool call, result, and timestamp. Index relevant turns; invoked skills and their used actions or resources; applicable `AGENTS.md`; and task-relevant memory references. Do not read maintained sources yet. + - Record one timeline row per message, call, or result, including failures and retries. Link call/result pairs; retain full text, arguments, and output. + - Preserve transcript order. Identify agents, linked calls, and parallel activity when exposed; do not invent hidden calls. 4. **Measure.** Use only timestamps, elapsed records, and exposed host usage. - Group tool time as `research and diagnosis`, `implementation`, `validation`, or `unattributed`. + - Attribute each call duration once; never sum overlapping intervals into elapsed time. Mark unknown durations `unavailable`. - Include tokens, requests, and cost only when exposed; otherwise mark them `unavailable`. - Separate background-process lifetime from blocking time. - Treat private reasoning duration as unavailable unless exposed directly; never infer it. -5. **Deliver.** Order known activity times descending and cite every value. +5. **Deliver.** Cite every measured value. Order only activity totals by descending time; keep the timeline chronological. diff --git a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md index 9180a14d0..49d54832a 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md +++ b/plugins/aidd-refine/skills/05-improve/actions/02-recommend.md @@ -4,11 +4,11 @@ Reduce avoidable work without reducing reliability. ## Input -The evidence boundary, timing and usage tables, complete conversation, and resource index. +The evidence boundary, timeline, timing and usage tables, complete conversation, and resource index. ## Output -An updated `Scope | Status | Evidence` index, a `## Recommendations` table with `ID | Focus | Type | Diagnostic | Evidence | Smallest change | Source | Saving`, and non-actionable limitations. +Coverage with `Scope | Status | Evidence`, and recommendations with `ID | Focus | Type | Diagnostic | Evidence | Smallest change | Source | Saving`. Pass through the boundary, timeline, and measurements unchanged. ## Process @@ -18,20 +18,16 @@ An updated `Scope | Status | Evidence` index, a `## Recommendations` table with - Prefer a cheap model and low reasoning effort when per-agent overrides are supported; otherwise inherit host defaults. - Give the behavior analyst the frozen transcript. Give artifact analysts the boundary, indexed turns, and verified excerpts with evidence; allow a targeted read-only refresh only when validity is uncertain. - Give every analyst steps 4–6 and require table rows. -4. **Reflect.** Ask yourself, for each scope; keep this assessment internal: - - How could the next run be faster or better? - - What should be removed or clarified? What was counterproductive? - - Where should change live? How could it save time or tokens? - - Which back-and-forth, bottlenecks, or tool calls could be removed, batched, parallelized, or replaced? +4. **Reflect.** Ask yourself one question per scope, internally: "Based on the conversation, what shorter, equally reliable path could have achieved the same result, and what minimal reusable change would reduce time, tokens, or cost next run?" 5. **Assess.** Use `obsolete`, `over-specific-or-time-bound`, `duplicate`, `inconsistent`, `counterproductive`, or `correct`. - - Compare observed work with the shortest reliable path to the requested result. Keep checks for mutable state or unresolved uncertainty; reuse still-valid results instead of repeating research. + - Assess repeated reads, back-and-forth, bottlenecks, and calls to remove, batch, parallelize, or replace. Keep checks for mutable state or unresolved uncertainty; reuse valid results. - Raw call count never proves waste. Ground the smallest general correction in evidence. - Recommend memory only for concise, durable knowledge that prevents recurring rediscovery, never transient state or one-off implementation details. - For skills, especially knowledge: delete evidence-backed waste or inconsistency first, without quota; then consolidate or clarify. - Reuse or strengthen existing content; add only for a demonstrated gap it cannot cover. Preserve useful context and requirements; classify sufficient content as `correct` => `no change`. - - Claim time savings from blocking-time or resource-impact evidence, never background-process uptime alone. + - Ground savings in blocking-time, resource-impact, or exposed cost evidence. State trade-offs; background-process uptime alone proves no saving. 6. **Deliver.** Deduplicate, verify cited evidence, and order by focus, then `behavior`, `skill`, `knowledge`. Report findings, not the questionnaire, in the fewest actionable words. - - Every recommendation needs an exact maintained source path, current excerpt, minimal correction or explicit `+`/`−` edit, and a short target/action prompt. - - If no source resolves or no useful persistent edit exists, report a non-actionable limitation. + - Every recommendation needs an exact maintained source path, current excerpt, explicit `−`/`+` correction, relevant event IDs, and a short target/action prompt. Leave `+` empty for deletion-only edits. + - Disclose unresolved sources in coverage. Omit findings without a useful persistent edit; do not invent recommendations. - Type: `skill`, `behavior`, `knowledge`, or `tooling`. - - Saving: `time`, `tokens`, `both`, or `unknown`. + - Saving: `time`, `tokens`, `cost`, their combination, or `unknown`. diff --git a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md index ec8f9c812..aa8dbe195 100644 --- a/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md +++ b/plugins/aidd-refine/skills/05-improve/actions/03-target-edits.md @@ -1,10 +1,10 @@ # 03 - Target edits -Map minimal edits and render the report. +Render the evidence and minimal edits. ## Input -The timing, usage, recommendations, and scope index tables. +The evidence boundary, timeline, timing, usage, coverage, and recommendations. ## Output @@ -12,9 +12,11 @@ An HTML report in a temporary directory with local `report.css` and `report.js`, ## Process -1. **Target.** Consolidate the provided recommendations by source into `File | + Added | − Removed or clarified`; use `No recommended file | — | —` when empty. Never resolve, judge, or apply sources here. -2. **Render.** Fill [the report template](../assets/report-template.html) from the provided data; remove its sample and copy its local CSS and JavaScript into the allowed unique temporary directory. - - Escape injected values; allow only template markup and local assets. Localize the HTML `lang`, visible text, and dynamic `data-label-*` text to the user's language. - - Put findings first with diagnostic/type tags, filters, accept controls, the target table, and an editable global prompt. Copy each recommendation's short target/action prompt into `data-prompt` without redrafting it. - - Show source paths and minimal edits. Collapse raw evidence, secondary metrics, timing, coverage, source records, and limitations; keep the run ID in source records and avoid repeated metadata. -3. **Return.** Return its path. Ask the user, in their language, which small general change in intent to apply next run for cumulative, measurable improvement. +1. **Render.** Fill [the report template](../assets/report-template.html) from the provided data; remove its sample and copy its local CSS and JavaScript into the allowed unique temporary directory. Never resolve, judge, or apply sources here. + - Escape injected values; allow only template markup and local assets. Localize HTML `lang`, UI text, and dynamic `data-label-*` text; preserve quoted evidence. + - Start with elapsed seconds, tokens, and call counts only. Use exposed usage only; distinguish linked subcalls. No visible dates or coverage. Keep text sizes uniform except titles and user prompts. + - Show full user prompts in labelled blockquotes, followed by compact calls: tool, relative path or command, and elapsed gap since the previous displayed event. Nest only exposed subcalls; preserve chronological order and parallel relationships. + - Derive gaps only from recorded timestamps; distinguish them from call durations. Omit assistant prose, raw payloads, and evidence explanations. No dropdowns. + - Put recommendations below: an actionable title, one relative path, and unlabelled red `−` / green `+` columns. Use `—` for an empty side; omit descriptions and proof links. + - Keep black emoji tags, accept controls, and the editable prompt; copy each `data-prompt` unchanged. No filters, print control, target table, or limitations section. +2. **Return.** Return its path. Ask the user, in their language, which small general change in intent to apply next run for cumulative, measurable improvement. diff --git a/plugins/aidd-refine/skills/05-improve/assets/report-template.html b/plugins/aidd-refine/skills/05-improve/assets/report-template.html index bc239f99e..d571102aa 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report-template.html +++ b/plugins/aidd-refine/skills/05-improve/assets/report-template.html @@ -11,88 +11,108 @@ -
+

Improve report

-

Example only · fictional content · partial coverage · 1 recommendation

+

Fictional example.

-
-

Recommended changes

1 recommendation

+
+
+

Elapsed9 s

+

TokensUnavailable

+

Calls · includes 2 subcalls6

+
+
-
- Filters -
-
- Diagnostic - - - - - - -
-
- Target - - - - - -
+
+

Conversation

+ +
+
+
User promptStart
+

Review the project guidance for repetition and unnecessary output. Suggest changes only.

+
+ +
+
+
read_fileAGENTS.md+2 s
+
+
+
parallel2 linked read_file subcalls+2 s
+
+
+
read_filedocs/example-review.md+0.1 s
+
+
+
read_filedocs/example-guidance.md+0.1 s
+
+
+
+
+
read_filedocs/example-review-notes.md ENOENT+0.8 s
+
+
+
shellrg --files docs -g '*review*' retry+1 s
+
-
+ + +
+
+
User prompt+2 s
+

Show replacements and deletions side by side. Do not invent a new rule for the deleted duplicate.

+
+
+
+ +
+

Recommended changes

-
+
+
+
Counterproductive Behavior

Replace the full-output requirement

+ +
+
+
+
docs/example-review.md
+
+
−Before replying, print every tool result in full.
+
+Report the result and the shortest evidence needed to verify it.
+
+
+
+
+
-
F-001
-
DuplicateKnowledge

Example · remove a fictional duplicate rule

+
Duplicate Knowledge

Remove the second publishing rule

-
docs/example-guidance.md
-
 Validate before publishing.
-−Validate before publishing.
+
docs/example-guidance.md
+
+
−Validate before publishing.
+

—

+
-

One rule carries the same requirement.

-
Evidence

The fictional source contains the same rule twice.

-
- - -
Non-actionable limitations · 1

A cited source could not be resolved, so no change is proposed.

-
-

Target files

1 file
-
- -
File+ Added− Removed or clarified
docs/example-guidance.md—Second duplicate rule
-
- -
- Metrics, timeline, and coverage -

Active time24 min

Tokens128.4k

Requests42

CostUnknown

-
ActivityTimeShare
Research and diagnosis12 min50%
Implementation8 min33%
Validation4 min17%
-

Coverage

AGENTS.md · checked
aidd_docs/memory/testing.md · not reviewed
skill export · missing

-
Raw sources

run_2026-09-12_01 · aidd telemetry report v8 · 2026-09-12T02:00Z → 03:00Z

-
-
diff --git a/plugins/aidd-refine/skills/05-improve/assets/report.css b/plugins/aidd-refine/skills/05-improve/assets/report.css index b7da388d5..2fea03391 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report.css +++ b/plugins/aidd-refine/skills/05-improve/assets/report.css @@ -3,11 +3,10 @@ --red-dark: color-mix(in oklch, var(--red) 72%, black); --red-soft: color-mix(in oklch, var(--red) 12%, white); --green: #66cc99; - --green-dark: color-mix(in oklch, var(--green) 62%, black); + --green-dark: color-mix(in oklch, var(--green) 58%, black); --green-soft: color-mix(in oklch, var(--green) 15%, white); --ink: oklch(0.16 0.01 25); --soft-ink: oklch(0.4 0.01 25); - --muted: oklch(0.55 0.008 25); --line: oklch(0.87 0.006 25); --strong-line: oklch(0.72 0.01 25); --surface: white; @@ -17,106 +16,92 @@ } * { box-sizing: border-box; } -body { margin: 0; background: var(--surface); color: var(--ink); font: 14px/1.6 var(--mono); } +body { margin: 0; background: var(--surface); color: var(--ink); font: 13px/1.6 var(--mono); } button, textarea { font: inherit; } -button:focus-visible, summary:focus-visible, textarea:focus { outline: 3px solid var(--focus); outline-offset: 3px; } +button:focus-visible, a:focus-visible, textarea:focus { outline: 3px solid var(--focus); outline-offset: 3px; } code { font-family: var(--mono); } +a { color: var(--red-dark); text-underline-offset: 3px; } -.site-header { min-height: 58px; padding: 0 32px; border-bottom: 1px solid var(--line); display: flex; align-items: center; justify-content: space-between; position: sticky; top: 0; z-index: 10; background: color-mix(in oklch, white 94%, transparent); backdrop-filter: blur(12px); } +.site-header { min-height: 58px; padding: 0 32px; border-bottom: 1px solid var(--line); display: flex; align-items: center; position: sticky; top: 0; z-index: 10; background: color-mix(in oklch, white 94%, transparent); backdrop-filter: blur(12px); } .brand { color: var(--ink); text-decoration: none; display: flex; align-items: center; gap: 10px; font-weight: 700; } .brand-mark { width: 25px; height: 25px; display: grid; place-items: center; background: var(--red); color: var(--ink); } .button { border: 1px solid var(--ink); padding: 7px 12px; background: var(--ink); color: white; cursor: pointer; } .button:hover { border-color: var(--red); background: var(--red); color: var(--ink); } main { width: min(1040px, calc(100% - 48px)); margin: auto; } -.intro { padding: 58px 0 30px; border-bottom: 1px solid var(--ink); } -.intro h1 { margin: 0; font-size: clamp(34px, 6vw, 48px); line-height: 1.05; letter-spacing: -0.045em; } -.intro p { margin: 16px 0 0; color: var(--muted); font-size: 12px; } -.section { padding: 42px 0; border-bottom: 1px solid var(--line); } +.intro { padding: 24px 0 16px; } +.intro h1 { margin: 0; font-size: 28px; line-height: 1.2; letter-spacing: -0.03em; } +.intro p { margin: 8px 0 0; color: var(--soft-ink); } +.section { padding: 24px 0; border-bottom: 1px solid var(--line); } .section-heading, .prompt-heading { display: flex; align-items: end; justify-content: space-between; gap: 20px; margin-bottom: 20px; } -.section-heading h2, .prompt-heading h2 { margin: 0; font-size: 22px; letter-spacing: -0.03em; } -.result-count, .section-total, .prompt-heading span, .prompt-actions span { color: var(--muted); font-size: 11px; } +.section-heading h2, .prompt-heading h2 { margin: 0; font-size: 18px; letter-spacing: -0.03em; } +.prompt-heading span { color: var(--soft-ink); } -details > summary { cursor: pointer; color: var(--soft-ink); font-weight: 700; } -.filter-panel { margin-bottom: 18px; border-block: 1px solid var(--line); padding: 11px 0; } -.filters { display: flex; flex-wrap: wrap; gap: 12px 30px; padding-top: 12px; } -.filter-group { display: flex; flex-wrap: wrap; align-items: center; gap: 6px; } -.filter-group > span { margin-right: 4px; color: var(--muted); font-size: 11px; } -.filter-group button { border: 1px solid var(--strong-line); border-radius: 99px; padding: 4px 8px; background: white; color: var(--soft-ink); font-size: 11px; cursor: pointer; } -.filter-group button[aria-pressed="true"] { border-color: var(--ink); background: var(--ink); color: white; } +.recap { padding-bottom: 16px; border-bottom: 1px solid var(--line); } +.metrics { display: grid; grid-template-columns: repeat(3, minmax(0, 1fr)); border: 1px solid var(--line); font-variant-numeric: tabular-nums; } +.metrics p { margin: 0; padding: 12px 16px; display: grid; gap: 4px; align-content: start; overflow-wrap: anywhere; } +.metrics p + p { border-left: 1px solid var(--line); } +.metrics span { color: var(--soft-ink); } + +.conversation-turn + .conversation-turn { margin-top: 30px; } +.user-message { margin: 0 0 16px; padding: 16px; border-left: 4px solid var(--ink); background: #f2f2f2; } +.message-meta { display: flex; flex-wrap: wrap; align-items: center; gap: 12px; color: var(--soft-ink); font-variant-numeric: tabular-nums; } +.message-meta > span:first-child { color: var(--ink); font-weight: 700; } +.message-meta .event-delta { margin-left: auto; } +.user-prompt { margin: 8px 0 0; font-size: 18px; line-height: 1.45; letter-spacing: -0.02em; } +.tool-calls { padding-left: 28px; } +.tool-call { margin: 12px 0; } +.tool-line { display: grid; grid-template-columns: 92px minmax(0, 1fr) auto; align-items: baseline; gap: 10px; padding: 4px 0; } +.tool-name { color: var(--soft-ink); font-weight: 700; } +.tool-target { overflow-wrap: anywhere; } +.event-delta { color: var(--soft-ink); white-space: nowrap; font-variant-numeric: tabular-nums; } +.call-error { color: var(--red-dark); background: var(--red-soft); padding: 1px 4px; } +.call-status { color: var(--soft-ink); } +.subcalls { margin-left: 18px; padding-left: 16px; border-left: 1px solid var(--line); } .findings { border-top: 1px solid var(--ink); } .finding { border-bottom: 1px solid var(--ink); } -.finding[hidden] { display: none; } -.finding-header { padding: 22px 0; display: grid; grid-template-columns: 58px minmax(0, 1fr) auto; gap: 18px; align-items: start; } -.finding-id { color: var(--red-dark); font-size: 11px; font-weight: 700; } +.finding-header { padding: 22px 0; display: grid; grid-template-columns: minmax(0, 1fr) auto; gap: 18px; align-items: start; } .finding-title h3 { margin: 9px 0 0; font-size: 18px; line-height: 1.3; } -.tags { display: flex; gap: 6px; } -.tag { border: 1px solid var(--strong-line); padding: 3px 7px; font-size: 11px; } -.tag-signal { border-color: var(--red); background: var(--red); color: var(--ink); } -.accept-toggle { min-height: 31px; border: 1px solid var(--ink); padding: 5px 9px; display: flex; align-items: center; gap: 7px; font-size: 11px; font-weight: 700; cursor: pointer; } +.tags { display: flex; flex-wrap: wrap; gap: 6px; } +.tag { background: black; color: white; padding: 3px 7px; } +.accept-toggle { min-height: 31px; border: 1px solid var(--ink); padding: 5px 9px; display: flex; align-items: center; gap: 7px; font-weight: 700; cursor: pointer; } .accept-input { width: 14px; height: 14px; margin: 0; accent-color: var(--red); } .finding-collapse { display: grid; grid-template-rows: 1fr; transition: grid-template-rows 180ms ease-out, opacity 180ms ease-out; } -.finding-body { min-height: 0; padding: 0 0 26px 76px; overflow: hidden; } +.finding-body { min-height: 0; padding-bottom: 26px; overflow: hidden; } .finding.is-accepted .finding-header { padding-block: 12px; opacity: 0.55; } .finding.is-accepted .finding-collapse { grid-template-rows: 0fr; opacity: 0.35; } +.finding.is-accepted .finding-body { padding-block: 0; } .finding.is-accepted .accept-toggle { border-color: var(--red); background: var(--red-soft); color: var(--red-dark); } -.change { border: 1px solid var(--strong-line); background: var(--soft-surface); overflow: hidden; } -.change-header { min-height: 38px; padding: 8px 11px; border-bottom: 1px solid var(--line); display: flex; justify-content: space-between; gap: 16px; font-size: 11px; } -.change-header span { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } -.copy-path { border: 0; padding: 0; background: transparent; color: var(--red-dark); font-size: 11px; cursor: pointer; } -.diff { margin: 0; padding: 9px 0; overflow-x: auto; font: 11px/1.8 var(--mono); } -.diff code { display: block; min-width: max-content; } -.diff-line { display: block; padding: 0 13px; } +.change { border: 1px solid var(--strong-line); overflow: hidden; } +.change-header { padding: 8px 11px; border-bottom: 1px solid var(--line); } +.change-header code { overflow-wrap: anywhere; } +.diff-columns { display: grid; grid-template-columns: repeat(2, minmax(0, 1fr)); } +.diff-pane + .diff-pane { border-left: 1px solid var(--line); } +.diff { margin: 0; padding: 11px 0; font: inherit; white-space: pre-wrap; overflow-wrap: anywhere; } +.diff code { display: block; } +.diff-line { display: block; padding: 0 11px; } .diff-line b { display: inline-block; width: 18px; user-select: none; } .diff-del { background: var(--red-soft); color: var(--red-dark); } -.diff-add { background: var(--green-soft); color: var(--ink); } -.why { margin: 12px 0; color: var(--soft-ink); font-size: 11px; } -.evidence-details { margin-top: 10px; font-size: 11px; } -.evidence-details p { margin: 8px 0 0; padding: 10px; background: var(--soft-surface); } -.empty-state { padding: 35px 0; color: var(--muted); text-align: center; } -.limitations { margin-top: 18px; padding: 12px 0; border-bottom: 1px solid var(--line); font-size: 11px; } - -.table-wrap { overflow-x: auto; border-top: 1px solid var(--ink); } -table { width: 100%; min-width: 650px; border-collapse: collapse; font-size: 11px; text-align: left; } -th { padding: 10px 12px; border-bottom: 1px solid var(--ink); color: var(--muted); font-size: 11px; font-weight: 500; } -td { padding: 13px 12px; border-bottom: 1px solid var(--line); vertical-align: top; } -th:first-child, td:first-child { padding-left: 0; } -th:last-child, td:last-child { padding-right: 0; } -td code { color: var(--red-dark); font-weight: 700; } +.diff-add { background: var(--green-soft); color: var(--green-dark); } +.diff-empty { margin: 0; padding: 11px; color: var(--soft-ink); } -.raw-section { padding: 20px 0; } -.raw-section > summary { font-size: 15px; } -.raw-section[open] > summary { margin-bottom: 18px; } -.metrics { display: grid; grid-template-columns: repeat(4, 1fr); border-top: 1px solid var(--ink); } -.metrics p { margin: 0; padding: 14px; border-right: 1px solid var(--line); display: grid; gap: 5px; } -.metrics p:last-child { border-right: 0; } -.metrics span { color: var(--muted); font-size: 11px; } -.metrics strong { font-size: 18px; } -.raw-section h3 { margin: 20px 0 5px; font-size: 12px; } -.raw-section p { color: var(--soft-ink); font-size: 11px; } -.source-records { margin-top: 14px; font-size: 11px; } - -.prompt-builder { margin: 54px 0; padding-top: 24px; border-top: 4px solid var(--red); } -#execution-prompt { width: 100%; min-height: 220px; padding: 15px; border: 1px solid var(--ink); resize: vertical; background: var(--soft-surface); color: var(--ink); font: 12px/1.7 var(--mono); } -.prompt-actions { margin-top: 10px; display: flex; align-items: center; justify-content: space-between; gap: 20px; } +.prompt-builder { margin: 28px 0; padding-top: 24px; border-top: 4px solid var(--red); } +#execution-prompt { width: 100%; min-height: 220px; padding: 15px; border: 1px solid var(--ink); resize: vertical; background: var(--soft-surface); color: var(--ink); } +.prompt-actions { margin-top: 10px; display: flex; justify-content: end; } @media (max-width: 680px) { .site-header { padding: 0 16px; } main { width: min(100% - 32px, 1040px); } + .metrics { grid-template-columns: 1fr; } + .metrics p + p { border-left: 0; border-top: 1px solid var(--line); } + .tool-calls { padding-left: 20px; } + .tool-line { grid-template-columns: 76px minmax(0, 1fr) auto; gap: 6px; } + .subcalls { margin-left: 8px; padding-left: 8px; } .finding-header { grid-template-columns: 1fr; gap: 8px; } .accept-toggle { justify-self: start; } - .finding-body { padding-left: 0; } - .metrics { grid-template-columns: repeat(2, 1fr); } - .metrics p:nth-child(2) { border-right: 0; } - .section-heading, .prompt-heading, .prompt-actions { align-items: start; flex-direction: column; } + .section-heading, .prompt-heading { align-items: start; flex-direction: column; } } @media (prefers-reduced-motion: reduce) { .finding-collapse { transition-duration: 0.001ms; } } -@media print { - .site-header, .filter-panel, .copy-path, .accept-toggle, .prompt-actions { display: none !important; } - main { width: 100%; } - .finding[hidden] { display: block; } - .finding.is-accepted { display: none; } -} diff --git a/plugins/aidd-refine/skills/05-improve/assets/report.js b/plugins/aidd-refine/skills/05-improve/assets/report.js index 0b91b3bec..a7801ce11 100644 --- a/plugins/aidd-refine/skills/05-improve/assets/report.js +++ b/plugins/aidd-refine/skills/05-improve/assets/report.js @@ -28,56 +28,17 @@ if (typeof module !== "undefined") module.exports = { composePrompt, formatCount }; if (typeof document === "undefined") return; - const state = { signal: "all", target: "all" }; const report = document.querySelector("#report"); const findings = [...document.querySelectorAll(".finding")]; - const count = document.querySelector("#result-count"); - const empty = document.querySelector("#empty-state"); const prompt = document.querySelector("#execution-prompt"); const acceptedCount = document.querySelector("#accepted-count"); const savedPromptLines = new Map(); - const findingIds = new Set(findings.map((finding) => finding.querySelector(".finding-id").textContent.trim())); - - function render() { - let visible = 0; - for (const finding of findings) { - const matchesSignal = state.signal === "all" || finding.dataset.signal === state.signal; - const matchesTarget = state.target === "all" || finding.dataset.target === state.target; - finding.hidden = !(matchesSignal && matchesTarget); - if (!finding.hidden) visible += 1; - } - count.textContent = formatCount(visible, report.dataset.labelResultOne, report.dataset.labelResultOther); - empty.hidden = visible !== 0; - } - - document.querySelectorAll("[data-filter]").forEach((button) => { - button.addEventListener("click", () => { - const kind = button.dataset.filter; - state[kind] = button.dataset.value; - document.querySelectorAll(`[data-filter="${kind}"]`).forEach((candidate) => { - candidate.setAttribute("aria-pressed", String(candidate === button)); - }); - render(); - }); - }); - - document.querySelectorAll(".copy-path").forEach((button) => { - button.addEventListener("click", async () => { - const original = button.textContent; - try { - await navigator.clipboard.writeText(button.dataset.copy); - button.textContent = button.dataset.labelSuccess || original; - } catch { - button.textContent = button.dataset.labelError || original; - } - window.setTimeout(() => { button.textContent = original; }, 1400); - }); - }); + const findingIds = new Set(findings.map((finding) => finding.dataset.id)); function updatePrompt() { const accepted = findings.filter((finding) => finding.querySelector(".accept-input").checked); const entries = accepted.map((finding) => ({ - id: finding.querySelector(".finding-id").textContent.trim(), + id: finding.dataset.id, prompt: finding.querySelector(".accept-input").dataset.prompt, })); prompt.value = composePrompt(prompt.value, entries, savedPromptLines, findingIds); @@ -109,6 +70,4 @@ } window.setTimeout(() => { button.textContent = original; }, 1600); }); - - document.querySelector("#print-report").addEventListener("click", () => window.print()); })(); diff --git a/scripts/__tests__/improve-report.test.js b/scripts/__tests__/improve-report.test.js index 1eb1f41a1..8a81779ea 100644 --- a/scripts/__tests__/improve-report.test.js +++ b/scripts/__tests__/improve-report.test.js @@ -1,6 +1,8 @@ const assert = require('node:assert/strict'); +const { readFileSync } = require('node:fs'); const { resolve } = require('node:path'); const test = require('node:test'); +const { runInNewContext } = require('node:vm'); const { composePrompt, formatCount } = require(resolve( __dirname, @@ -75,3 +77,100 @@ test('count labels come from localized HTML data', () => { assert.equal(formatCount(1, '{count} recommendation', '{count} recommendations'), '1 recommendation'); assert.equal(formatCount(2, '{count} recommandation', '{count} recommandations'), '2 recommandations'); }); + +function loadReport() { + function element(fields = {}) { + const listeners = new Map(); + return { + ...fields, + addEventListener: (name, listener) => listeners.set(name, listener), + dispatch(name) { return listeners.get(name)({ currentTarget: this }); }, + }; + } + + const prompt = element({ value: 'Manual introduction.\n\n---\nManual closing.' }); + const acceptedCount = element({ textContent: '' }); + const copy = element({ textContent: 'Copy', dataset: { labelSuccess: 'Copied' } }); + const writes = []; + const findings = ['concise-output', 'duplicate-publishing-rule'].map((id) => { + const classes = new Set(); + const label = { textContent: 'Accepter' }; + const input = element({ + checked: false, + dataset: { prompt: `Apply ${id}.`, labelOff: 'Accepter', labelOn: 'Accepté' }, + }); + const finding = { + dataset: { id }, + classes, + input, + label, + classList: { toggle: (name, enabled) => enabled ? classes.add(name) : classes.delete(name) }, + querySelector: (selector) => ({ + '.accept-input': input, + '.accept-toggle span': label, + })[selector], + }; + input.closest = () => finding; + return finding; + }); + const document = { + querySelector: (selector) => ({ + '#report': { dataset: { labelAcceptedOne: '{count} acceptée', labelAcceptedOther: '{count} acceptées' } }, + '#execution-prompt': prompt, + '#accepted-count': acceptedCount, + '#copy-prompt': copy, + })[selector] || null, + querySelectorAll: (selector) => ({ + '.finding': findings, + '.accept-input': findings.map(({ input }) => input), + })[selector] || [], + }; + + runInNewContext(readFileSync(resolve( + __dirname, + '../../plugins/aidd-refine/skills/05-improve/assets/report.js', + ), 'utf8'), { + document, + navigator: { clipboard: { writeText: async (value) => writes.push(value) } }, + window: { setTimeout() {} }, + }); + return { prompt, acceptedCount, copy, findings, writes }; +} + +test('acceptance uses metadata IDs without visible ID nodes and retains manual edits', () => { + const { prompt, acceptedCount, findings } = loadReport(); + const second = findings[1]; + second.input.checked = true; + second.input.dispatch('change'); + + assert.ok(second.classes.has('is-accepted')); + assert.equal(second.label.textContent, 'Accepté'); + assert.equal(acceptedCount.textContent, '1 acceptée'); + assert.match(prompt.value, /\[duplicate-publishing-rule\] Apply duplicate-publishing-rule\./); + + prompt.value = prompt.value.replace('Apply duplicate-publishing-rule.', 'Keep my edited instruction.'); + second.input.checked = false; + second.input.dispatch('change'); + assert.ok(!second.classes.has('is-accepted')); + assert.equal(second.label.textContent, 'Accepter'); + assert.equal(acceptedCount.textContent, '0 acceptées'); + assert.doesNotMatch(prompt.value, /\[duplicate-publishing-rule\]/); + + second.input.checked = true; + second.input.dispatch('change'); + assert.match(prompt.value, /\[duplicate-publishing-rule\] Keep my edited instruction\./); + assert.match(prompt.value, /^Manual introduction\./); + assert.match(prompt.value, /Manual closing\.$/); +}); + +test('copy reads the current editable prompt after acceptance', async () => { + const { prompt, copy, findings, writes } = loadReport(); + findings[0].input.checked = true; + findings[0].input.dispatch('change'); + prompt.value += '\nAn extra manual instruction.'; + + await copy.dispatch('click'); + + assert.deepEqual(writes, [prompt.value]); + assert.equal(copy.textContent, 'Copied'); +});