mirror of
https://github.com/openai/skills.git
synced 2026-09-11 19:47:26 +03:00
Remove experimental skills except wrapped (#126)
Co-authored-by: Ed Bayes <@openai.com>
This commit is contained in:
co-authored by
Ed Bayes <@openai.com>
parent
a5119697b8
commit
ea6b206c68
@@ -1,202 +0,0 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -1,77 +0,0 @@
|
||||
---
|
||||
name: codex-readiness-integration-test
|
||||
description: Run the Codex Readiness integration test. Use when you need an end-to-end agentic loop with build/test scoring.
|
||||
metadata:
|
||||
short-description: Run Codex Readiness integration test
|
||||
---
|
||||
|
||||
# LLM Codex Readiness Integration Test
|
||||
|
||||
This skill runs a multi-stage integration test to validate agentic execution quality. It always runs in execute mode (no read-only mode).
|
||||
|
||||
## Outputs
|
||||
|
||||
Each run writes to `.codex-readiness-integration-test/<timestamp>/` and updates `.codex-readiness-integration-test/latest.json`.
|
||||
|
||||
New outputs per run:
|
||||
- `agentic_summary.json` and `logs/agentic.log` (agentic loop execution)
|
||||
- `llm_results.json` (automatic LLM evaluation)
|
||||
- `summary.txt` (human-readable summary)
|
||||
|
||||
## Pre-conditions (Required)
|
||||
|
||||
- Authenticate with the Codex CLI using the repo-local HOME before running the test.
|
||||
Run these in your own terminal (not via the integration test):
|
||||
HOME=$PWD/.codex-home XDG_CACHE_HOME=$PWD/.codex-home/.cache codex login
|
||||
HOME=$PWD/.codex-home XDG_CACHE_HOME=$PWD/.codex-home/.cache codex login status
|
||||
- The integration test creates {repo_root}/.codex-home and {repo_root}/.codex-home/.cache/codex as its first step.
|
||||
|
||||
## Workflow
|
||||
|
||||
0) Ask the user how to source the task.
|
||||
- Offer two explicit options: (a) user provides a custom task/prompt, or (b) auto-generate a task.
|
||||
- Do not run the entry point until the user chooses one option.
|
||||
1) Generate or load `{out_dir}/prompt.pending.json`.
|
||||
- Use the integration test's expected prompt path, not `prompt.json` at the repo root.
|
||||
- With the default out dir, this path is `.codex-readiness-integration-test/prompt.pending.json`.
|
||||
- If `--seed-task` is provided, it is used as the starting task.
|
||||
- If not provided, generate a task with `skills/codex-readiness-integration-test/references/generate_prompt.md` and save the JSON to `{out_dir}/prompt.pending.json`.
|
||||
- The user must approve the prompt before execution (no auto-approve mode). Make sure to output a summary of the prompt when asking the user to approve.
|
||||
2) Execute the agentic loop via Codex CLI (uses `AGENTS.md` and `change_prompt`).
|
||||
3) Run build/test commands from the prompt plan via `skills/codex-readiness-integration-test/scripts/run_plan.py`.
|
||||
4) Collect evidence (`evidence.json`), deterministic checks, and run automatic LLM evals via Codex CLI.
|
||||
5) Score and write the report + summary output.
|
||||
|
||||
## Configuration
|
||||
|
||||
Optional fields in `{out_dir}/prompt.pending.json`:
|
||||
- `agentic_loop`: configure Codex CLI invocation for the agentic loop.
|
||||
- `llm_eval`: configure Codex CLI invocation for automatic evals.
|
||||
|
||||
If these fields are omitted, defaults are used.
|
||||
|
||||
## Requirements
|
||||
|
||||
- The LLM evaluator must fail if evidence mentions the phrase `Context compaction enabled`.
|
||||
- Use qualitative context-usage evaluation (no strict thresholds).
|
||||
|
||||
|
||||
## What this test covers well
|
||||
|
||||
- Runs Codex CLI against the real repo root, producing real filesystem edits and git diffs.
|
||||
- Executes the approved change prompt and then runs the build/test plan in-repo.
|
||||
- Captures evidence, deterministic checks, and LLM eval artifacts for review.
|
||||
|
||||
## What this test does not represent
|
||||
|
||||
- The agentic loop may use non-default flags (e.g., bypass approvals/sandbox), so interactive guardrails differ.
|
||||
- Uses a dedicated HOME (`.codex-home`), which can change auth/config/cache vs normal CLI use.
|
||||
- Auto-generated prompts and one-shot execution do not simulate interactive guidance.
|
||||
- MCP servers/tools are not exercised unless explicitly configured.
|
||||
|
||||
## Notes
|
||||
|
||||
- The prompts in `skills/codex-readiness-integration-test/references/` expect strict JSON.
|
||||
- Use `skills/codex-readiness-integration-test/references/json_fix.md` to repair invalid JSON output.
|
||||
- This skill calls the `codex` CLI. Ensure it is installed and available on PATH, or override the command in `{out_dir}/prompt.pending.json`.
|
||||
- If the agentic loop detects sandbox-blocked tool access, it now writes `requires_escalation: true` to `{run_dir}/agentic_summary.json` and exits with code `3`. Re-run the integration test with escalated permissions in that case.
|
||||
@@ -1,25 +0,0 @@
|
||||
You are evaluating whether the agentic loop executed successfully. You will be given a JSON input payload with:
|
||||
- prompt: prompt.json content
|
||||
- evidence: evidence.json content (includes agents_md and logs_index)
|
||||
- git_diff
|
||||
- execution_summary
|
||||
- agentic_summary
|
||||
|
||||
Hard fails:
|
||||
- If any evidence or logs mention the phrase "Context compaction enabled", status MUST be FAIL.
|
||||
|
||||
Prefer PASS only if agentic_summary.status is PASS and exit_code is 0.
|
||||
|
||||
Return strict JSON with this schema:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...","quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- Use WARN for partial success (e.g., agentic_summary exists but status is FAIL).
|
||||
- Reference concrete evidence paths in evidence_quotes.
|
||||
- Output JSON only. No markdown.
|
||||
@@ -1,28 +0,0 @@
|
||||
Evaluate code change quality relative to prompt.json and the actual diff. You will be given a JSON input payload with:
|
||||
- prompt: prompt.json content
|
||||
- evidence: evidence.json content
|
||||
- git_diff
|
||||
- execution_summary
|
||||
- agentic_summary
|
||||
|
||||
Score quality across these dimensions:
|
||||
- correctness vs change_prompt
|
||||
- context usage (qualitative)
|
||||
- maintainability/readability
|
||||
- risk/regression assessment
|
||||
- builds/tests passing (from execution_summary)
|
||||
|
||||
Return strict JSON with this schema:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...","quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- If builds/tests FAIL, status should be FAIL unless the prompt explicitly allows it.
|
||||
- Use WARN for partial correctness or limited context usage.
|
||||
- Call out mismatches between change_prompt and git_diff.
|
||||
- Output JSON only. No markdown.
|
||||
@@ -1,129 +0,0 @@
|
||||
{
|
||||
"schema_version": "1.0",
|
||||
"checks": [
|
||||
{
|
||||
"id": "agentic_run_success",
|
||||
"title": "P0: Agentic loop execution",
|
||||
"description": "Agentic loop ran and exited with code 0.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "run_dir",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "agentic_run_success",
|
||||
"deterministic_rule_params": {
|
||||
"path": "agentic_summary.json"
|
||||
},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "exec_plan_before_code_changes",
|
||||
"title": "P0: Planning signal before code changes",
|
||||
"description": "A planning signal (for example: update_plan or \"Plan:\") appears in agentic.log before any non-doc, non-.codex file changes.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "run_dir",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "exec_plan_before_code_changes",
|
||||
"deterministic_rule_params": {
|
||||
"prompt_path": "prompt.json",
|
||||
"agentic_log_path": "logs/agentic.log"
|
||||
},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "verification_after_code_changes",
|
||||
"title": "P0: Verification after code changes",
|
||||
"description": "A build, test, or lint verification command appears in agentic.log after the first non-doc, non-.codex file change.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "run_dir",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "verification_after_code_changes",
|
||||
"deterministic_rule_params": {
|
||||
"prompt_path": "prompt.json",
|
||||
"agentic_log_path": "logs/agentic.log"
|
||||
},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "repo_root_only_changes",
|
||||
"title": "P1: Repo-root-only changes",
|
||||
"description": "Git diff paths resolve under repo root.",
|
||||
"priority": 1,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "run_dir",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "repo_root_only_changes",
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "execution_summary_status",
|
||||
"title": "P0: Build/test execution status",
|
||||
"description": "Execution summary is present and overall status is PASS or WARN.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "run_dir",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "execution_summary_status",
|
||||
"deterministic_rule_params": {
|
||||
"summary_path": "execution_summary.json"
|
||||
},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "execution_logs_no_errors",
|
||||
"title": "P2: Execution logs show no obvious errors",
|
||||
"description": "Execution logs do not contain obvious error markers.",
|
||||
"priority": 2,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "run_dir",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "execution_logs_no_errors",
|
||||
"deterministic_rule_params": {
|
||||
"logs_dir": "logs"
|
||||
},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "agentic_loop_eval",
|
||||
"title": "P0: Agentic loop success",
|
||||
"description": "LLM evaluation of agentic loop success, AGENTS reference, and compaction avoidance.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "LLM",
|
||||
"scope": "run_dir",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": "agentic_loop_eval",
|
||||
"deterministic_rule_id": null,
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "change_quality_eval",
|
||||
"title": "P0: Change quality evaluation",
|
||||
"description": "LLM evaluation of correctness, context usage, maintainability, and risk.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "LLM",
|
||||
"scope": "run_dir",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": "change_quality",
|
||||
"deterministic_rule_id": null,
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,38 +0,0 @@
|
||||
You are generating a change prompt for an integration test. First, check whether AGENTS.md exists at the repo root and incorporate any build/test guidance from it. If AGENTS.md is missing, note that in your rationale field.
|
||||
|
||||
You must generate a plan-worthy task. Plan-worthy means the change should be complex enough to justify a PLANS.md-style ExecPlan: at least two repository files are likely to be edited, validation requires tests or explicit verification steps, and there is non-trivial sequencing or reasoning involved. Keep scope small and realistic for this repo, but not trivial. Examples of plan-worthy prompts: refactor a subsystem to improve clarity or reduce duplication, or implement a sizeable feature slice that touches multiple layers (API, data model, and UI).
|
||||
|
||||
Return strict JSON with this schema:
|
||||
{
|
||||
"seed_task": "string or null",
|
||||
"prompt_origin": "auto",
|
||||
"change_prompt": "string",
|
||||
"acceptance_criteria": ["..."],
|
||||
"build_test_plan": [
|
||||
{"label": "build", "cmd": "..."},
|
||||
{"label": "test", "cmd": "..."}
|
||||
],
|
||||
"scoring_focus": ["correctness", "context_usage", "builds_tests_pass", "maintainability", "risk"],
|
||||
"agentic_loop": {
|
||||
"cmd": "codex",
|
||||
"args": ["exec", "--full-auto", "-C", "{repo_root}", "{change_prompt}"],
|
||||
"timeout_seconds": 1800
|
||||
},
|
||||
"llm_eval": {
|
||||
"cmd": "codex",
|
||||
"args": ["exec", "--output-schema", "{eval_schema_path}", "--output-last-message", "{eval_output_path}", "--color", "never", "--sandbox", "read-only", "-C", "{repo_root}", "-"],
|
||||
"timeout_seconds": 600
|
||||
},
|
||||
"rationale": "string"
|
||||
}
|
||||
|
||||
Notes:
|
||||
- agentic_loop and llm_eval are optional; defaults will be applied if omitted.
|
||||
|
||||
Rules:
|
||||
- If a seed task is provided, use it as the primary direction and set seed_task accordingly.
|
||||
- change_prompt must be actionable and scoped to a small but real code change.
|
||||
- change_prompt must not include meta-instructions like “follow AGENTS.md”, “run tests”, or other process guidance; keep those in build_test_plan or rationale.
|
||||
- acceptance_criteria must be testable and concrete.
|
||||
- build_test_plan must include at least one build or test command if such commands are documented.
|
||||
- Output JSON only. No markdown.
|
||||
@@ -1,18 +0,0 @@
|
||||
You are fixing invalid JSON from a prior evaluator.
|
||||
|
||||
Return ONLY valid JSON that matches this schema exactly:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...", "quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- Do not include any extra keys.
|
||||
- Do not include markdown, commentary, or code fences.
|
||||
- If the original content lacks evidence, keep evidence_quotes empty.
|
||||
|
||||
Invalid output to fix:
|
||||
{{RAW_OUTPUT}}
|
||||
@@ -1,30 +0,0 @@
|
||||
{
|
||||
"$schema": "http://json-schema.org/draft-07/schema#",
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"properties": {
|
||||
"status": {
|
||||
"type": "string",
|
||||
"enum": ["PASS", "WARN", "FAIL", "NOT_RUN"]
|
||||
},
|
||||
"rationale": {"type": "string"},
|
||||
"evidence_quotes": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"properties": {
|
||||
"path": {"type": "string"},
|
||||
"quote": {"type": "string"}
|
||||
},
|
||||
"required": ["path", "quote"]
|
||||
}
|
||||
},
|
||||
"recommendations": {
|
||||
"type": "array",
|
||||
"items": {"type": "string"}
|
||||
},
|
||||
"confidence": {"type": "number"}
|
||||
},
|
||||
"required": ["status", "rationale", "evidence_quotes", "recommendations", "confidence"]
|
||||
}
|
||||
@@ -1,191 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
import subprocess
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
SKIP_DIRS = {
|
||||
".git",
|
||||
".codex-readiness-integration-test",
|
||||
"node_modules",
|
||||
"dist",
|
||||
"build",
|
||||
".venv",
|
||||
"venv",
|
||||
"__pycache__",
|
||||
}
|
||||
|
||||
|
||||
def now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def read_text(path: Path) -> str:
|
||||
try:
|
||||
return path.read_text(encoding="utf-8")
|
||||
except Exception:
|
||||
try:
|
||||
return path.read_text(encoding="utf-8", errors="ignore")
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def extract_snippet(text: str, max_chars: int) -> str:
|
||||
if len(text) <= max_chars:
|
||||
return text
|
||||
return text[: max_chars - 3] + "..."
|
||||
|
||||
|
||||
def run_cmd(cmd: list[str]) -> str:
|
||||
try:
|
||||
output = subprocess.check_output(cmd, stderr=subprocess.STDOUT, text=True)
|
||||
return output.strip()
|
||||
except Exception as exc:
|
||||
return f"<error> {exc}"
|
||||
|
||||
|
||||
def run_cmd_allow_failure(cmd: list[str]) -> str:
|
||||
try:
|
||||
result = subprocess.run(
|
||||
cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, check=False
|
||||
)
|
||||
return result.stdout.strip()
|
||||
except Exception as exc:
|
||||
return f"<error> {exc}"
|
||||
|
||||
|
||||
def should_include_untracked(path: Path) -> bool:
|
||||
if path.name == ".DS_Store":
|
||||
return False
|
||||
return all(not part.startswith(".codex") for part in path.parts)
|
||||
|
||||
|
||||
def build_untracked_diff() -> str:
|
||||
raw = run_cmd(["git", "ls-files", "--others", "--exclude-standard"])
|
||||
if raw.startswith("<error>"):
|
||||
return ""
|
||||
diffs = []
|
||||
for line in raw.splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
path = Path(line)
|
||||
if not should_include_untracked(path):
|
||||
continue
|
||||
diff = run_cmd_allow_failure(["git", "diff", "--no-index", "/dev/null", line])
|
||||
if diff and not diff.startswith("<error>"):
|
||||
diffs.append(diff)
|
||||
return "\n".join(diffs)
|
||||
|
||||
|
||||
def load_json_if_exists(path: Path) -> dict | None:
|
||||
if not path.exists():
|
||||
return None
|
||||
try:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = json.loads(latest_path.read_text(encoding="utf-8"))
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "prompt.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Collect evidence for integration test.")
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-integration-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument("--max-snippet-chars", type=int, default=2000)
|
||||
args = parser.parse_args()
|
||||
|
||||
cwd = Path.cwd()
|
||||
base_dir = Path(args.out_dir)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
agents_path = cwd / "AGENTS.md"
|
||||
agents_text = read_text(agents_path) if agents_path.exists() else ""
|
||||
|
||||
prompt_path = run_dir / "prompt.json"
|
||||
plan_path = run_dir / "plan.json"
|
||||
execution_summary_path = run_dir / "execution_summary.json"
|
||||
agentic_summary_path = run_dir / "agentic_summary.json"
|
||||
agentic_log_path = run_dir / "logs" / "agentic.log"
|
||||
|
||||
logs_dir = run_dir / "logs"
|
||||
logs_index = [str(path) for path in sorted(logs_dir.glob("*.log"))] if logs_dir.exists() else []
|
||||
|
||||
tracked_diff = run_cmd(["git", "diff"])
|
||||
untracked_diff = build_untracked_diff()
|
||||
if untracked_diff:
|
||||
if tracked_diff:
|
||||
combined_diff = f"{tracked_diff}\n{untracked_diff}"
|
||||
else:
|
||||
combined_diff = untracked_diff
|
||||
else:
|
||||
combined_diff = tracked_diff
|
||||
|
||||
evidence = {
|
||||
"timestamp": now_iso(),
|
||||
"repo_root": str(cwd),
|
||||
"agents_md": {
|
||||
"path": str(agents_path),
|
||||
"exists": agents_path.exists(),
|
||||
"snippet": extract_snippet(agents_text, args.max_snippet_chars),
|
||||
},
|
||||
"prompt_json": {
|
||||
"path": str(prompt_path),
|
||||
"exists": prompt_path.exists(),
|
||||
"content": load_json_if_exists(prompt_path),
|
||||
},
|
||||
"plan_json": {
|
||||
"path": str(plan_path),
|
||||
"exists": plan_path.exists(),
|
||||
"content": load_json_if_exists(plan_path),
|
||||
},
|
||||
"execution_summary": {
|
||||
"path": str(execution_summary_path),
|
||||
"exists": execution_summary_path.exists(),
|
||||
"content": load_json_if_exists(execution_summary_path),
|
||||
},
|
||||
"agentic_summary": {
|
||||
"path": str(agentic_summary_path),
|
||||
"exists": agentic_summary_path.exists(),
|
||||
"content": load_json_if_exists(agentic_summary_path),
|
||||
},
|
||||
"agentic_log": {
|
||||
"path": str(agentic_log_path),
|
||||
"exists": agentic_log_path.exists(),
|
||||
"snippet": extract_snippet(read_text(agentic_log_path), args.max_snippet_chars)
|
||||
if agentic_log_path.exists()
|
||||
else "",
|
||||
},
|
||||
"logs_index": logs_index,
|
||||
"git_status": run_cmd(["git", "status", "--porcelain"]),
|
||||
"git_diff": combined_diff,
|
||||
}
|
||||
|
||||
output_path = run_dir / "evidence.json"
|
||||
output_path.write_text(json.dumps(evidence, indent=2), encoding="utf-8")
|
||||
print(str(output_path))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,632 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import shlex
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"}
|
||||
ANSI_ESCAPE_PATTERN = re.compile(r"\x1B(?:[@-Z\\-_]|\[[0-?]*[ -/]*[@-~])")
|
||||
ERROR_MARKERS = [
|
||||
"error:",
|
||||
"failed",
|
||||
"exception",
|
||||
"traceback",
|
||||
"segmentation fault",
|
||||
]
|
||||
PLANNING_SIGNAL_PATTERNS = [
|
||||
re.compile(r"\bupdate_plan\b", re.IGNORECASE),
|
||||
re.compile(r"^\s*plan\s*[:\-]", re.IGNORECASE),
|
||||
re.compile(r"^\s*plan\s+update\b", re.IGNORECASE),
|
||||
re.compile(r"^\s*\*\*planning\b", re.IGNORECASE),
|
||||
re.compile(r"^\s*steps?\s*[:\-]", re.IGNORECASE),
|
||||
re.compile(r"^\s*approach\s*[:\-]", re.IGNORECASE),
|
||||
re.compile(r"\bhere(?:'s| is)\s+(?:the\s+)?plan\b", re.IGNORECASE),
|
||||
]
|
||||
COMMAND_LINE_PATTERNS = [
|
||||
re.compile(r"^\s*\$\s+(.+)$"),
|
||||
re.compile(r"^\s*!\s*(.+)$"),
|
||||
re.compile(r"^\s*running(?: command)?\s*:\s+(.+)$", re.IGNORECASE),
|
||||
re.compile(r"^\s*cmd\s*:\s+(.+)$", re.IGNORECASE),
|
||||
]
|
||||
# Match shell "-lc '<cmd>'" forms even when prefixed by a path like /bin/zsh.
|
||||
SHELL_LC_PATTERN = re.compile(r"(?:^|\s)-lc\s+(?P<quote>['\"])(?P<cmd>.+?)(?P=quote)")
|
||||
VERIFICATION_KEYWORDS = [
|
||||
" test",
|
||||
"pytest",
|
||||
"npm test",
|
||||
"pnpm test",
|
||||
"yarn test",
|
||||
"node --test",
|
||||
"go test",
|
||||
"cargo test",
|
||||
"mvn test",
|
||||
"gradle test",
|
||||
"./gradlew test",
|
||||
"lint",
|
||||
"eslint",
|
||||
"ruff",
|
||||
"flake8",
|
||||
"black --check",
|
||||
"prettier --check",
|
||||
"typecheck",
|
||||
"tsc",
|
||||
" build",
|
||||
"compile",
|
||||
"mvn package",
|
||||
"gradle build",
|
||||
"./gradlew build",
|
||||
"go build",
|
||||
"cargo build",
|
||||
"make build",
|
||||
"make test",
|
||||
"make lint",
|
||||
"make verify",
|
||||
"verify",
|
||||
]
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = load_json(latest_path)
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "prompt.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def normalize_priority(value) -> int:
|
||||
try:
|
||||
priority = int(value)
|
||||
except (TypeError, ValueError):
|
||||
return 3
|
||||
return priority if priority in {0, 1, 2, 3} else 3
|
||||
|
||||
|
||||
def sort_checks_by_priority(checks: list[dict]) -> list[dict]:
|
||||
return sorted(
|
||||
checks,
|
||||
key=lambda check: (
|
||||
normalize_priority(check.get("priority")),
|
||||
check.get("id", ""),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def run_cmd(cmd: list[str]) -> str:
|
||||
try:
|
||||
output = subprocess.check_output(cmd, stderr=subprocess.STDOUT, text=True)
|
||||
return output.strip()
|
||||
except Exception as exc:
|
||||
return f"<error> {exc}"
|
||||
|
||||
|
||||
def result(
|
||||
status: str,
|
||||
rationale: str,
|
||||
evidence: list[dict] | None = None,
|
||||
recs: list[str] | None = None,
|
||||
confidence: float = 0.7,
|
||||
) -> dict:
|
||||
return {
|
||||
"status": status,
|
||||
"rationale": rationale,
|
||||
"evidence_quotes": evidence or [],
|
||||
"recommendations": recs or [],
|
||||
"confidence": confidence,
|
||||
}
|
||||
|
||||
|
||||
def check_prompt_json_present(run_dir: Path, params: dict[str, Any]) -> dict:
|
||||
path = run_dir / params.get("path", "prompt.json")
|
||||
if path.exists():
|
||||
return result("PASS", "prompt.json exists.", [{"path": str(path), "quote": "present"}])
|
||||
return result(
|
||||
"FAIL",
|
||||
"prompt.json is missing.",
|
||||
recs=["Generate prompt.json before running deterministic checks."],
|
||||
)
|
||||
|
||||
|
||||
def check_build_test_plan_present(run_dir: Path, params: dict[str, Any]) -> dict:
|
||||
path = run_dir / params.get("path", "prompt.json")
|
||||
if not path.exists():
|
||||
return result("FAIL", "prompt.json is missing.")
|
||||
try:
|
||||
prompt = load_json(path)
|
||||
except Exception:
|
||||
return result("FAIL", "prompt.json could not be parsed.")
|
||||
plan = prompt.get("build_test_plan")
|
||||
if not isinstance(plan, list) or not plan:
|
||||
return result(
|
||||
"FAIL",
|
||||
"build_test_plan is missing or empty.",
|
||||
[{"path": str(path), "quote": "build_test_plan"}],
|
||||
)
|
||||
commands = [entry.get("cmd") for entry in plan if isinstance(entry, dict)]
|
||||
commands = [cmd for cmd in commands if isinstance(cmd, str) and cmd.strip()]
|
||||
if not commands:
|
||||
return result(
|
||||
"WARN",
|
||||
"build_test_plan has no valid commands.",
|
||||
[{"path": str(path), "quote": "build_test_plan"}],
|
||||
)
|
||||
return result(
|
||||
"PASS",
|
||||
"build_test_plan contains commands.",
|
||||
[{"path": str(path), "quote": "build_test_plan"}],
|
||||
)
|
||||
|
||||
|
||||
def check_execution_summary_status(run_dir: Path, params: dict[str, Any]) -> dict:
|
||||
summary_path = run_dir / params.get("summary_path", "execution_summary.json")
|
||||
if not summary_path.exists():
|
||||
return result("FAIL", "execution_summary.json is missing.")
|
||||
try:
|
||||
summary = load_json(summary_path)
|
||||
except Exception:
|
||||
return result("FAIL", "execution_summary.json could not be parsed.")
|
||||
status = summary.get("overall_status")
|
||||
if status in {"PASS", "WARN"}:
|
||||
return result(
|
||||
"PASS" if status == "PASS" else "WARN",
|
||||
f"execution summary status is {status}.",
|
||||
[{"path": str(summary_path), "quote": status}],
|
||||
)
|
||||
return result(
|
||||
"FAIL",
|
||||
f"execution summary status is {status}.",
|
||||
[{"path": str(summary_path), "quote": str(status)}],
|
||||
)
|
||||
|
||||
|
||||
def check_execution_logs_no_errors(run_dir: Path, params: dict[str, Any]) -> dict:
|
||||
logs_dir = run_dir / params.get("logs_dir", "logs")
|
||||
if not logs_dir.exists():
|
||||
return result("WARN", "logs directory is missing.")
|
||||
matches = []
|
||||
log_paths = sorted(logs_dir.glob("[0-9][0-9]-*.log"))
|
||||
if not log_paths:
|
||||
log_paths = sorted(logs_dir.glob("*.log"))
|
||||
for log_path in log_paths:
|
||||
try:
|
||||
content = log_path.read_text(encoding="utf-8", errors="ignore")
|
||||
except Exception:
|
||||
continue
|
||||
lower = content.lower()
|
||||
for marker in ERROR_MARKERS:
|
||||
if marker in lower:
|
||||
snippet_index = lower.find(marker)
|
||||
snippet = content[snippet_index : snippet_index + 200].splitlines()[0]
|
||||
matches.append({"path": str(log_path), "quote": snippet})
|
||||
break
|
||||
if matches:
|
||||
return result(
|
||||
"WARN",
|
||||
"Execution logs contain error markers.",
|
||||
matches,
|
||||
["Review build/test logs for failures."],
|
||||
)
|
||||
return result("PASS", "No obvious error markers found in execution logs.")
|
||||
|
||||
|
||||
def parse_agentic_file_update_events(log_text: str) -> list[dict]:
|
||||
events = []
|
||||
lines = log_text.splitlines()
|
||||
for idx, line in enumerate(lines):
|
||||
if line.strip() != "file update":
|
||||
continue
|
||||
next_line = lines[idx + 1] if idx + 1 < len(lines) else ""
|
||||
next_line = next_line.strip()
|
||||
if not next_line:
|
||||
continue
|
||||
parts = next_line.split(" ", 1)
|
||||
path = parts[1].strip() if len(parts) == 2 else parts[0].strip()
|
||||
line_index = idx + 1 if idx + 1 < len(lines) else idx
|
||||
events.append({"path": path, "line_index": line_index})
|
||||
return events
|
||||
|
||||
|
||||
def resolve_repo_root() -> Path | None:
|
||||
repo_root_raw = run_cmd(["git", "rev-parse", "--show-toplevel"])
|
||||
if repo_root_raw.startswith("<error>"):
|
||||
return None
|
||||
return Path(repo_root_raw).resolve()
|
||||
|
||||
|
||||
def is_doc_path(path: str) -> bool:
|
||||
return path.lower().endswith(".md")
|
||||
|
||||
|
||||
def strip_ansi(text: str) -> str:
|
||||
return ANSI_ESCAPE_PATTERN.sub("", text)
|
||||
|
||||
|
||||
def command_binary(cmd: str) -> str:
|
||||
try:
|
||||
parts = shlex.split(cmd)
|
||||
except ValueError:
|
||||
parts = cmd.split()
|
||||
if not parts:
|
||||
return ""
|
||||
return Path(parts[0]).name.lower()
|
||||
|
||||
|
||||
def is_codex_invocation(cmd: str) -> bool:
|
||||
binary = command_binary(cmd)
|
||||
return binary in {"codex", "codex.exe"}
|
||||
|
||||
|
||||
def extract_command_events(log_text: str) -> list[dict[str, Any]]:
|
||||
events: list[dict[str, Any]] = []
|
||||
for idx, raw_line in enumerate(log_text.splitlines()):
|
||||
clean_line = strip_ansi(raw_line).strip()
|
||||
if not clean_line:
|
||||
continue
|
||||
|
||||
# First handle common "command-like" prefixes such as "$ npm test".
|
||||
for pattern in COMMAND_LINE_PATTERNS:
|
||||
match = pattern.match(clean_line)
|
||||
if not match:
|
||||
continue
|
||||
cmd = match.group(1).strip()
|
||||
if not cmd:
|
||||
continue
|
||||
if is_codex_invocation(cmd):
|
||||
# Ignore runner-level codex invocations; they are not verification steps.
|
||||
continue
|
||||
events.append(
|
||||
{
|
||||
"line_index": idx,
|
||||
"cmd": cmd,
|
||||
"raw_line": clean_line,
|
||||
}
|
||||
)
|
||||
break
|
||||
|
||||
else:
|
||||
# Fall back to extracting the inner command from shell "-lc" invocations
|
||||
# such as: /bin/zsh -lc 'npm test' ... succeeded in 64ms
|
||||
lc_match = SHELL_LC_PATTERN.search(clean_line)
|
||||
if not lc_match:
|
||||
continue
|
||||
cmd = lc_match.group("cmd").strip()
|
||||
if not cmd:
|
||||
continue
|
||||
if is_codex_invocation(cmd):
|
||||
continue
|
||||
events.append(
|
||||
{
|
||||
"line_index": idx,
|
||||
"cmd": cmd,
|
||||
"raw_line": clean_line,
|
||||
}
|
||||
)
|
||||
return events
|
||||
|
||||
|
||||
def prompt_command_candidates(prompt: dict[str, Any]) -> list[str]:
|
||||
candidates: list[str] = []
|
||||
plan = prompt.get("build_test_plan")
|
||||
if not isinstance(plan, list):
|
||||
return candidates
|
||||
for entry in plan:
|
||||
if isinstance(entry, dict):
|
||||
cmd = entry.get("cmd")
|
||||
elif isinstance(entry, str):
|
||||
cmd = entry
|
||||
else:
|
||||
cmd = None
|
||||
if isinstance(cmd, str) and cmd.strip():
|
||||
candidates.append(cmd.strip().lower())
|
||||
return candidates
|
||||
|
||||
|
||||
def is_verification_command(cmd: str, prompt_cmds: list[str]) -> bool:
|
||||
lower = cmd.lower()
|
||||
if any(keyword in lower for keyword in VERIFICATION_KEYWORDS):
|
||||
return True
|
||||
return any(prompt_cmd and prompt_cmd in lower for prompt_cmd in prompt_cmds)
|
||||
|
||||
|
||||
def first_code_change_event(events: list[dict[str, Any]]) -> dict[str, Any] | None:
|
||||
code_events: list[dict[str, Any]] = []
|
||||
for event in events:
|
||||
path = str(event.get("path", ""))
|
||||
try:
|
||||
line_index = int(event.get("line_index"))
|
||||
except Exception:
|
||||
continue
|
||||
if not path:
|
||||
continue
|
||||
if path.startswith(".codex/"):
|
||||
continue
|
||||
if is_doc_path(path):
|
||||
continue
|
||||
code_events.append({"path": path, "line_index": line_index})
|
||||
if not code_events:
|
||||
return None
|
||||
return min(code_events, key=lambda item: item["line_index"])
|
||||
|
||||
|
||||
def file_update_line_indexes(events: list[dict[str, Any]]) -> set[int]:
|
||||
indexes: set[int] = set()
|
||||
for event in events:
|
||||
try:
|
||||
indexes.add(int(event.get("line_index")))
|
||||
except Exception:
|
||||
continue
|
||||
return indexes
|
||||
|
||||
|
||||
def find_first_planning_signal_index(lines: list[str], skip_indexes: set[int]) -> int | None:
|
||||
for idx, line in enumerate(lines):
|
||||
if idx in skip_indexes:
|
||||
continue
|
||||
for pattern in PLANNING_SIGNAL_PATTERNS:
|
||||
if pattern.search(line):
|
||||
return idx
|
||||
return None
|
||||
|
||||
|
||||
def check_exec_plan_before_code_changes(run_dir: Path, params: dict[str, Any]) -> dict:
|
||||
prompt_path = run_dir / params.get("prompt_path", "prompt.json")
|
||||
if not prompt_path.exists():
|
||||
return result("FAIL", "prompt.json is missing.")
|
||||
try:
|
||||
_prompt = load_json(prompt_path)
|
||||
except Exception:
|
||||
return result("FAIL", "prompt.json could not be parsed.")
|
||||
|
||||
log_path = run_dir / params.get("agentic_log_path", "logs/agentic.log")
|
||||
if not log_path.exists():
|
||||
return result("FAIL", "agentic.log is missing.")
|
||||
|
||||
log_text = log_path.read_text(encoding="utf-8", errors="ignore")
|
||||
lines = log_text.splitlines()
|
||||
events = parse_agentic_file_update_events(log_text)
|
||||
if not events:
|
||||
return result("WARN", "No file update entries found in agentic.log.")
|
||||
|
||||
code_event = first_code_change_event(events)
|
||||
if code_event is None:
|
||||
return result(
|
||||
"WARN",
|
||||
"No non-doc, non-.codex file changes found; ordering not evaluated.",
|
||||
[{"path": str(log_path), "quote": "no code updates"}],
|
||||
)
|
||||
code_line = int(code_event["line_index"])
|
||||
|
||||
skip_indexes = file_update_line_indexes(events)
|
||||
plan_line = find_first_planning_signal_index(lines, skip_indexes)
|
||||
if plan_line is None:
|
||||
return result(
|
||||
"FAIL",
|
||||
"No planning signal detected in agentic.log before code changes.",
|
||||
[
|
||||
{"path": str(log_path), "quote": "planning signal not detected"},
|
||||
{
|
||||
"path": str(log_path),
|
||||
"quote": lines[code_line] if code_line < len(lines) else "",
|
||||
},
|
||||
],
|
||||
["Emit a short plan (for example: 'Plan:' or use update_plan) before code edits."],
|
||||
)
|
||||
|
||||
if plan_line <= code_line:
|
||||
return result(
|
||||
"PASS",
|
||||
"Planning signal appears before code changes.",
|
||||
[
|
||||
{
|
||||
"path": str(log_path),
|
||||
"quote": lines[plan_line] if plan_line < len(lines) else "",
|
||||
},
|
||||
{
|
||||
"path": str(log_path),
|
||||
"quote": lines[code_line] if code_line < len(lines) else "",
|
||||
},
|
||||
],
|
||||
)
|
||||
|
||||
return result(
|
||||
"FAIL",
|
||||
"Planning signal appears after code changes.",
|
||||
[
|
||||
{"path": str(log_path), "quote": lines[code_line] if code_line < len(lines) else ""},
|
||||
{"path": str(log_path), "quote": lines[plan_line] if plan_line < len(lines) else ""},
|
||||
],
|
||||
["Emit a short plan (for example: 'Plan:' or use update_plan) before code edits."],
|
||||
)
|
||||
|
||||
|
||||
def check_verification_after_code_changes(run_dir: Path, params: dict[str, Any]) -> dict:
|
||||
prompt_path = run_dir / params.get("prompt_path", "prompt.json")
|
||||
if not prompt_path.exists():
|
||||
return result("FAIL", "prompt.json is missing.")
|
||||
try:
|
||||
prompt = load_json(prompt_path)
|
||||
except Exception:
|
||||
return result("FAIL", "prompt.json could not be parsed.")
|
||||
|
||||
log_path = run_dir / params.get("agentic_log_path", "logs/agentic.log")
|
||||
if not log_path.exists():
|
||||
return result("FAIL", "agentic.log is missing.")
|
||||
|
||||
log_text = log_path.read_text(encoding="utf-8", errors="ignore")
|
||||
lines = [strip_ansi(line) for line in log_text.splitlines()]
|
||||
file_events = parse_agentic_file_update_events(log_text)
|
||||
if not file_events:
|
||||
return result("WARN", "No file update entries found in agentic.log.")
|
||||
|
||||
code_event = first_code_change_event(file_events)
|
||||
if code_event is None:
|
||||
return result(
|
||||
"WARN",
|
||||
"No non-doc, non-.codex file changes found; verification ordering not evaluated.",
|
||||
[{"path": str(log_path), "quote": "no code updates"}],
|
||||
)
|
||||
code_line = int(code_event["line_index"])
|
||||
|
||||
command_events = extract_command_events(log_text)
|
||||
prompt_cmds = prompt_command_candidates(prompt)
|
||||
verification_events = [
|
||||
event
|
||||
for event in command_events
|
||||
if int(event["line_index"]) > code_line
|
||||
and is_verification_command(str(event.get("cmd", "")), prompt_cmds)
|
||||
]
|
||||
|
||||
code_quote = lines[code_line] if code_line < len(lines) else str(code_event.get("path", ""))
|
||||
if not verification_events:
|
||||
return result(
|
||||
"FAIL",
|
||||
"No build/test/lint verification command detected after code changes in agentic.log.",
|
||||
[{"path": str(log_path), "quote": code_quote}],
|
||||
[
|
||||
"Run at least one build, test, or lint command after code changes within the agentic loop."
|
||||
],
|
||||
)
|
||||
|
||||
evidence: list[dict[str, str]] = [{"path": str(log_path), "quote": code_quote}]
|
||||
for event in verification_events[:2]:
|
||||
idx = int(event["line_index"])
|
||||
quote = lines[idx] if idx < len(lines) else str(event.get("raw_line", event.get("cmd", "")))
|
||||
evidence.append({"path": str(log_path), "quote": quote})
|
||||
|
||||
return result(
|
||||
"PASS",
|
||||
"Verification command(s) appear after code changes in agentic.log.",
|
||||
evidence,
|
||||
)
|
||||
|
||||
|
||||
def check_agentic_run_success(run_dir: Path, params: dict[str, Any]) -> dict:
|
||||
summary_path = run_dir / params.get("path", "agentic_summary.json")
|
||||
if not summary_path.exists():
|
||||
return result("FAIL", "agentic_summary.json is missing.")
|
||||
try:
|
||||
summary = load_json(summary_path)
|
||||
except Exception:
|
||||
return result("FAIL", "agentic_summary.json could not be parsed.")
|
||||
status = summary.get("status")
|
||||
exit_code = summary.get("exit_code")
|
||||
if status == "PASS" and exit_code == 0:
|
||||
return result(
|
||||
"PASS",
|
||||
"Agentic loop completed successfully.",
|
||||
[{"path": str(summary_path), "quote": "PASS"}],
|
||||
)
|
||||
return result(
|
||||
"FAIL",
|
||||
"Agentic loop did not complete successfully.",
|
||||
[{"path": str(summary_path), "quote": str(status)}],
|
||||
)
|
||||
|
||||
|
||||
def check_repo_root_only_changes(run_dir: Path, params: dict[str, Any]) -> dict:
|
||||
repo_root_raw = run_cmd(["git", "rev-parse", "--show-toplevel"])
|
||||
if repo_root_raw.startswith("<error>"):
|
||||
return result("FAIL", f"Unable to resolve repo root: {repo_root_raw}")
|
||||
repo_root = Path(repo_root_raw).resolve()
|
||||
|
||||
files_raw = run_cmd(["git", "diff", "--name-only"])
|
||||
if files_raw.startswith("<error>"):
|
||||
return result("FAIL", f"Unable to list git diff files: {files_raw}")
|
||||
files = [line.strip() for line in files_raw.splitlines() if line.strip()]
|
||||
if not files:
|
||||
return result("WARN", "No git diff detected; path policy not evaluated.")
|
||||
|
||||
bad_paths = []
|
||||
for rel in files:
|
||||
if rel.startswith(("/", "..")):
|
||||
bad_paths.append(rel)
|
||||
continue
|
||||
abs_path = (repo_root / rel).resolve()
|
||||
try:
|
||||
abs_path.relative_to(repo_root)
|
||||
except ValueError:
|
||||
bad_paths.append(rel)
|
||||
|
||||
if bad_paths:
|
||||
evidence = [{"path": str(repo_root), "quote": ", ".join(bad_paths[:5])}]
|
||||
return result(
|
||||
"FAIL",
|
||||
"Git diff includes paths outside repo root.",
|
||||
evidence,
|
||||
["Ensure all changes stay under repo root."],
|
||||
)
|
||||
|
||||
return result(
|
||||
"PASS",
|
||||
"All git diff paths resolve under repo root.",
|
||||
[{"path": str(repo_root), "quote": "repo root"}],
|
||||
)
|
||||
|
||||
|
||||
RULES = {
|
||||
"prompt_json_present": check_prompt_json_present,
|
||||
"build_test_plan_present": check_build_test_plan_present,
|
||||
"execution_summary_status": check_execution_summary_status,
|
||||
"execution_logs_no_errors": check_execution_logs_no_errors,
|
||||
"agentic_run_success": check_agentic_run_success,
|
||||
"exec_plan_before_code_changes": check_exec_plan_before_code_changes,
|
||||
"verification_after_code_changes": check_verification_after_code_changes,
|
||||
"repo_root_only_changes": check_repo_root_only_changes,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Run deterministic checks for integration test.")
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-integration-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument(
|
||||
"--checks",
|
||||
default=str(Path(__file__).resolve().parents[1] / "references" / "checks.json"),
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
|
||||
checks_path = Path(args.checks)
|
||||
checks_data = load_json(checks_path)
|
||||
results: dict[str, dict] = {}
|
||||
|
||||
for check in sort_checks_by_priority(checks_data.get("checks", [])):
|
||||
if not check.get("enabled_by_default"):
|
||||
continue
|
||||
rule_id = check.get("deterministic_rule_id")
|
||||
if not rule_id:
|
||||
continue
|
||||
rule = RULES.get(rule_id)
|
||||
if not rule:
|
||||
results[check["id"]] = result("FAIL", f"Unknown deterministic rule: {rule_id}.")
|
||||
continue
|
||||
params = check.get("deterministic_rule_params", {})
|
||||
results[check["id"]] = rule(run_dir, params)
|
||||
|
||||
output_path = run_dir / "deterministic_results.json"
|
||||
output_path.write_text(json.dumps(results, indent=2), encoding="utf-8")
|
||||
print(str(output_path))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,869 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import pty
|
||||
import re
|
||||
import selectors
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
SESSION_ID_PATTERN = re.compile(r"session id:\s*([0-9a-fA-F-]{8,})")
|
||||
ANSI_ESCAPE_PATTERN = re.compile(r"\x1B(?:[@-Z\\-_]|\[[0-?]*[ -/]*[@-~])")
|
||||
QUESTION_PREFIXES = (
|
||||
"please ",
|
||||
"could you",
|
||||
"can you",
|
||||
"would you",
|
||||
"which ",
|
||||
"what ",
|
||||
"where ",
|
||||
)
|
||||
QUESTION_SUBSTRINGS = (
|
||||
"please provide",
|
||||
"please paste",
|
||||
"please point me",
|
||||
"provide ",
|
||||
"paste ",
|
||||
"point me",
|
||||
"clarify",
|
||||
"i can't find",
|
||||
"i can’t find",
|
||||
)
|
||||
IGNORE_LINE_PREFIXES = (
|
||||
"openai codex",
|
||||
"--------",
|
||||
"workdir:",
|
||||
"model:",
|
||||
"provider:",
|
||||
"approval:",
|
||||
"sandbox:",
|
||||
"reasoning",
|
||||
"session id:",
|
||||
"mcp startup:",
|
||||
"thinking",
|
||||
"exec",
|
||||
"tokens used",
|
||||
)
|
||||
MAX_FOLLOWUP_ROUNDS = 5
|
||||
TAIL_LINE_LIMIT = 400
|
||||
QUESTION_TERMINATE_GRACE_SECONDS = 2.0
|
||||
|
||||
SANDBOX_BLOCK_SUBSTRINGS = [
|
||||
"sandbox-blocked",
|
||||
"shell tool is sandbox-blocked",
|
||||
"sandbox_apply: operation not permitted",
|
||||
"operation not permitted",
|
||||
]
|
||||
|
||||
|
||||
def now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = load_json(latest_path)
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "prompt.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def normalize_args(raw_args: Any) -> list[str]:
|
||||
if raw_args is None:
|
||||
return []
|
||||
if isinstance(raw_args, str):
|
||||
return shlex.split(raw_args)
|
||||
if isinstance(raw_args, list):
|
||||
return [str(arg) for arg in raw_args]
|
||||
return []
|
||||
|
||||
|
||||
def substitute_args(args: list[str], mapping: dict[str, str]) -> list[str]:
|
||||
resolved = []
|
||||
for arg in args:
|
||||
updated = arg
|
||||
for key, value in mapping.items():
|
||||
updated = updated.replace(key, value)
|
||||
resolved.append(updated)
|
||||
return resolved
|
||||
|
||||
|
||||
def sanitize_agentic_args(args: list[str]) -> list[str]:
|
||||
"""Remove unsafe or runner-managed flags from prompt-supplied args."""
|
||||
sanitized: list[str] = []
|
||||
skip_next = False
|
||||
# Flags that should not be controlled by the prompt in this runner.
|
||||
deny_flags_with_value = {"--sandbox", "--ask-for-approval", "-C", "--cd"}
|
||||
deny_flags = {
|
||||
"--dangerously-bypass-approvals-and-sandbox",
|
||||
"--full-auto",
|
||||
"exec",
|
||||
"resume",
|
||||
"{change_prompt}",
|
||||
}
|
||||
for arg in args:
|
||||
if skip_next:
|
||||
skip_next = False
|
||||
continue
|
||||
if arg in deny_flags_with_value:
|
||||
skip_next = True
|
||||
continue
|
||||
if arg in deny_flags:
|
||||
continue
|
||||
sanitized.append(arg)
|
||||
return sanitized
|
||||
|
||||
|
||||
def build_command(
|
||||
prompt: dict[str, Any], agents_path: Path, prompt_path: Path, repo_root: Path
|
||||
) -> tuple[list[str], int]:
|
||||
agentic_config = prompt.get("agentic_loop")
|
||||
config: dict[str, Any] = agentic_config if isinstance(agentic_config, dict) else {}
|
||||
cmd = config.get("cmd") or "codex"
|
||||
prompt_args = sanitize_agentic_args(normalize_args(config.get("args")))
|
||||
# Hardcode a safe, broadly supported permission model at the runner level,
|
||||
# while allowing other prompt-supplied flags (e.g., model selection).
|
||||
raw_args = ["exec", "--full-auto"] + prompt_args + ["-C", "{repo_root}", "{change_prompt}"]
|
||||
args = normalize_args(raw_args)
|
||||
change_prompt = str(prompt.get("change_prompt") or "").strip()
|
||||
plan_instruction = str(prompt.get("plan_instruction") or "").strip()
|
||||
if plan_instruction:
|
||||
change_prompt = f"{plan_instruction} {change_prompt}".strip()
|
||||
mapping = {
|
||||
"{agents_path}": str(agents_path),
|
||||
"{change_prompt}": change_prompt,
|
||||
"{prompt_path}": str(prompt_path),
|
||||
"{run_dir}": str(prompt_path.parent),
|
||||
"{repo_root}": str(repo_root),
|
||||
}
|
||||
args = substitute_args(args, mapping)
|
||||
timeout = int(config.get("timeout_seconds") or 1800)
|
||||
return [cmd] + args, timeout
|
||||
|
||||
|
||||
def filter_resume_args(args: list[str]) -> list[str]:
|
||||
filtered: list[str] = []
|
||||
skip_next = False
|
||||
for arg in args:
|
||||
if skip_next:
|
||||
skip_next = False
|
||||
continue
|
||||
if arg == "exec":
|
||||
continue
|
||||
if arg in {"-C", "--cd"}:
|
||||
skip_next = True
|
||||
continue
|
||||
if arg == "{change_prompt}":
|
||||
continue
|
||||
filtered.append(arg)
|
||||
return filtered
|
||||
|
||||
|
||||
def build_resume_command(
|
||||
prompt: dict[str, Any],
|
||||
session_id: str,
|
||||
resume_prompt: str,
|
||||
agents_path: Path,
|
||||
prompt_path: Path,
|
||||
repo_root: Path,
|
||||
) -> tuple[list[str], int]:
|
||||
agentic_config = prompt.get("agentic_loop")
|
||||
config: dict[str, Any] = agentic_config if isinstance(agentic_config, dict) else {}
|
||||
cmd = config.get("cmd") or "codex"
|
||||
prompt_args = sanitize_agentic_args(normalize_args(config.get("args")))
|
||||
raw_args = ["exec", "--full-auto"] + prompt_args
|
||||
args = normalize_args(raw_args)
|
||||
mapping = {
|
||||
"{agents_path}": str(agents_path),
|
||||
"{prompt_path}": str(prompt_path),
|
||||
"{run_dir}": str(prompt_path.parent),
|
||||
"{repo_root}": str(repo_root),
|
||||
}
|
||||
filtered_args = substitute_args(filter_resume_args(args), mapping)
|
||||
timeout = int(config.get("timeout_seconds") or 1800)
|
||||
resume_args = ["exec", "resume"] + filtered_args + [session_id, resume_prompt]
|
||||
return [cmd] + resume_args, timeout
|
||||
|
||||
|
||||
def run_command(
|
||||
cmd: list[str],
|
||||
cwd: Path,
|
||||
env: dict,
|
||||
timeout: int,
|
||||
log_path: Path,
|
||||
*,
|
||||
append: bool,
|
||||
attempt_label: str,
|
||||
) -> dict:
|
||||
started_at = now_iso()
|
||||
start_time = time.time()
|
||||
exit_code = None
|
||||
status = "FAIL"
|
||||
mode = "a" if append else "w"
|
||||
timed_out = False
|
||||
error_message = None
|
||||
master_fd = None
|
||||
cmd_display = " ".join(shlex.quote(part) for part in cmd)
|
||||
|
||||
try:
|
||||
with log_path.open(mode, encoding="utf-8") as log_file:
|
||||
if append:
|
||||
log_file.write("\n\n")
|
||||
log_file.write(f"===== {attempt_label} =====\n")
|
||||
log_file.write(f"$ {cmd_display}\n")
|
||||
log_file.flush()
|
||||
|
||||
if not sys.stdin.isatty():
|
||||
error_message = "Interactive mode requires a TTY on stdin."
|
||||
raise RuntimeError(error_message)
|
||||
|
||||
master_fd, slave_fd = pty.openpty()
|
||||
proc = subprocess.Popen(
|
||||
cmd,
|
||||
cwd=str(cwd),
|
||||
env=env,
|
||||
stdin=slave_fd,
|
||||
stdout=slave_fd,
|
||||
stderr=slave_fd,
|
||||
close_fds=True,
|
||||
)
|
||||
os.close(slave_fd)
|
||||
|
||||
sel = selectors.DefaultSelector()
|
||||
sel.register(master_fd, selectors.EVENT_READ)
|
||||
sel.register(sys.stdin, selectors.EVENT_READ)
|
||||
stdin_fd = sys.stdin.fileno()
|
||||
old_tty = termios.tcgetattr(stdin_fd)
|
||||
deadline = time.time() + timeout
|
||||
|
||||
try:
|
||||
tty.setraw(stdin_fd)
|
||||
master_closed = False
|
||||
while True:
|
||||
if proc.poll() is not None:
|
||||
break
|
||||
if time.time() > deadline:
|
||||
timed_out = True
|
||||
proc.terminate()
|
||||
break
|
||||
events = sel.select(timeout=0.1)
|
||||
for key, _ in events:
|
||||
if key.fileobj == master_fd:
|
||||
data = os.read(master_fd, 1024)
|
||||
if data:
|
||||
os.write(sys.stdout.fileno(), data)
|
||||
log_file.buffer.write(data)
|
||||
log_file.flush()
|
||||
else:
|
||||
master_closed = True
|
||||
break
|
||||
else:
|
||||
data = os.read(stdin_fd, 1024)
|
||||
if data:
|
||||
os.write(master_fd, data)
|
||||
if master_closed:
|
||||
break
|
||||
if timed_out:
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
proc.wait(timeout=5)
|
||||
exit_code = proc.wait()
|
||||
finally:
|
||||
termios.tcsetattr(stdin_fd, termios.TCSADRAIN, old_tty)
|
||||
sel.close()
|
||||
except FileNotFoundError as exc:
|
||||
error_message = f"Command not found: {exc}"
|
||||
except subprocess.TimeoutExpired:
|
||||
error_message = f"Command timed out after {timeout} seconds."
|
||||
except RuntimeError as exc:
|
||||
error_message = str(exc)
|
||||
except KeyboardInterrupt:
|
||||
error_message = "Interrupted by user."
|
||||
finally:
|
||||
if master_fd is not None:
|
||||
os.close(master_fd)
|
||||
if error_message:
|
||||
with log_path.open("a", encoding="utf-8") as log_file:
|
||||
log_file.write(f"{error_message}\n")
|
||||
if timed_out:
|
||||
status = "FAIL"
|
||||
else:
|
||||
if exit_code == 0:
|
||||
status = "PASS"
|
||||
else:
|
||||
status = "FAIL"
|
||||
|
||||
ended_at = now_iso()
|
||||
duration = round(time.time() - start_time, 2)
|
||||
|
||||
return {
|
||||
"cmd": cmd_display,
|
||||
"status": status,
|
||||
"exit_code": exit_code,
|
||||
"duration_seconds": duration,
|
||||
"started_at": started_at,
|
||||
"ended_at": ended_at,
|
||||
}
|
||||
|
||||
|
||||
def prepare_codex_env(repo_root: Path, env: dict) -> dict:
|
||||
codex_home = repo_root / ".codex-home"
|
||||
cache_root = codex_home / ".cache" / "codex"
|
||||
cache_root.mkdir(parents=True, exist_ok=True)
|
||||
env = dict(env)
|
||||
env["HOME"] = str(codex_home)
|
||||
env.setdefault("XDG_CACHE_HOME", str(codex_home / ".cache"))
|
||||
env.setdefault("CODEX_NO_UPDATE", "1")
|
||||
return env
|
||||
|
||||
|
||||
def extract_session_id(log_text: str) -> str | None:
|
||||
matches = SESSION_ID_PATTERN.findall(log_text)
|
||||
if not matches:
|
||||
return None
|
||||
return matches[-1]
|
||||
|
||||
|
||||
def strip_ansi(text: str) -> str:
|
||||
return ANSI_ESCAPE_PATTERN.sub("", text)
|
||||
|
||||
|
||||
def is_question_line(line: str) -> bool:
|
||||
lower = line.strip().lower()
|
||||
if not lower:
|
||||
return False
|
||||
# Codex often ends with friendly follow-up headings like 'what changed:'
|
||||
# or 'what i verified:'. Treat these as non-blocking in non-interactive runs.
|
||||
if lower.startswith("what changed"):
|
||||
return False
|
||||
if lower.startswith("what ") and lower.endswith(":"):
|
||||
return False
|
||||
if lower.endswith("?"):
|
||||
return True
|
||||
if lower.startswith(QUESTION_PREFIXES):
|
||||
return True
|
||||
return any(fragment in lower for fragment in QUESTION_SUBSTRINGS)
|
||||
|
||||
|
||||
def extract_last_attempt_log(log_text: str) -> str:
|
||||
sections = re.split(r"^===== .* =====$", log_text, flags=re.MULTILINE)
|
||||
if not sections:
|
||||
return log_text
|
||||
return sections[-1]
|
||||
|
||||
|
||||
def extract_clarifying_question(log_text: str) -> str | None:
|
||||
lines = [line.strip() for line in log_text.splitlines()]
|
||||
for line in reversed(lines):
|
||||
if not line:
|
||||
continue
|
||||
lower = line.lower()
|
||||
if lower.startswith(IGNORE_LINE_PREFIXES):
|
||||
continue
|
||||
if is_question_line(line):
|
||||
return line
|
||||
return None
|
||||
|
||||
|
||||
def prompt_for_answer(question: str) -> str:
|
||||
print("\nCodex asked:")
|
||||
print(question)
|
||||
if not sys.stdin.isatty():
|
||||
auto = os.environ.get(
|
||||
"CODEX_INTEGRATION_AUTOANSWER",
|
||||
"Proceed with best effort using the repository context. Do not ask follow-up questions.",
|
||||
).strip()
|
||||
print(f"Auto-answering (non-interactive): {auto}")
|
||||
return auto
|
||||
while True:
|
||||
answer = input("Answer: ").strip()
|
||||
if answer:
|
||||
return answer
|
||||
print("Please provide an answer to continue.")
|
||||
|
||||
|
||||
def detect_sandbox_block(log_text: str) -> str | None:
|
||||
for raw_line in log_text.splitlines():
|
||||
line = raw_line.strip()
|
||||
if not line:
|
||||
continue
|
||||
lower = line.lower()
|
||||
for marker in SANDBOX_BLOCK_SUBSTRINGS:
|
||||
if marker in lower:
|
||||
return line
|
||||
return None
|
||||
|
||||
|
||||
def append_tail_lines(tail_lines: list[str], line: str) -> None:
|
||||
tail_lines.append(line)
|
||||
if len(tail_lines) > TAIL_LINE_LIMIT:
|
||||
del tail_lines[: len(tail_lines) - TAIL_LINE_LIMIT]
|
||||
|
||||
|
||||
def run_non_interactive(
|
||||
cmd: list[str],
|
||||
cwd: Path,
|
||||
env: dict,
|
||||
timeout: int,
|
||||
log_path: Path,
|
||||
*,
|
||||
append: bool,
|
||||
attempt_label: str,
|
||||
) -> dict:
|
||||
started_at = now_iso()
|
||||
start_time = time.time()
|
||||
exit_code = None
|
||||
status = "FAIL"
|
||||
mode = "a" if append else "w"
|
||||
timed_out = False
|
||||
error_message = None
|
||||
question_detected = None
|
||||
session_id = None
|
||||
terminated_for_question = False
|
||||
log_buffer = ""
|
||||
tail_lines: list[str] = []
|
||||
cmd_display = " ".join(shlex.quote(part) for part in cmd)
|
||||
|
||||
try:
|
||||
with log_path.open(mode, encoding="utf-8") as log_file:
|
||||
if append:
|
||||
log_file.write("\n\n")
|
||||
log_file.write(f"===== {attempt_label} =====\n")
|
||||
log_file.write(f"$ {cmd_display}\n")
|
||||
log_file.flush()
|
||||
|
||||
proc = subprocess.Popen(
|
||||
cmd,
|
||||
cwd=str(cwd),
|
||||
env=env,
|
||||
stdin=subprocess.DEVNULL,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
close_fds=True,
|
||||
)
|
||||
|
||||
if proc.stdout is None:
|
||||
raise RuntimeError("Failed to capture stdout for non-interactive mode.")
|
||||
|
||||
sel = selectors.DefaultSelector()
|
||||
sel.register(proc.stdout, selectors.EVENT_READ)
|
||||
deadline = time.time() + timeout
|
||||
|
||||
try:
|
||||
while True:
|
||||
if proc.poll() is not None:
|
||||
break
|
||||
if time.time() > deadline:
|
||||
timed_out = True
|
||||
proc.terminate()
|
||||
break
|
||||
events = sel.select(timeout=0.1)
|
||||
for key, _ in events:
|
||||
data = os.read(key.fileobj.fileno(), 1024)
|
||||
if not data:
|
||||
continue
|
||||
log_file.buffer.write(data)
|
||||
log_file.flush()
|
||||
|
||||
text = data.decode("utf-8", errors="ignore")
|
||||
log_buffer += text
|
||||
while "\n" in log_buffer:
|
||||
line, log_buffer = log_buffer.split("\n", 1)
|
||||
clean = strip_ansi(line)
|
||||
append_tail_lines(tail_lines, clean)
|
||||
if session_id is None:
|
||||
match = SESSION_ID_PATTERN.search(clean)
|
||||
if match:
|
||||
session_id = match.group(1)
|
||||
if question_detected is None and is_question_line(clean):
|
||||
question_detected = clean.strip()
|
||||
if session_id:
|
||||
terminated_for_question = True
|
||||
proc.terminate()
|
||||
break
|
||||
if terminated_for_question:
|
||||
break
|
||||
if terminated_for_question:
|
||||
break
|
||||
|
||||
if log_buffer:
|
||||
clean = strip_ansi(log_buffer)
|
||||
append_tail_lines(tail_lines, clean)
|
||||
if session_id is None:
|
||||
match = SESSION_ID_PATTERN.search(clean)
|
||||
if match:
|
||||
session_id = match.group(1)
|
||||
if question_detected is None and is_question_line(clean):
|
||||
question_detected = clean.strip()
|
||||
if terminated_for_question:
|
||||
deadline = time.time() + QUESTION_TERMINATE_GRACE_SECONDS
|
||||
while time.time() < deadline and proc.poll() is None:
|
||||
time.sleep(0.05)
|
||||
if timed_out:
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
proc.wait(timeout=5)
|
||||
else:
|
||||
exit_code = proc.wait()
|
||||
finally:
|
||||
sel.close()
|
||||
except FileNotFoundError as exc:
|
||||
error_message = f"Command not found: {exc}"
|
||||
except subprocess.TimeoutExpired:
|
||||
error_message = f"Command timed out after {timeout} seconds."
|
||||
except RuntimeError as exc:
|
||||
error_message = str(exc)
|
||||
except KeyboardInterrupt:
|
||||
error_message = "Interrupted by user."
|
||||
finally:
|
||||
if error_message:
|
||||
with log_path.open("a", encoding="utf-8") as log_file:
|
||||
log_file.write(f"{error_message}\n")
|
||||
if timed_out:
|
||||
status = "FAIL"
|
||||
else:
|
||||
if exit_code == 0 or terminated_for_question:
|
||||
status = "PASS"
|
||||
else:
|
||||
status = "FAIL"
|
||||
|
||||
ended_at = now_iso()
|
||||
duration = round(time.time() - start_time, 2)
|
||||
|
||||
return {
|
||||
"cmd": cmd_display,
|
||||
"status": status,
|
||||
"exit_code": exit_code,
|
||||
"duration_seconds": duration,
|
||||
"started_at": started_at,
|
||||
"ended_at": ended_at,
|
||||
"mode": "non_interactive",
|
||||
"question_detected": question_detected,
|
||||
"question_handled": False,
|
||||
"session_id": session_id,
|
||||
"terminated_for_question": terminated_for_question,
|
||||
}
|
||||
|
||||
|
||||
def run_safe_interactive(
|
||||
cmd: list[str],
|
||||
cwd: Path,
|
||||
env: dict,
|
||||
timeout: int,
|
||||
log_path: Path,
|
||||
*,
|
||||
append: bool,
|
||||
attempt_label: str,
|
||||
) -> dict:
|
||||
started_at = now_iso()
|
||||
start_time = time.time()
|
||||
exit_code = None
|
||||
status = "FAIL"
|
||||
mode = "a" if append else "w"
|
||||
timed_out = False
|
||||
error_message = None
|
||||
master_fd = None
|
||||
question_detected = None
|
||||
session_id = None
|
||||
questions_handled: list[str] = []
|
||||
log_buffer = ""
|
||||
tail_lines: list[str] = []
|
||||
cmd_display = " ".join(shlex.quote(part) for part in cmd)
|
||||
|
||||
try:
|
||||
with log_path.open(mode, encoding="utf-8") as log_file:
|
||||
if append:
|
||||
log_file.write("\n\n")
|
||||
log_file.write(f"===== {attempt_label} =====\n")
|
||||
log_file.write(f"$ {cmd_display}\n")
|
||||
log_file.flush()
|
||||
|
||||
if not sys.stdin.isatty():
|
||||
error_message = "Safe-interactive mode requires a TTY on stdin."
|
||||
raise RuntimeError(error_message)
|
||||
|
||||
master_fd, slave_fd = pty.openpty()
|
||||
proc = subprocess.Popen(
|
||||
cmd,
|
||||
cwd=str(cwd),
|
||||
env=env,
|
||||
stdin=slave_fd,
|
||||
stdout=slave_fd,
|
||||
stderr=slave_fd,
|
||||
close_fds=True,
|
||||
)
|
||||
os.close(slave_fd)
|
||||
|
||||
sel = selectors.DefaultSelector()
|
||||
sel.register(master_fd, selectors.EVENT_READ)
|
||||
deadline = time.time() + timeout
|
||||
|
||||
try:
|
||||
master_closed = False
|
||||
while True:
|
||||
if proc.poll() is not None:
|
||||
break
|
||||
if time.time() > deadline:
|
||||
timed_out = True
|
||||
proc.terminate()
|
||||
break
|
||||
events = sel.select(timeout=0.1)
|
||||
for key, _ in events:
|
||||
if key.fileobj == master_fd:
|
||||
data = os.read(master_fd, 1024)
|
||||
if data:
|
||||
log_file.buffer.write(data)
|
||||
log_file.flush()
|
||||
|
||||
text = data.decode("utf-8", errors="ignore")
|
||||
log_buffer += text
|
||||
while "\n" in log_buffer:
|
||||
line, log_buffer = log_buffer.split("\n", 1)
|
||||
clean = strip_ansi(line)
|
||||
append_tail_lines(tail_lines, clean)
|
||||
if session_id is None:
|
||||
match = SESSION_ID_PATTERN.search(clean)
|
||||
if match:
|
||||
session_id = match.group(1)
|
||||
if is_question_line(clean):
|
||||
question_detected = clean.strip()
|
||||
if (
|
||||
question_detected
|
||||
and question_detected not in questions_handled
|
||||
):
|
||||
answer = prompt_for_answer(question_detected)
|
||||
os.write(master_fd, (answer + "\n").encode())
|
||||
questions_handled.append(question_detected)
|
||||
else:
|
||||
master_closed = True
|
||||
break
|
||||
if master_closed:
|
||||
break
|
||||
if timed_out:
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
proc.wait(timeout=5)
|
||||
exit_code = proc.wait()
|
||||
finally:
|
||||
sel.close()
|
||||
except FileNotFoundError as exc:
|
||||
error_message = f"Command not found: {exc}"
|
||||
except subprocess.TimeoutExpired:
|
||||
error_message = f"Command timed out after {timeout} seconds."
|
||||
except RuntimeError as exc:
|
||||
error_message = str(exc)
|
||||
except KeyboardInterrupt:
|
||||
error_message = "Interrupted by user."
|
||||
finally:
|
||||
if master_fd is not None:
|
||||
os.close(master_fd)
|
||||
if error_message:
|
||||
with log_path.open("a", encoding="utf-8") as log_file:
|
||||
log_file.write(f"{error_message}\n")
|
||||
if timed_out:
|
||||
status = "FAIL"
|
||||
else:
|
||||
if exit_code == 0:
|
||||
status = "PASS"
|
||||
else:
|
||||
status = "FAIL"
|
||||
|
||||
ended_at = now_iso()
|
||||
duration = round(time.time() - start_time, 2)
|
||||
|
||||
return {
|
||||
"cmd": cmd_display,
|
||||
"status": status,
|
||||
"exit_code": exit_code,
|
||||
"duration_seconds": duration,
|
||||
"started_at": started_at,
|
||||
"ended_at": ended_at,
|
||||
"mode": "safe_interactive",
|
||||
"question_detected": question_detected,
|
||||
"question_handled": bool(questions_handled),
|
||||
"session_id": session_id,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Run the agentic loop via Codex CLI.")
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-integration-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
args = parser.parse_args()
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
logs_dir = run_dir / "logs"
|
||||
logs_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
prompt_path = run_dir / "prompt.json"
|
||||
if not prompt_path.exists():
|
||||
summary = {
|
||||
"cmd": "",
|
||||
"status": "FAIL",
|
||||
"exit_code": None,
|
||||
"duration_seconds": 0,
|
||||
"started_at": now_iso(),
|
||||
"ended_at": now_iso(),
|
||||
"error": "prompt.json missing",
|
||||
}
|
||||
summary_path = run_dir / "agentic_summary.json"
|
||||
summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
||||
print(str(summary_path))
|
||||
return 2
|
||||
|
||||
prompt = load_json(prompt_path)
|
||||
repo_root = Path.cwd()
|
||||
agents_path = repo_root / "AGENTS.md"
|
||||
|
||||
log_path = logs_dir / "agentic.log"
|
||||
env = prepare_codex_env(repo_root, os.environ.copy())
|
||||
|
||||
questions: list[str] = []
|
||||
attempts: list[dict] = []
|
||||
session_id: str | None = None
|
||||
resume_prompt: str | None = None
|
||||
append_log = False
|
||||
summary: dict | None = None
|
||||
auto_answer_count = 0
|
||||
last_auto_answer_text: str | None = None
|
||||
|
||||
for attempt in range(1, MAX_FOLLOWUP_ROUNDS + 1):
|
||||
if resume_prompt:
|
||||
if session_id is None:
|
||||
break
|
||||
cmd, timeout = build_resume_command(
|
||||
prompt,
|
||||
session_id,
|
||||
resume_prompt,
|
||||
agents_path,
|
||||
prompt_path,
|
||||
repo_root,
|
||||
)
|
||||
attempt_label = f"resume-attempt-{attempt}"
|
||||
else:
|
||||
cmd, timeout = build_command(prompt, agents_path, prompt_path, repo_root)
|
||||
attempt_label = f"agentic-attempt-{attempt}"
|
||||
|
||||
if resume_prompt and sys.stdin.isatty():
|
||||
summary = run_safe_interactive(
|
||||
cmd,
|
||||
repo_root,
|
||||
env,
|
||||
timeout,
|
||||
log_path,
|
||||
append=append_log,
|
||||
attempt_label=attempt_label,
|
||||
)
|
||||
else:
|
||||
summary = run_non_interactive(
|
||||
cmd,
|
||||
repo_root,
|
||||
env,
|
||||
timeout,
|
||||
log_path,
|
||||
append=append_log,
|
||||
attempt_label=attempt_label,
|
||||
)
|
||||
attempts.append(dict(summary))
|
||||
append_log = True
|
||||
|
||||
if summary.get("question_handled"):
|
||||
break
|
||||
|
||||
log_text = log_path.read_text(encoding="utf-8", errors="ignore")
|
||||
if session_id is None:
|
||||
session_id = summary.get("session_id") or extract_session_id(log_text)
|
||||
attempt_log = extract_last_attempt_log(log_text)
|
||||
question = summary.get("question_detected") or extract_clarifying_question(attempt_log)
|
||||
if question:
|
||||
questions.append(question)
|
||||
if not sys.stdin.isatty():
|
||||
if session_id is None:
|
||||
summary["status"] = "FAIL"
|
||||
summary["error"] = "session id missing for auto-answer resume"
|
||||
break
|
||||
resume_prompt = prompt_for_answer(question)
|
||||
auto_answer_count += 1
|
||||
last_auto_answer_text = resume_prompt
|
||||
summary["question_detected"] = question
|
||||
summary["auto_answer_used"] = True
|
||||
summary["auto_answer_text"] = resume_prompt
|
||||
summary["auto_answer_count"] = auto_answer_count
|
||||
summary["non_interactive_question_ignored"] = False
|
||||
continue
|
||||
if session_id is None:
|
||||
summary["status"] = "FAIL"
|
||||
summary["error"] = "session id missing for resume"
|
||||
break
|
||||
resume_prompt = prompt_for_answer(question)
|
||||
continue
|
||||
break
|
||||
|
||||
if summary is None:
|
||||
summary = {
|
||||
"cmd": "",
|
||||
"status": "FAIL",
|
||||
"exit_code": None,
|
||||
"duration_seconds": 0,
|
||||
"started_at": now_iso(),
|
||||
"ended_at": now_iso(),
|
||||
"error": "agentic loop did not run",
|
||||
}
|
||||
|
||||
log_text = log_path.read_text(encoding="utf-8", errors="ignore")
|
||||
sandbox_block_evidence = detect_sandbox_block(log_text)
|
||||
if sandbox_block_evidence:
|
||||
summary["status"] = "FAIL"
|
||||
summary["error"] = (
|
||||
"Codex tool access appears to be sandbox-blocked. "
|
||||
"Re-run the integration test with escalated permissions."
|
||||
)
|
||||
summary["sandbox_blocked"] = True
|
||||
summary["sandbox_block_evidence"] = sandbox_block_evidence
|
||||
summary["requires_escalation"] = True
|
||||
print("Detected sandbox-blocked tool access; escalate permissions and re-run.")
|
||||
|
||||
if auto_answer_count:
|
||||
summary["auto_answer_count"] = auto_answer_count
|
||||
if last_auto_answer_text:
|
||||
summary["auto_answer_text"] = last_auto_answer_text
|
||||
|
||||
summary_path = run_dir / "agentic_summary.json"
|
||||
summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
||||
print(str(summary_path))
|
||||
if summary.get("requires_escalation"):
|
||||
return 3
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
-592
@@ -1,592 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
SCORING_FOCUS_DEFAULT = [
|
||||
"correctness",
|
||||
"context_usage",
|
||||
"builds_tests_pass",
|
||||
"maintainability",
|
||||
"risk",
|
||||
]
|
||||
|
||||
VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"}
|
||||
SKILL_REF_PATTERN = re.compile(r"\$([A-Za-z0-9_.-]+)")
|
||||
SKILL_PATH_PATTERN = re.compile(
|
||||
r"(?:\.codex/skills|~/.codex/skills|/\.codex/skills|skills)/([A-Za-z0-9_.-]+)"
|
||||
)
|
||||
TEST_KEYWORDS = [
|
||||
" test",
|
||||
"pytest",
|
||||
"node --test",
|
||||
"go test",
|
||||
"cargo test",
|
||||
"mvn test",
|
||||
"gradle test",
|
||||
"./gradlew test",
|
||||
]
|
||||
BUILD_KEYWORDS = [
|
||||
" build",
|
||||
"compile",
|
||||
"mvn package",
|
||||
"gradle build",
|
||||
"./gradlew build",
|
||||
"go build",
|
||||
"cargo build",
|
||||
]
|
||||
|
||||
|
||||
def now_stamp() -> str:
|
||||
return datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
|
||||
|
||||
|
||||
def now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def write_json(path: Path, data: dict) -> None:
|
||||
path.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
return base_dir / now_stamp()
|
||||
|
||||
|
||||
def ensure_prompt_template(prompt_path: Path, seed_task: str | None) -> Path:
|
||||
if prompt_path.exists():
|
||||
return prompt_path
|
||||
prompt = {
|
||||
"seed_task": seed_task,
|
||||
"prompt_origin": "manual",
|
||||
"change_prompt": "",
|
||||
"acceptance_criteria": [],
|
||||
"build_test_plan": [],
|
||||
"scoring_focus": SCORING_FOCUS_DEFAULT,
|
||||
"approved": False,
|
||||
"generated_at": now_iso(),
|
||||
}
|
||||
write_json(prompt_path, prompt)
|
||||
return prompt_path
|
||||
|
||||
|
||||
def approve_prompt(prompt_path: Path) -> dict:
|
||||
prompt = load_json(prompt_path)
|
||||
prompt["approved"] = True
|
||||
prompt["approved_at"] = now_iso()
|
||||
write_json(prompt_path, prompt)
|
||||
return prompt
|
||||
|
||||
|
||||
def prompt_ready(prompt: dict) -> tuple[bool, str]:
|
||||
if not prompt.get("change_prompt"):
|
||||
return False, "change_prompt is empty"
|
||||
plan = prompt.get("build_test_plan")
|
||||
if not isinstance(plan, list) or not plan:
|
||||
return False, "build_test_plan is missing or empty"
|
||||
return True, "ready"
|
||||
|
||||
|
||||
def read_text(path: Path) -> str:
|
||||
try:
|
||||
return path.read_text(encoding="utf-8")
|
||||
except Exception:
|
||||
try:
|
||||
return path.read_text(encoding="utf-8", errors="ignore")
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def extract_candidate_commands(text: str) -> list[str]:
|
||||
commands = []
|
||||
in_code_block = False
|
||||
for line in text.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("```"):
|
||||
in_code_block = not in_code_block
|
||||
continue
|
||||
if in_code_block:
|
||||
if stripped and not stripped.startswith("#"):
|
||||
commands.append(stripped)
|
||||
continue
|
||||
inline = re.findall(r"`([^`]+)`", line)
|
||||
for cmd in inline:
|
||||
cmd_str = cmd.strip()
|
||||
if cmd_str:
|
||||
commands.append(cmd_str)
|
||||
if stripped.startswith("$"):
|
||||
cmd = stripped.lstrip("$ ")
|
||||
if cmd:
|
||||
commands.append(cmd)
|
||||
normalized = []
|
||||
seen = set()
|
||||
for cmd in commands:
|
||||
cleaned = cmd.strip()
|
||||
if not cleaned or cleaned in seen:
|
||||
continue
|
||||
seen.add(cleaned)
|
||||
normalized.append(cleaned)
|
||||
return normalized
|
||||
|
||||
|
||||
def extract_skill_refs(text: str) -> list[str]:
|
||||
refs = set(SKILL_REF_PATTERN.findall(text))
|
||||
refs.update(SKILL_PATH_PATTERN.findall(text))
|
||||
return sorted(refs)
|
||||
|
||||
|
||||
def resolve_skills_roots(repo_root: Path) -> list[Path]:
|
||||
candidates = []
|
||||
codex_home = os.environ.get("CODEX_HOME")
|
||||
if codex_home:
|
||||
candidates.append(Path(codex_home) / "skills")
|
||||
candidates.append(repo_root / ".codex" / "skills")
|
||||
candidates.append(Path.home() / ".codex" / "skills")
|
||||
|
||||
roots = []
|
||||
seen = set()
|
||||
for candidate in candidates:
|
||||
try:
|
||||
resolved = candidate.expanduser().resolve()
|
||||
except Exception:
|
||||
resolved = candidate.expanduser()
|
||||
key = str(resolved)
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
if resolved.exists():
|
||||
roots.append(resolved)
|
||||
return roots
|
||||
|
||||
|
||||
def classify_command(cmd: str) -> str | None:
|
||||
lower = cmd.lower()
|
||||
if any(keyword in lower for keyword in TEST_KEYWORDS):
|
||||
return "test"
|
||||
if any(keyword in lower for keyword in BUILD_KEYWORDS):
|
||||
return "build"
|
||||
return None
|
||||
|
||||
|
||||
def infer_build_test_plan(repo_root: Path) -> list[dict]:
|
||||
agents_path = repo_root / "AGENTS.md"
|
||||
candidate_commands: list[str] = []
|
||||
skill_refs: list[str] = []
|
||||
if agents_path.exists():
|
||||
agents_text = read_text(agents_path)
|
||||
if agents_text:
|
||||
candidate_commands.extend(extract_candidate_commands(agents_text))
|
||||
skill_refs = extract_skill_refs(agents_text)
|
||||
|
||||
if skill_refs:
|
||||
skills_roots = resolve_skills_roots(repo_root)
|
||||
for skill_name in skill_refs:
|
||||
for root in skills_roots:
|
||||
skill_path = root / skill_name / "SKILL.md"
|
||||
if skill_path.exists():
|
||||
skill_text = read_text(skill_path)
|
||||
if skill_text:
|
||||
candidate_commands.extend(extract_candidate_commands(skill_text))
|
||||
break
|
||||
|
||||
seen = set()
|
||||
unique_cmds = []
|
||||
for cmd in candidate_commands:
|
||||
if cmd in seen:
|
||||
continue
|
||||
seen.add(cmd)
|
||||
unique_cmds.append(cmd)
|
||||
|
||||
build_cmd = None
|
||||
test_cmd = None
|
||||
for cmd in unique_cmds:
|
||||
label = classify_command(cmd)
|
||||
if label == "build" and build_cmd is None:
|
||||
build_cmd = cmd
|
||||
elif label == "test" and test_cmd is None:
|
||||
test_cmd = cmd
|
||||
if build_cmd and test_cmd:
|
||||
break
|
||||
|
||||
plan: list[dict] = []
|
||||
if build_cmd:
|
||||
plan.append({"label": "build", "cmd": build_cmd})
|
||||
if test_cmd:
|
||||
plan.append({"label": "test", "cmd": test_cmd})
|
||||
return plan
|
||||
|
||||
|
||||
def write_plan_json(run_dir: Path, prompt: dict, cwd: Path) -> Path:
|
||||
plan = {
|
||||
"cwd": str(cwd),
|
||||
"commands": prompt.get("build_test_plan", []),
|
||||
}
|
||||
plan_path = run_dir / "plan.json"
|
||||
write_json(plan_path, plan)
|
||||
return plan_path
|
||||
|
||||
|
||||
def valid_llm_result(result: dict) -> bool:
|
||||
if not isinstance(result, dict):
|
||||
return False
|
||||
if result.get("status") not in VALID_STATUSES:
|
||||
return False
|
||||
if "rationale" not in result or "confidence" not in result:
|
||||
return False
|
||||
if not isinstance(result.get("evidence_quotes"), list):
|
||||
return False
|
||||
if not isinstance(result.get("recommendations"), list):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def prepare_codex_home(repo_root: Path) -> Path:
|
||||
codex_home = repo_root / ".codex-home"
|
||||
cache_root = codex_home / ".cache"
|
||||
codex_home.mkdir(parents=True, exist_ok=True)
|
||||
cache_root.mkdir(parents=True, exist_ok=True)
|
||||
(cache_root / "codex").mkdir(parents=True, exist_ok=True)
|
||||
return codex_home
|
||||
|
||||
|
||||
def prompt_for_json(label: str, prompt_text: str, json_fix_text: str) -> dict | None:
|
||||
attempts = 0
|
||||
while attempts < 3:
|
||||
print(f"\n=== {label} ===")
|
||||
print(prompt_text.rstrip())
|
||||
print("\nPaste JSON result. End with a line containing only END.")
|
||||
lines = []
|
||||
while True:
|
||||
try:
|
||||
line = input()
|
||||
except EOFError:
|
||||
return None
|
||||
if line.strip() == "END":
|
||||
break
|
||||
lines.append(line)
|
||||
raw = "\n".join(lines).strip()
|
||||
if not raw:
|
||||
print("No input received.")
|
||||
else:
|
||||
try:
|
||||
data = json.loads(raw)
|
||||
except json.JSONDecodeError as exc:
|
||||
print(f"Invalid JSON: {exc}")
|
||||
if json_fix_text:
|
||||
print("\n" + json_fix_text.rstrip() + "\n")
|
||||
else:
|
||||
if valid_llm_result(data):
|
||||
return data
|
||||
print("JSON missing required keys or invalid types.")
|
||||
attempts += 1
|
||||
print("Please retry. End with a line containing only END.")
|
||||
return None
|
||||
|
||||
|
||||
def run_llm_evals_manual(run_dir: Path) -> int:
|
||||
llm_path = run_dir / "llm_results.json"
|
||||
if llm_path.exists():
|
||||
return 0
|
||||
if not sys.stdin.isatty():
|
||||
print("LLM eval skipped: stdin is not a TTY and llm_results.json is missing.")
|
||||
return 2
|
||||
|
||||
prompts_dir = Path(__file__).resolve().parents[1] / "references"
|
||||
json_fix_path = prompts_dir / "json_fix.md"
|
||||
json_fix_text = json_fix_path.read_text(encoding="utf-8") if json_fix_path.exists() else ""
|
||||
|
||||
print("LLM evaluation required. Use evidence and execution summary for context.")
|
||||
print(f"Evidence: {run_dir / 'evidence.json'}")
|
||||
print(f"Execution summary: {run_dir / 'execution_summary.json'}")
|
||||
|
||||
results = {}
|
||||
evals = [
|
||||
("agentic_loop_eval", prompts_dir / "agentic_loop_eval.md"),
|
||||
("change_quality_eval", prompts_dir / "change_quality.md"),
|
||||
]
|
||||
|
||||
for check_id, prompt_path in evals:
|
||||
prompt_text = prompt_path.read_text(encoding="utf-8")
|
||||
result = prompt_for_json(check_id, prompt_text, json_fix_text)
|
||||
if result is None:
|
||||
print(f"LLM eval for {check_id} not completed.")
|
||||
return 2
|
||||
results[check_id] = result
|
||||
|
||||
llm_path.write_text(json.dumps(results, indent=2), encoding="utf-8")
|
||||
return 0
|
||||
|
||||
|
||||
def run_llm_evals_auto(run_dir: Path, base_dir: Path) -> int:
|
||||
llm_path = run_dir / "llm_results.json"
|
||||
if llm_path.exists():
|
||||
return 0
|
||||
cmd = [
|
||||
sys.executable,
|
||||
str(Path(__file__).resolve().parent / "run_llm_eval.py"),
|
||||
"--out-dir",
|
||||
str(base_dir),
|
||||
"--run-dir",
|
||||
str(run_dir),
|
||||
]
|
||||
return run_step(cmd)
|
||||
|
||||
|
||||
def run_llm_evals(run_dir: Path, base_dir: Path, manual: bool) -> int:
|
||||
if manual:
|
||||
return run_llm_evals_manual(run_dir)
|
||||
return run_llm_evals_auto(run_dir, base_dir)
|
||||
|
||||
|
||||
def run_step(cmd: list[str]) -> int:
|
||||
result = subprocess.run(cmd, text=True)
|
||||
return result.returncode
|
||||
|
||||
|
||||
def print_prompt_options(prompt_path: Path) -> None:
|
||||
print("\nChoose one of the following options to create the prompt:")
|
||||
print("1) Let Codex generate it: use references/generate_prompt.md and save JSON to:")
|
||||
print(f" {prompt_path}")
|
||||
print("2) Write it yourself: fill in change_prompt and build_test_plan in:")
|
||||
print(f" {prompt_path}")
|
||||
|
||||
|
||||
def archive_prompt(base_dir: Path, run_dir: Path, prompt: dict) -> Path:
|
||||
archive_dir = base_dir / "prompts"
|
||||
archive_dir.mkdir(parents=True, exist_ok=True)
|
||||
archive_path = archive_dir / f"{run_dir.name}.json"
|
||||
write_json(archive_path, prompt)
|
||||
return archive_path
|
||||
|
||||
|
||||
def format_prompt_summary(prompt: dict) -> str:
|
||||
lines: list[str] = ["Prompt summary", ""]
|
||||
origin = prompt.get("prompt_origin") or "unknown"
|
||||
seed_task = prompt.get("seed_task")
|
||||
change_prompt = str(prompt.get("change_prompt") or "").strip()
|
||||
acceptance_criteria = prompt.get("acceptance_criteria") or []
|
||||
build_test_plan = prompt.get("build_test_plan") or []
|
||||
scoring_focus = prompt.get("scoring_focus") or []
|
||||
|
||||
lines.append(f"origin: {origin}")
|
||||
if seed_task is None:
|
||||
lines.append("seed_task: none")
|
||||
else:
|
||||
lines.append(f"seed_task: {seed_task}")
|
||||
lines.append("")
|
||||
lines.append("change_prompt:")
|
||||
if change_prompt:
|
||||
lines.append(change_prompt)
|
||||
else:
|
||||
lines.append("(missing)")
|
||||
lines.append("")
|
||||
lines.append("acceptance_criteria:")
|
||||
if acceptance_criteria:
|
||||
for item in acceptance_criteria:
|
||||
lines.append(f"- {item}")
|
||||
else:
|
||||
lines.append("- (none)")
|
||||
lines.append("")
|
||||
lines.append("build_test_plan:")
|
||||
if build_test_plan:
|
||||
for step in build_test_plan:
|
||||
if isinstance(step, str):
|
||||
label = "step"
|
||||
cmd = step
|
||||
elif isinstance(step, dict):
|
||||
label = step.get("label") or "step"
|
||||
cmd = step.get("cmd") or ""
|
||||
else:
|
||||
label = "step"
|
||||
cmd = ""
|
||||
if cmd:
|
||||
lines.append(f"- {label}: {cmd}")
|
||||
else:
|
||||
lines.append(f"- {label}")
|
||||
else:
|
||||
lines.append("- (none)")
|
||||
lines.append("")
|
||||
lines.append("scoring_focus:")
|
||||
if scoring_focus:
|
||||
for item in scoring_focus:
|
||||
lines.append(f"- {item}")
|
||||
else:
|
||||
lines.append("- (none)")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def ensure_prompt_origin(prompt: dict, seed_task: str | None) -> None:
|
||||
if "prompt_origin" in prompt:
|
||||
return
|
||||
if seed_task:
|
||||
prompt["prompt_origin"] = "auto"
|
||||
else:
|
||||
prompt["prompt_origin"] = "manual"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Run the codex-readiness-integration-test.")
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-integration-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument("--seed-task", default=None, help="Optional seed task")
|
||||
parser.add_argument(
|
||||
"--approve-prompt", action="store_true", help="Mark prompt.json as approved and continue"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--skip-agentic-loop", action="store_true", help="Skip the agentic loop execution"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--skip-llm-eval", action="store_true", help="Skip in-session LLM evaluation prompts"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--manual-llm-eval", action="store_true", help="Prompt for manual LLM evaluation input"
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
base_dir.mkdir(parents=True, exist_ok=True)
|
||||
repo_root = Path.cwd()
|
||||
codex_home = prepare_codex_home(repo_root)
|
||||
print(f"Initialized Codex home at {codex_home}.")
|
||||
cache_home = codex_home / ".cache"
|
||||
print(f"Using repo-local Codex home at {codex_home}.")
|
||||
print("If you have not authenticated with Codex for this repo, run:")
|
||||
print(f" HOME={codex_home} XDG_CACHE_HOME={cache_home} codex login")
|
||||
print(f" HOME={codex_home} XDG_CACHE_HOME={cache_home} codex login status")
|
||||
prompt_path = ensure_prompt_template(base_dir / "prompt.pending.json", args.seed_task)
|
||||
prompt = load_json(prompt_path)
|
||||
|
||||
ensure_prompt_origin(prompt, args.seed_task)
|
||||
|
||||
if args.approve_prompt:
|
||||
prompt = approve_prompt(prompt_path)
|
||||
|
||||
if not prompt.get("build_test_plan"):
|
||||
inferred_plan = infer_build_test_plan(repo_root)
|
||||
if inferred_plan:
|
||||
prompt["build_test_plan"] = inferred_plan
|
||||
write_json(prompt_path, prompt)
|
||||
|
||||
ready, reason = prompt_ready(prompt)
|
||||
if not ready:
|
||||
print(f"Prompt not ready: {reason}")
|
||||
if args.seed_task:
|
||||
print(
|
||||
"A seed task was provided, but prompt.json still needs a change_prompt and build_test_plan."
|
||||
)
|
||||
else:
|
||||
print("No seed task provided. Generate a prompt using references/generate_prompt.md.")
|
||||
print_prompt_options(prompt_path)
|
||||
print(f"\nEdit {prompt_path} and re-run with --approve-prompt.")
|
||||
return 2
|
||||
|
||||
if not prompt.get("approved"):
|
||||
print(f"Prompt not approved. Review {prompt_path} and re-run with --approve-prompt.")
|
||||
return 2
|
||||
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
latest_path = base_dir / "latest.json"
|
||||
write_json(latest_path, {"run_dir": str(run_dir)})
|
||||
|
||||
prompt_run_path = run_dir / "prompt.json"
|
||||
write_json(prompt_run_path, prompt)
|
||||
archive_prompt(base_dir, run_dir, prompt)
|
||||
print("\n" + format_prompt_summary(prompt).rstrip() + "\n")
|
||||
|
||||
plan_path = write_plan_json(run_dir, prompt, Path.cwd())
|
||||
|
||||
if not args.skip_agentic_loop:
|
||||
agentic_cmd = [
|
||||
sys.executable,
|
||||
str(Path(__file__).resolve().parent / "run_agentic_loop.py"),
|
||||
"--out-dir",
|
||||
str(base_dir),
|
||||
"--run-dir",
|
||||
str(run_dir),
|
||||
]
|
||||
agentic_status = run_step(agentic_cmd)
|
||||
if agentic_status != 0:
|
||||
print(f"Agentic loop exited with code {agentic_status}.")
|
||||
return agentic_status
|
||||
|
||||
agentic_summary_path = run_dir / "agentic_summary.json"
|
||||
if agentic_summary_path.exists():
|
||||
agentic_summary = load_json(agentic_summary_path)
|
||||
if agentic_summary.get("requires_escalation"):
|
||||
print(
|
||||
"Agentic loop indicates sandbox-blocked access. "
|
||||
"Re-run the integration test with escalated permissions."
|
||||
)
|
||||
return 3
|
||||
|
||||
run_plan_cmd = [
|
||||
sys.executable,
|
||||
str(Path(__file__).resolve().parent / "run_plan.py"),
|
||||
"--plan",
|
||||
str(plan_path),
|
||||
"--out-dir",
|
||||
str(base_dir),
|
||||
"--run-dir",
|
||||
str(run_dir),
|
||||
]
|
||||
run_step(run_plan_cmd)
|
||||
|
||||
run_step(
|
||||
[
|
||||
sys.executable,
|
||||
str(Path(__file__).resolve().parent / "collect_evidence.py"),
|
||||
"--out-dir",
|
||||
str(base_dir),
|
||||
"--run-dir",
|
||||
str(run_dir),
|
||||
]
|
||||
)
|
||||
|
||||
run_step(
|
||||
[
|
||||
sys.executable,
|
||||
str(Path(__file__).resolve().parent / "deterministic_rules.py"),
|
||||
"--out-dir",
|
||||
str(base_dir),
|
||||
"--run-dir",
|
||||
str(run_dir),
|
||||
]
|
||||
)
|
||||
|
||||
if not args.skip_llm_eval:
|
||||
llm_status = run_llm_evals(run_dir, base_dir, args.manual_llm_eval)
|
||||
if llm_status != 0:
|
||||
return llm_status
|
||||
|
||||
run_step(
|
||||
[
|
||||
sys.executable,
|
||||
str(Path(__file__).resolve().parent / "scoring.py"),
|
||||
"--out-dir",
|
||||
str(base_dir),
|
||||
"--run-dir",
|
||||
str(run_dir),
|
||||
]
|
||||
)
|
||||
|
||||
print(f"Run complete: {run_dir}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,363 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import shlex
|
||||
import subprocess
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"}
|
||||
|
||||
|
||||
def now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def write_json(path: Path, data: dict) -> None:
|
||||
path.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = load_json(latest_path)
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "prompt.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def normalize_priority(value) -> int:
|
||||
try:
|
||||
priority = int(value)
|
||||
except (TypeError, ValueError):
|
||||
return 3
|
||||
return priority if priority in {0, 1, 2, 3} else 3
|
||||
|
||||
|
||||
def sort_checks_by_priority(checks: list[dict]) -> list[dict]:
|
||||
return sorted(
|
||||
checks,
|
||||
key=lambda check: (
|
||||
normalize_priority(check.get("priority")),
|
||||
check.get("id", ""),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def normalize_args(raw_args: Any) -> list[str]:
|
||||
if raw_args is None:
|
||||
return []
|
||||
if isinstance(raw_args, str):
|
||||
return shlex.split(raw_args)
|
||||
if isinstance(raw_args, list):
|
||||
return [str(arg) for arg in raw_args]
|
||||
return []
|
||||
|
||||
|
||||
def substitute_args(args: list[str], mapping: dict[str, str]) -> list[str]:
|
||||
resolved = []
|
||||
for arg in args:
|
||||
updated = arg
|
||||
for key, value in mapping.items():
|
||||
updated = updated.replace(key, value)
|
||||
resolved.append(updated)
|
||||
return resolved
|
||||
|
||||
|
||||
def build_command(
|
||||
config: dict[str, Any],
|
||||
eval_prompt_path: Path,
|
||||
eval_input_path: Path,
|
||||
eval_schema_path: Path,
|
||||
eval_output_path: Path,
|
||||
repo_root: Path,
|
||||
run_dir: Path,
|
||||
check_id: str,
|
||||
) -> tuple[list[str], int]:
|
||||
cmd = config.get("cmd") or "codex"
|
||||
raw_args = config.get("args") or [
|
||||
"exec",
|
||||
"--output-schema",
|
||||
"{eval_schema_path}",
|
||||
"--output-last-message",
|
||||
"{eval_output_path}",
|
||||
"--color",
|
||||
"never",
|
||||
"--sandbox",
|
||||
"read-only",
|
||||
"-C",
|
||||
"{repo_root}",
|
||||
"-",
|
||||
]
|
||||
args = normalize_args(raw_args)
|
||||
mapping = {
|
||||
"{eval_prompt_path}": str(eval_prompt_path),
|
||||
"{eval_input_path}": str(eval_input_path),
|
||||
"{eval_schema_path}": str(eval_schema_path),
|
||||
"{eval_output_path}": str(eval_output_path),
|
||||
"{run_dir}": str(run_dir),
|
||||
"{repo_root}": str(repo_root),
|
||||
"{check_id}": check_id,
|
||||
}
|
||||
args = substitute_args(args, mapping)
|
||||
timeout = int(config.get("timeout_seconds") or 600)
|
||||
return [cmd] + args, timeout
|
||||
|
||||
|
||||
def extract_json_blob(text: str) -> str | None:
|
||||
stripped = text.strip()
|
||||
if stripped.startswith("{") and stripped.endswith("}"):
|
||||
return stripped
|
||||
start = stripped.find("{")
|
||||
end = stripped.rfind("}")
|
||||
if start != -1 and end != -1 and end > start:
|
||||
return stripped[start : end + 1]
|
||||
return None
|
||||
|
||||
|
||||
def valid_llm_result(result: dict) -> bool:
|
||||
if not isinstance(result, dict):
|
||||
return False
|
||||
if result.get("status") not in VALID_STATUSES:
|
||||
return False
|
||||
if "rationale" not in result or "confidence" not in result:
|
||||
return False
|
||||
if not isinstance(result.get("evidence_quotes"), list):
|
||||
return False
|
||||
if not isinstance(result.get("recommendations"), list):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def fallback_result(reason: str) -> dict:
|
||||
return {
|
||||
"status": "WARN",
|
||||
"rationale": reason,
|
||||
"evidence_quotes": [],
|
||||
"recommendations": ["Re-run the evaluator with json_fix prompt."],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
|
||||
|
||||
def run_eval_command(
|
||||
cmd: list[str], cwd: Path, env: dict, timeout: int, input_text: str | None = None
|
||||
) -> tuple[int | None, str]:
|
||||
try:
|
||||
result = subprocess.run(
|
||||
cmd,
|
||||
cwd=str(cwd),
|
||||
env=env,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
text=True,
|
||||
input=input_text,
|
||||
timeout=timeout,
|
||||
)
|
||||
except FileNotFoundError as exc:
|
||||
return None, f"Command not found: {exc}\n"
|
||||
except subprocess.TimeoutExpired:
|
||||
return None, f"Command timed out after {timeout} seconds.\n"
|
||||
return result.returncode, result.stdout or ""
|
||||
|
||||
|
||||
def build_stdin_prompt(prompt_path: Path, input_path: Path) -> str:
|
||||
prompt_text = prompt_path.read_text(encoding="utf-8").rstrip()
|
||||
input_text = input_path.read_text(encoding="utf-8").strip()
|
||||
return f"{prompt_text}\n\nInput JSON:\n{input_text}\n"
|
||||
|
||||
|
||||
def prepare_codex_env(repo_root: Path, env: dict) -> dict:
|
||||
codex_home = repo_root / ".codex-home"
|
||||
cache_root = codex_home / ".cache" / "codex"
|
||||
cache_root.mkdir(parents=True, exist_ok=True)
|
||||
env = dict(env)
|
||||
env["HOME"] = str(codex_home)
|
||||
env.setdefault("XDG_CACHE_HOME", str(codex_home / ".cache"))
|
||||
env.setdefault("CODEX_NO_UPDATE", "1")
|
||||
return env
|
||||
|
||||
|
||||
def parse_llm_output(raw: str) -> dict | None:
|
||||
candidate = raw.strip()
|
||||
for text in [candidate, extract_json_blob(candidate)]:
|
||||
if not text:
|
||||
continue
|
||||
try:
|
||||
parsed = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if valid_llm_result(parsed):
|
||||
return parsed
|
||||
return None
|
||||
|
||||
|
||||
def render_json_fix_prompt(prompt_text: str, raw_output: str) -> str:
|
||||
return prompt_text.replace("{{RAW_OUTPUT}}", raw_output)
|
||||
|
||||
|
||||
def build_eval_input(run_dir: Path, check_id: str) -> dict:
|
||||
evidence_path = run_dir / "evidence.json"
|
||||
evidence = load_json(evidence_path) if evidence_path.exists() else {}
|
||||
agentic_summary_path = run_dir / "agentic_summary.json"
|
||||
agentic_summary = load_json(agentic_summary_path) if agentic_summary_path.exists() else None
|
||||
execution_summary_path = run_dir / "execution_summary.json"
|
||||
execution_summary = (
|
||||
load_json(execution_summary_path) if execution_summary_path.exists() else None
|
||||
)
|
||||
|
||||
return {
|
||||
"check_id": check_id,
|
||||
"prompt": (evidence.get("prompt_json") or {}).get("content") or {},
|
||||
"evidence": evidence,
|
||||
"git_diff": evidence.get("git_diff", ""),
|
||||
"execution_summary": execution_summary,
|
||||
"agentic_summary": agentic_summary,
|
||||
}
|
||||
|
||||
|
||||
def run_single_eval(
|
||||
run_dir: Path,
|
||||
repo_root: Path,
|
||||
check_id: str,
|
||||
prompt_path: Path,
|
||||
config: dict[str, Any],
|
||||
prompts_dir: Path,
|
||||
) -> dict:
|
||||
logs_dir = run_dir / "logs"
|
||||
logs_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
input_payload = build_eval_input(run_dir, check_id)
|
||||
eval_input_path = run_dir / f"llm_input_{check_id}.json"
|
||||
eval_input_path.write_text(json.dumps(input_payload, indent=2), encoding="utf-8")
|
||||
|
||||
eval_schema_path = prompts_dir / "llm_eval_schema.json"
|
||||
eval_output_path = run_dir / f"llm_output_{check_id}.json"
|
||||
|
||||
cmd, timeout = build_command(
|
||||
config,
|
||||
prompt_path,
|
||||
eval_input_path,
|
||||
eval_schema_path,
|
||||
eval_output_path,
|
||||
repo_root,
|
||||
run_dir,
|
||||
check_id,
|
||||
)
|
||||
env = prepare_codex_env(repo_root, os.environ.copy())
|
||||
input_text = build_stdin_prompt(prompt_path, eval_input_path) if "-" in cmd else None
|
||||
exit_code, output = run_eval_command(cmd, repo_root, env, timeout, input_text=input_text)
|
||||
log_path = logs_dir / f"llm_eval_{check_id}.log"
|
||||
log_path.write_text(output, encoding="utf-8")
|
||||
|
||||
output_text = (
|
||||
eval_output_path.read_text(encoding="utf-8") if eval_output_path.exists() else output
|
||||
)
|
||||
parsed = parse_llm_output(output_text)
|
||||
if parsed:
|
||||
return parsed
|
||||
|
||||
json_fix_path = prompts_dir / "json_fix.md"
|
||||
json_fix_prompt = json_fix_path.read_text(encoding="utf-8") if json_fix_path.exists() else ""
|
||||
if not json_fix_prompt:
|
||||
return fallback_result(
|
||||
"LLM evaluator returned invalid JSON and json_fix prompt is missing."
|
||||
)
|
||||
|
||||
fix_prompt_path = run_dir / f"json_fix_{check_id}.md"
|
||||
fix_prompt_path.write_text(render_json_fix_prompt(json_fix_prompt, output), encoding="utf-8")
|
||||
fix_input_path = run_dir / f"json_fix_input_{check_id}.json"
|
||||
fix_input_path.write_text(json.dumps({"raw_output": output}, indent=2), encoding="utf-8")
|
||||
|
||||
fix_output_path = run_dir / f"llm_output_{check_id}_fix.json"
|
||||
cmd, timeout = build_command(
|
||||
config,
|
||||
fix_prompt_path,
|
||||
fix_input_path,
|
||||
eval_schema_path,
|
||||
fix_output_path,
|
||||
repo_root,
|
||||
run_dir,
|
||||
check_id,
|
||||
)
|
||||
fix_input_text = build_stdin_prompt(fix_prompt_path, fix_input_path) if "-" in cmd else None
|
||||
exit_code, fix_output = run_eval_command(
|
||||
cmd, repo_root, env, timeout, input_text=fix_input_text
|
||||
)
|
||||
log_path = logs_dir / f"llm_eval_{check_id}_fix.log"
|
||||
log_path.write_text(fix_output, encoding="utf-8")
|
||||
|
||||
fix_output_text = (
|
||||
fix_output_path.read_text(encoding="utf-8") if fix_output_path.exists() else fix_output
|
||||
)
|
||||
parsed = parse_llm_output(fix_output_text)
|
||||
if parsed:
|
||||
return parsed
|
||||
|
||||
return fallback_result("LLM evaluator returned invalid JSON after json_fix.")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Run automatic LLM evaluations via Codex CLI.")
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-integration-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument(
|
||||
"--checks",
|
||||
default=str(Path(__file__).resolve().parents[1] / "references" / "checks.json"),
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
repo_root = Path.cwd()
|
||||
|
||||
prompt_path = run_dir / "prompt.json"
|
||||
prompt_config = load_json(prompt_path) if prompt_path.exists() else {}
|
||||
raw_llm_config = prompt_config.get("llm_eval")
|
||||
llm_config: dict[str, Any] = raw_llm_config if isinstance(raw_llm_config, dict) else {}
|
||||
|
||||
checks_data = load_json(Path(args.checks))
|
||||
prompts_dir = Path(__file__).resolve().parents[1] / "references"
|
||||
|
||||
results = {}
|
||||
for check in sort_checks_by_priority(checks_data.get("checks", [])):
|
||||
if not check.get("enabled_by_default"):
|
||||
continue
|
||||
if check.get("type") != "LLM":
|
||||
continue
|
||||
check_id = check.get("id")
|
||||
prompt_id = check.get("evaluator_prompt_id")
|
||||
if not prompt_id:
|
||||
continue
|
||||
prompt_file = prompts_dir / f"{prompt_id}.md"
|
||||
if not prompt_file.exists():
|
||||
results[check_id] = fallback_result(f"Evaluator prompt missing: {prompt_file}")
|
||||
continue
|
||||
results[check_id] = run_single_eval(
|
||||
run_dir, repo_root, check_id, prompt_file, llm_config, prompts_dir
|
||||
)
|
||||
|
||||
llm_path = run_dir / "llm_results.json"
|
||||
write_json(llm_path, results)
|
||||
print(str(llm_path))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,277 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import selectors
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, TypedDict
|
||||
|
||||
DENYLIST_PATTERNS = [
|
||||
r"\brm\s+-rf\b",
|
||||
r"\brm\s+-fr\b",
|
||||
r"\brm\s+-r\b",
|
||||
r"\bgit\s+clean\s+-xfd\b",
|
||||
r"\bmkfs\b",
|
||||
r"\bdd\s+if=",
|
||||
r"\bdiskutil\s+erase\b",
|
||||
r"\b:;\s*\b", # basic fork bomb patterns
|
||||
r"\bmkfs\.[a-z0-9]+\b",
|
||||
]
|
||||
|
||||
|
||||
class PlanCommand(TypedDict, total=False):
|
||||
label: str
|
||||
cmd: str
|
||||
timeout_soft_seconds: int
|
||||
timeout_hard_seconds: int
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = load_json(latest_path)
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "evidence.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def is_denylisted(cmd: str) -> bool:
|
||||
lower = cmd.lower()
|
||||
return any(re.search(pattern, lower) for pattern in DENYLIST_PATTERNS)
|
||||
|
||||
|
||||
def normalize_plan(plan_data: dict[str, Any]) -> dict[str, Any]:
|
||||
if "commands" not in plan_data:
|
||||
raise ValueError("Plan JSON must include 'commands'.")
|
||||
commands: list[PlanCommand] = []
|
||||
for entry in plan_data.get("commands", []):
|
||||
if isinstance(entry, str):
|
||||
command_str: PlanCommand = {"label": "step", "cmd": entry}
|
||||
commands.append(command_str)
|
||||
elif isinstance(entry, dict):
|
||||
cmd = entry.get("cmd")
|
||||
if not cmd:
|
||||
raise ValueError("Each command entry must include 'cmd'.")
|
||||
command: PlanCommand = {
|
||||
"label": entry.get("label") or "step",
|
||||
"cmd": cmd,
|
||||
}
|
||||
soft = entry.get("timeout_soft_seconds")
|
||||
hard = entry.get("timeout_hard_seconds")
|
||||
if soft is not None:
|
||||
command["timeout_soft_seconds"] = int(soft)
|
||||
if hard is not None:
|
||||
command["timeout_hard_seconds"] = int(hard)
|
||||
commands.append(command)
|
||||
else:
|
||||
raise ValueError("Commands must be strings or objects with 'cmd'.")
|
||||
plan_data["commands"] = commands
|
||||
return plan_data
|
||||
|
||||
|
||||
def run_command(
|
||||
cmd: str, cwd: Path, env: dict, soft_timeout: int, hard_timeout: int, log_path: Path
|
||||
) -> dict:
|
||||
started_at = now_iso()
|
||||
start_time = time.time()
|
||||
soft_exceeded = False
|
||||
hard_exceeded = False
|
||||
exit_code = None
|
||||
|
||||
with log_path.open("w", encoding="utf-8") as log_file:
|
||||
proc = subprocess.Popen(
|
||||
cmd,
|
||||
shell=True,
|
||||
cwd=str(cwd),
|
||||
env=env,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
text=True,
|
||||
bufsize=1,
|
||||
)
|
||||
selector = selectors.DefaultSelector()
|
||||
if proc.stdout:
|
||||
selector.register(proc.stdout, selectors.EVENT_READ)
|
||||
|
||||
while True:
|
||||
now = time.time()
|
||||
if not soft_exceeded and now - start_time > soft_timeout:
|
||||
soft_exceeded = True
|
||||
if now - start_time > hard_timeout:
|
||||
hard_exceeded = True
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
break
|
||||
events = selector.select(timeout=0.2)
|
||||
for key, _ in events:
|
||||
file_obj = key.fileobj
|
||||
if isinstance(file_obj, io.TextIOBase):
|
||||
line = file_obj.readline()
|
||||
if line:
|
||||
log_file.write(line)
|
||||
if proc.poll() is not None:
|
||||
break
|
||||
|
||||
# Drain remaining output
|
||||
if proc.stdout:
|
||||
for line in proc.stdout:
|
||||
log_file.write(line)
|
||||
|
||||
exit_code = proc.returncode
|
||||
|
||||
ended_at = now_iso()
|
||||
duration = time.time() - start_time
|
||||
|
||||
if hard_exceeded:
|
||||
status = "FAIL"
|
||||
elif exit_code == 0:
|
||||
status = "WARN" if soft_exceeded else "PASS"
|
||||
else:
|
||||
status = "FAIL"
|
||||
|
||||
return {
|
||||
"cmd": cmd,
|
||||
"status": status,
|
||||
"exit_code": exit_code,
|
||||
"duration_seconds": round(duration, 2),
|
||||
"soft_timeout_seconds": soft_timeout,
|
||||
"hard_timeout_seconds": hard_timeout,
|
||||
"soft_timeout_exceeded": soft_exceeded,
|
||||
"hard_timeout_exceeded": hard_exceeded,
|
||||
"log_path": str(log_path),
|
||||
"started_at": started_at,
|
||||
"ended_at": ended_at,
|
||||
}
|
||||
|
||||
|
||||
def sanitize_label(label: str) -> str:
|
||||
cleaned = re.sub(r"[^a-zA-Z0-9_.-]+", "-", label.strip().lower())
|
||||
return cleaned.strip("-") or "step"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Execute a documented dev/build/test plan.")
|
||||
parser.add_argument("--plan", required=True, help="Path to plan JSON")
|
||||
parser.add_argument(
|
||||
"--out-dir",
|
||||
default=".codex-readiness-integration-test",
|
||||
help="Base output directory",
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument(
|
||||
"--soft-timeout-seconds", type=int, default=600, help="Soft timeout per command"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--hard-timeout-multiplier", type=int, default=3, help="Hard timeout multiplier"
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
plan_path = Path(args.plan)
|
||||
if not plan_path.exists():
|
||||
raise SystemExit(f"Plan file not found: {plan_path}")
|
||||
|
||||
plan_data = normalize_plan(load_json(plan_path))
|
||||
|
||||
cwd = Path(plan_data.get("cwd") or plan_data.get("project_dir") or Path.cwd())
|
||||
if not cwd.is_absolute():
|
||||
cwd = (Path.cwd() / cwd).resolve()
|
||||
|
||||
env = os.environ.copy()
|
||||
env.update(plan_data.get("env", {}))
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
base_dir.mkdir(parents=True, exist_ok=True)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
logs_dir = run_dir / "logs"
|
||||
logs_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
steps = []
|
||||
for index, entry in enumerate(plan_data.get("commands", []), start=1):
|
||||
label = entry.get("label") or f"step-{index}"
|
||||
cmd = entry.get("cmd", "").strip()
|
||||
soft_timeout = entry.get("timeout_soft_seconds") or args.soft_timeout_seconds
|
||||
hard_timeout = entry.get("timeout_hard_seconds") or (
|
||||
soft_timeout * args.hard_timeout_multiplier
|
||||
)
|
||||
|
||||
log_path = logs_dir / f"{index:02d}-{sanitize_label(label)}.log"
|
||||
if is_denylisted(cmd):
|
||||
steps.append(
|
||||
{
|
||||
"label": label,
|
||||
"cmd": cmd,
|
||||
"status": "FAIL",
|
||||
"exit_code": None,
|
||||
"duration_seconds": 0,
|
||||
"soft_timeout_seconds": soft_timeout,
|
||||
"hard_timeout_seconds": hard_timeout,
|
||||
"soft_timeout_exceeded": False,
|
||||
"hard_timeout_exceeded": False,
|
||||
"denylisted": True,
|
||||
"log_path": str(log_path),
|
||||
"started_at": now_iso(),
|
||||
"ended_at": now_iso(),
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
result = run_command(cmd, cwd, env, soft_timeout, hard_timeout, log_path)
|
||||
result["label"] = label
|
||||
result["denylisted"] = False
|
||||
steps.append(result)
|
||||
|
||||
overall_status = "PASS"
|
||||
for step in steps:
|
||||
if step["status"] == "FAIL":
|
||||
overall_status = "FAIL"
|
||||
break
|
||||
if step["status"] == "WARN":
|
||||
overall_status = "WARN"
|
||||
|
||||
summary = {
|
||||
"plan_path": str(plan_path),
|
||||
"project_dir": str(plan_data.get("project_dir", cwd)),
|
||||
"cwd": str(cwd),
|
||||
"soft_timeout_seconds": args.soft_timeout_seconds,
|
||||
"hard_timeout_multiplier": args.hard_timeout_multiplier,
|
||||
"steps": steps,
|
||||
"overall_status": overall_status,
|
||||
"started_at": steps[0]["started_at"] if steps else now_iso(),
|
||||
"ended_at": steps[-1]["ended_at"] if steps else now_iso(),
|
||||
}
|
||||
|
||||
summary_path = run_dir / "execution_summary.json"
|
||||
summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
||||
print(str(summary_path))
|
||||
|
||||
return 0 if overall_status == "PASS" else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,456 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
from decimal import ROUND_HALF_UP, Decimal
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"}
|
||||
PRIORITY_MULTIPLIERS = {0: 4, 1: 3, 2: 2, 3: 1}
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = load_json(latest_path)
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "evidence.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def round_half_up(value: float) -> int:
|
||||
return int(Decimal(value).quantize(Decimal("1"), rounding=ROUND_HALF_UP))
|
||||
|
||||
|
||||
def status_points(status: str) -> float:
|
||||
if status == "PASS":
|
||||
return 1.0
|
||||
if status == "WARN":
|
||||
return 0.5
|
||||
return 0.0
|
||||
|
||||
|
||||
def normalize_priority(value) -> int:
|
||||
try:
|
||||
priority = int(value)
|
||||
except (TypeError, ValueError):
|
||||
priority = 3
|
||||
if priority in PRIORITY_MULTIPLIERS:
|
||||
return priority
|
||||
return 3
|
||||
|
||||
|
||||
def sort_checks_by_priority(checks: list[dict]) -> list[dict]:
|
||||
return sorted(
|
||||
checks,
|
||||
key=lambda check: (
|
||||
normalize_priority(check.get("priority")),
|
||||
check.get("id", ""),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def priority_label(priority: int) -> str:
|
||||
return f"P{priority}"
|
||||
|
||||
|
||||
def validate_result(result: dict) -> dict | None:
|
||||
if not isinstance(result, dict):
|
||||
return None
|
||||
if result.get("status") not in VALID_STATUSES:
|
||||
return None
|
||||
if (
|
||||
"rationale" not in result
|
||||
or "evidence_quotes" not in result
|
||||
or "recommendations" not in result
|
||||
or "confidence" not in result
|
||||
):
|
||||
return None
|
||||
if not isinstance(result.get("evidence_quotes"), list):
|
||||
return None
|
||||
if not isinstance(result.get("recommendations"), list):
|
||||
return None
|
||||
return result
|
||||
|
||||
|
||||
def fallback_invalid_json() -> dict:
|
||||
return {
|
||||
"status": "WARN",
|
||||
"rationale": "Invalid JSON from evaluator after retries.",
|
||||
"evidence_quotes": [],
|
||||
"recommendations": ["Re-run the evaluator with the json_fix prompt."],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
|
||||
|
||||
def build_weights(checks: list[dict]) -> dict:
|
||||
enabled = [c for c in checks if c.get("enabled_by_default")]
|
||||
raw_weights = []
|
||||
for check in enabled:
|
||||
weight = check.get("weight")
|
||||
base_weight = weight if isinstance(weight, (int, float)) else 1.0
|
||||
priority = normalize_priority(check.get("priority"))
|
||||
raw_weights.append(base_weight * PRIORITY_MULTIPLIERS[priority])
|
||||
total = sum(raw_weights) if raw_weights else 1.0
|
||||
weights = {}
|
||||
for check, raw in zip(enabled, raw_weights):
|
||||
weights[check["id"]] = (raw / total) * 100.0
|
||||
return weights
|
||||
|
||||
|
||||
def build_results(
|
||||
checks: list[dict],
|
||||
deterministic_results: dict,
|
||||
llm_results: dict,
|
||||
execution_summary: dict | None,
|
||||
) -> dict:
|
||||
results = {}
|
||||
execution_status = None
|
||||
if execution_summary:
|
||||
execution_status = execution_summary.get("overall_status")
|
||||
|
||||
for check in checks:
|
||||
if not check.get("enabled_by_default"):
|
||||
continue
|
||||
check_id = check["id"]
|
||||
check_type = check.get("type")
|
||||
|
||||
if check_type == "DETERMINISTIC":
|
||||
result = deterministic_results.get(check_id)
|
||||
if result:
|
||||
valid = validate_result(result)
|
||||
results[check_id] = valid if valid else fallback_invalid_json()
|
||||
else:
|
||||
results[check_id] = {
|
||||
"status": "FAIL",
|
||||
"rationale": "Deterministic result missing for this check.",
|
||||
"evidence_quotes": [],
|
||||
"recommendations": ["Run deterministic_rules.py to populate results."],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
continue
|
||||
|
||||
if check_type == "LLM":
|
||||
llm_result = llm_results.get(check_id)
|
||||
if llm_result:
|
||||
valid = validate_result(llm_result)
|
||||
results[check_id] = valid if valid else fallback_invalid_json()
|
||||
else:
|
||||
results[check_id] = {
|
||||
"status": "WARN",
|
||||
"rationale": "LLM evaluation missing for this check.",
|
||||
"evidence_quotes": [],
|
||||
"recommendations": ["Run the evaluator prompt for this check."],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
continue
|
||||
|
||||
if check_type == "HYBRID":
|
||||
status = (
|
||||
execution_status
|
||||
or deterministic_results.get(check_id, {}).get("status")
|
||||
or "NOT_RUN"
|
||||
)
|
||||
llm_result = llm_results.get(check_id)
|
||||
if llm_result:
|
||||
valid = validate_result(llm_result) or fallback_invalid_json()
|
||||
valid["status"] = status
|
||||
results[check_id] = valid
|
||||
else:
|
||||
results[check_id] = {
|
||||
"status": status,
|
||||
"rationale": "Execution summary present but LLM rationale missing."
|
||||
if status != "NOT_RUN"
|
||||
else "Execution not run.",
|
||||
"evidence_quotes": [],
|
||||
"recommendations": ["Provide execution rationale using the evaluator prompt."],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
continue
|
||||
|
||||
return results
|
||||
|
||||
|
||||
def render_html(report: dict, prompt: dict | None = None, report_path: Path | None = None) -> str:
|
||||
score = report["scorecard"]["score_total"]
|
||||
status = report["scorecard"]["overall_status"]
|
||||
results = report.get("results", {})
|
||||
enabled_checks = report.get("enabled_checks", [])
|
||||
checks_by_id = {check["id"]: check for check in enabled_checks}
|
||||
prompt = prompt or {}
|
||||
change_prompt = prompt.get("change_prompt") or ""
|
||||
acceptance_criteria = prompt.get("acceptance_criteria") or []
|
||||
|
||||
def status_class(value: str) -> str:
|
||||
return value.lower()
|
||||
|
||||
html = [
|
||||
"<!doctype html>",
|
||||
"<html>",
|
||||
"<head>",
|
||||
"<meta charset='utf-8'>",
|
||||
"<title>Codex Readiness Integration Test Report</title>",
|
||||
"<style>",
|
||||
"body{font-family:Arial,sans-serif;margin:24px;color:#222;background:#fafafa;}",
|
||||
".badge{display:inline-block;padding:4px 10px;border-radius:12px;color:#fff;font-size:12px;text-transform:uppercase;}",
|
||||
".pass{background:#2e7d32;} .warn{background:#f9a825;} .fail{background:#c62828;} .not_run{background:#546e7a;}",
|
||||
"table{border-collapse:collapse;width:100%;margin:12px 0 20px;background:#fff;table-layout:fixed;}",
|
||||
"th,td{border:1px solid #ddd;padding:8px;font-size:13px;vertical-align:top;word-break:break-word;}",
|
||||
"th{background:#f1f1f1;text-align:left;}",
|
||||
"h3{margin:12px 0 6px;}",
|
||||
"h2{margin-top:24px;}",
|
||||
"</style>",
|
||||
"</head>",
|
||||
"<body>",
|
||||
"<h1>Codex Readiness Integration Test Report</h1>",
|
||||
]
|
||||
if report_path:
|
||||
html.append(f"<h2>{report_path}</h2>")
|
||||
html.append(
|
||||
f"<p>Overall score: <strong>{score}</strong> <span class='badge {status_class(status)}'>{status}</span></p>"
|
||||
)
|
||||
|
||||
if change_prompt:
|
||||
html.extend(
|
||||
[
|
||||
"<h2>Prompt</h2>",
|
||||
"<table>",
|
||||
"<tr><th>Change prompt</th></tr>",
|
||||
f"<tr><td>{change_prompt}</td></tr>",
|
||||
"</table>",
|
||||
]
|
||||
)
|
||||
|
||||
if isinstance(acceptance_criteria, list) and acceptance_criteria:
|
||||
html.extend(
|
||||
[
|
||||
"<h2>Acceptance criteria</h2>",
|
||||
"<table>",
|
||||
"<tr><th>Criteria</th></tr>",
|
||||
]
|
||||
)
|
||||
for item in acceptance_criteria:
|
||||
html.append(f"<tr><td>{item}</td></tr>")
|
||||
html.append("</table>")
|
||||
|
||||
html.append("<h2>Checks</h2>")
|
||||
|
||||
html.append("<table>")
|
||||
html.append("<tr><th>Check</th><th>Status</th><th>Rationale</th></tr>")
|
||||
for check_id, result in results.items():
|
||||
check = checks_by_id.get(check_id, {})
|
||||
title = check.get("title", check_id)
|
||||
status_value = result.get("status", "NOT_RUN")
|
||||
rationale = result.get("rationale", "")
|
||||
html.append(
|
||||
"<tr>"
|
||||
f"<td>{title}</td>"
|
||||
f"<td><span class='badge {status_class(status_value)}'>{status_value}</span></td>"
|
||||
f"<td>{rationale}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
html.append("</table>")
|
||||
html.append("</body></html>")
|
||||
return "\n".join(html)
|
||||
|
||||
|
||||
def summarize_diff(git_diff: str) -> dict[str, Any]:
|
||||
files: list[str] = []
|
||||
additions = 0
|
||||
deletions = 0
|
||||
for line in git_diff.splitlines():
|
||||
if line.startswith("diff --git "):
|
||||
parts = line.split()
|
||||
if len(parts) >= 4:
|
||||
left = parts[2].removeprefix("a/")
|
||||
right = parts[3].removeprefix("b/")
|
||||
if left and left not in files:
|
||||
files.append(left)
|
||||
if right and right not in files:
|
||||
files.append(right)
|
||||
continue
|
||||
if line.startswith(("+++ ", "--- ")):
|
||||
continue
|
||||
if line.startswith("+"):
|
||||
additions += 1
|
||||
elif line.startswith("-"):
|
||||
deletions += 1
|
||||
return {"files": files, "additions": additions, "deletions": deletions}
|
||||
|
||||
|
||||
def truncate_text(text: str, limit: int = 200) -> str:
|
||||
if len(text) <= limit:
|
||||
return text
|
||||
return text[: limit - 3] + "..."
|
||||
|
||||
|
||||
def render_summary_text(
|
||||
report: dict,
|
||||
evidence: dict,
|
||||
deterministic_results: dict,
|
||||
llm_results: dict,
|
||||
execution_summary: dict | None,
|
||||
agentic_summary: dict | None,
|
||||
summary_path: Path,
|
||||
) -> str:
|
||||
lines = [
|
||||
"# Codex Readiness Integration Test Report",
|
||||
f"## {summary_path}",
|
||||
"",
|
||||
]
|
||||
score = report["scorecard"]["score_total"]
|
||||
status = report["scorecard"]["overall_status"]
|
||||
lines.append(f"Overall: {status} (score {score})")
|
||||
|
||||
prompt = (evidence.get("prompt_json") or {}).get("content") or {}
|
||||
change_prompt = prompt.get("change_prompt") or ""
|
||||
if change_prompt:
|
||||
lines.append(f"Prompt: {truncate_text(change_prompt)}")
|
||||
|
||||
if agentic_summary:
|
||||
agentic_status = agentic_summary.get("status", "NOT_RUN")
|
||||
exit_code = agentic_summary.get("exit_code")
|
||||
duration = agentic_summary.get("duration_seconds")
|
||||
lines.append(
|
||||
f"Agentic loop: {agentic_status} (exit_code {exit_code}, duration {duration}s)"
|
||||
)
|
||||
else:
|
||||
lines.append("Agentic loop: NOT_RUN")
|
||||
|
||||
diff_stats = summarize_diff(evidence.get("git_diff", ""))
|
||||
if diff_stats.get("files"):
|
||||
lines.append(
|
||||
f"Diff: {len(diff_stats['files'])} file(s), +{diff_stats['additions']}/-{diff_stats['deletions']} lines"
|
||||
)
|
||||
|
||||
path_check = deterministic_results.get("repo_root_only_changes", {}).get("status")
|
||||
if path_check:
|
||||
lines.append(f"Path policy: {path_check}")
|
||||
|
||||
test_status = execution_summary.get("overall_status") if execution_summary else "NOT_RUN"
|
||||
lines.append(f"Tests: {test_status}")
|
||||
|
||||
agentic_eval = llm_results.get("agentic_loop_eval", {}).get("status", "NOT_RUN")
|
||||
change_eval = llm_results.get("change_quality_eval", {}).get("status", "NOT_RUN")
|
||||
lines.append(f"LLM eval: agentic_loop_eval={agentic_eval}, change_quality_eval={change_eval}")
|
||||
|
||||
if agentic_summary:
|
||||
questions = agentic_summary.get("clarifying_questions") or []
|
||||
if questions:
|
||||
last_question = truncate_text(str(questions[-1]))
|
||||
lines.append(f"Clarifying questions: {len(questions)}")
|
||||
lines.append(f"Last question: {last_question}")
|
||||
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Score integration test results.")
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-integration-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument(
|
||||
"--checks",
|
||||
default=str(Path(__file__).resolve().parents[1] / "references" / "checks.json"),
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
|
||||
checks_data = load_json(Path(args.checks))
|
||||
checks = sort_checks_by_priority(checks_data.get("checks", []))
|
||||
|
||||
deterministic_results = (
|
||||
load_json(run_dir / "deterministic_results.json")
|
||||
if (run_dir / "deterministic_results.json").exists()
|
||||
else {}
|
||||
)
|
||||
llm_results = (
|
||||
load_json(run_dir / "llm_results.json") if (run_dir / "llm_results.json").exists() else {}
|
||||
)
|
||||
execution_summary = (
|
||||
load_json(run_dir / "execution_summary.json")
|
||||
if (run_dir / "execution_summary.json").exists()
|
||||
else None
|
||||
)
|
||||
evidence = load_json(run_dir / "evidence.json") if (run_dir / "evidence.json").exists() else {}
|
||||
agentic_summary = (
|
||||
load_json(run_dir / "agentic_summary.json")
|
||||
if (run_dir / "agentic_summary.json").exists()
|
||||
else None
|
||||
)
|
||||
|
||||
weights = build_weights(checks)
|
||||
results = build_results(checks, deterministic_results, llm_results, execution_summary)
|
||||
|
||||
score_items = []
|
||||
for check_id, result in results.items():
|
||||
weight = weights.get(check_id, 0)
|
||||
score_items.append(weight * status_points(result.get("status", "NOT_RUN")))
|
||||
score_total = round_half_up(sum(score_items))
|
||||
|
||||
overall_status = "PASS"
|
||||
for result in results.values():
|
||||
if result["status"] == "FAIL":
|
||||
overall_status = "FAIL"
|
||||
break
|
||||
if result["status"] in {"WARN", "NOT_RUN"}:
|
||||
overall_status = "WARN"
|
||||
|
||||
report = {
|
||||
"scorecard": {
|
||||
"score_total": score_total,
|
||||
"overall_status": overall_status,
|
||||
"weights": weights,
|
||||
},
|
||||
"enabled_checks": [check for check in checks if check.get("enabled_by_default")],
|
||||
"results": results,
|
||||
}
|
||||
|
||||
report_path = run_dir / "report.json"
|
||||
report_path.write_text(json.dumps(report, indent=2), encoding="utf-8")
|
||||
|
||||
prompt_content = (evidence.get("prompt_json") or {}).get("content") or {}
|
||||
html_path = run_dir / "report.html"
|
||||
html_path.write_text(render_html(report, prompt_content, html_path), encoding="utf-8")
|
||||
|
||||
summary = {
|
||||
"overall_status": overall_status,
|
||||
"score_total": score_total,
|
||||
}
|
||||
summary_path = run_dir / "summary.json"
|
||||
summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
||||
|
||||
summary_text_path = run_dir / "summary.txt"
|
||||
summary_text = render_summary_text(
|
||||
report,
|
||||
evidence,
|
||||
deterministic_results,
|
||||
llm_results,
|
||||
execution_summary,
|
||||
agentic_summary,
|
||||
summary_text_path,
|
||||
)
|
||||
summary_text_path.write_text(summary_text, encoding="utf-8")
|
||||
print(summary_text)
|
||||
|
||||
print(str(report_path))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
-100
@@ -1,100 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def load_rules_module():
|
||||
module_path = Path(__file__).resolve().parents[1] / "scripts" / "deterministic_rules.py"
|
||||
spec = importlib.util.spec_from_file_location("deterministic_rules", module_path)
|
||||
if spec is None or spec.loader is None:
|
||||
raise RuntimeError(f"Unable to load deterministic_rules module from {module_path}")
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
RULES = load_rules_module()
|
||||
|
||||
|
||||
def write_run_dir(run_dir: Path, log_text: str) -> None:
|
||||
(run_dir / "logs").mkdir(parents=True, exist_ok=True)
|
||||
(run_dir / "prompt.json").write_text(json.dumps({"change_prompt": "x"}), encoding="utf-8")
|
||||
(run_dir / "logs" / "agentic.log").write_text(log_text, encoding="utf-8")
|
||||
|
||||
|
||||
def check(run_dir: Path) -> dict:
|
||||
return RULES.check_exec_plan_before_code_changes(
|
||||
run_dir,
|
||||
{"prompt_path": "prompt.json", "agentic_log_path": "logs/agentic.log"},
|
||||
)
|
||||
|
||||
|
||||
def test_planning_signal_before_code_change_passes(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-pass"
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"Plan: inspect code paths",
|
||||
"file update",
|
||||
"M src/app.py",
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "PASS"
|
||||
assert "before code changes" in result["rationale"]
|
||||
|
||||
|
||||
def test_planning_signal_after_code_change_fails(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-fail-ordering"
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"file update",
|
||||
"M src/app.py",
|
||||
"Plan: now I will describe the approach",
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "FAIL"
|
||||
assert "after code changes" in result["rationale"]
|
||||
|
||||
|
||||
def test_plan_file_path_does_not_count_as_planning_signal(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-plan-path"
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"file update",
|
||||
"M docs/exec-plan.md",
|
||||
"file update",
|
||||
"M src/app.py",
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "FAIL"
|
||||
assert "No planning signal detected" in result["rationale"]
|
||||
|
||||
|
||||
def test_no_code_changes_warns(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-warn-no-code"
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"Plan: investigate the issue",
|
||||
"file update",
|
||||
"M docs/notes.md",
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "WARN"
|
||||
assert "ordering not evaluated" in result["rationale"]
|
||||
-120
@@ -1,120 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def load_rules_module():
|
||||
module_path = Path(__file__).resolve().parents[1] / "scripts" / "deterministic_rules.py"
|
||||
spec = importlib.util.spec_from_file_location("deterministic_rules", module_path)
|
||||
if spec is None or spec.loader is None:
|
||||
raise RuntimeError(f"Unable to load deterministic_rules module from {module_path}")
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
RULES = load_rules_module()
|
||||
|
||||
|
||||
def write_run_dir(run_dir: Path, log_text: str, build_test_plan: list[dict] | None = None) -> None:
|
||||
(run_dir / "logs").mkdir(parents=True, exist_ok=True)
|
||||
prompt = {
|
||||
"change_prompt": "x",
|
||||
"build_test_plan": build_test_plan or [],
|
||||
}
|
||||
(run_dir / "prompt.json").write_text(json.dumps(prompt), encoding="utf-8")
|
||||
(run_dir / "logs" / "agentic.log").write_text(log_text, encoding="utf-8")
|
||||
|
||||
|
||||
def check(run_dir: Path) -> dict:
|
||||
return RULES.check_verification_after_code_changes(
|
||||
run_dir,
|
||||
{"prompt_path": "prompt.json", "agentic_log_path": "logs/agentic.log"},
|
||||
)
|
||||
|
||||
|
||||
def test_verification_command_after_code_change_passes(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-pass"
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"file update",
|
||||
"M src/app.py",
|
||||
"$ pytest -q",
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "PASS"
|
||||
assert "after code changes" in result["rationale"]
|
||||
|
||||
|
||||
def test_verification_only_before_code_change_fails(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-fail-order"
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"$ pytest -q",
|
||||
"file update",
|
||||
"M src/app.py",
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "FAIL"
|
||||
assert "No build/test/lint verification command detected" in result["rationale"]
|
||||
|
||||
|
||||
def test_prompt_plan_command_counts_as_verification(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-pass-prompt-command"
|
||||
build_test_plan = [{"label": "verify", "cmd": "make check-all"}]
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"file update",
|
||||
"M src/core.py",
|
||||
"$ make check-all",
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text, build_test_plan=build_test_plan)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "PASS"
|
||||
|
||||
|
||||
def test_codex_exec_lines_do_not_count_as_verification(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-fail-codex-line"
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"file update",
|
||||
"M src/app.py",
|
||||
'$ codex exec -C /repo "please run tests after this change"',
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "FAIL"
|
||||
assert "No build/test/lint verification command detected" in result["rationale"]
|
||||
|
||||
|
||||
def test_no_code_changes_warns(tmp_path: Path) -> None:
|
||||
run_dir = tmp_path / "run-warn-no-code"
|
||||
log_text = "\n".join(
|
||||
[
|
||||
"file update",
|
||||
"M docs/notes.md",
|
||||
"$ pytest -q",
|
||||
]
|
||||
)
|
||||
write_run_dir(run_dir, log_text)
|
||||
|
||||
result = check(run_dir)
|
||||
|
||||
assert result["status"] == "WARN"
|
||||
assert "ordering not evaluated" in result["rationale"]
|
||||
@@ -1,202 +0,0 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -1,129 +0,0 @@
|
||||
---
|
||||
name: codex-readiness-unit-test
|
||||
description: Run the Codex Readiness unit test report. Use when you need deterministic checks plus in-session LLM evals for AGENTS.md/PLANS.md.
|
||||
metadata:
|
||||
short-description: Run Codex Readiness unit test report
|
||||
---
|
||||
|
||||
# LLM Codex Readiness Unit Test
|
||||
|
||||
Instruction-first, in-session "readiness" for evaluating AGENTS/PLANS documentation quality without any external APIs or SDKs. All checks run against the current working directory (cwd), with no monorepo discovery. Each run writes to `.codex-readiness-unit-test/<timestamp>/` and updates `.codex-readiness-unit-test/latest.json`. Keep execution deterministic (filesystem scanning + local command execution only). All LLM evaluation happens in-session and must output strict JSON via the provided references.
|
||||
|
||||
## Quick Start
|
||||
|
||||
1) Collect evidence:
|
||||
- `python skills/codex-readiness-unit-test/bin/collect_evidence.py`
|
||||
2) Run deterministic checks:
|
||||
- `python skills/codex-readiness-unit-test/bin/deterministic_rules.py`
|
||||
3) Run LLM checks using references in `references/` and store `.codex-readiness-unit-test/<timestamp>/llm_results.json`.
|
||||
4) If execute mode is requested, build a plan, get confirmation, run:
|
||||
- `python skills/codex-readiness-unit-test/bin/run_plan.py --plan .codex-readiness-unit-test/<timestamp>/plan.json`
|
||||
5) Generate the report:
|
||||
- `python skills/codex-readiness-unit-test/bin/scoring.py --mode read-only|execute`
|
||||
|
||||
Outputs (per run, under `.codex-readiness-unit-test/<timestamp>/`):
|
||||
- `report.json`
|
||||
- `report.html`
|
||||
- `summary.json`
|
||||
- `logs/*` (execute mode)
|
||||
|
||||
## Runbook
|
||||
|
||||
This skill produces a deterministic evidence file plus an in-session LLM evaluation, then compiles a JSON report and HTML scorecard. It requires no OpenAI API key and makes no external HTTP calls.
|
||||
|
||||
### Minimal Inputs
|
||||
- `mode`: `read-only` or `execute` (required)
|
||||
- `soft_timeout_seconds`: optional (default 600)
|
||||
|
||||
### Modes (Read-only vs Execute)
|
||||
- **Read-only**: Collect evidence, run deterministic rules, and run LLM checks #3–#5. No commands are executed, check #6 is marked `NOT_RUN`, and no execution logs/summary are produced.
|
||||
- **Execute**: Everything in read-only **plus** a confirmed `plan.json` is executed via `run_plan.py`. This enables check #6 and produces execution logs + `execution_summary.json` for scoring.
|
||||
|
||||
Always ask the user which mode to run (read-only vs. execute) before proceeding.
|
||||
|
||||
### Check Types
|
||||
- **Deterministic**: filesystem-only checks (#1 AGENTS.md exists, #2 PLANS.md exists, #3 AGENTS.md <= 300 lines, #4 config.toml exists at repo root, repo .codex/, or user .codex/)
|
||||
- **LLM**: in-session Codex evaluation (#3 project context, #4 commands, #5 loops; commands may live in AGENTS or referenced skills)
|
||||
- **Hybrid**: deterministic execution + LLM rationale (#6 execution)
|
||||
|
||||
Skill references are discovered from AGENTS.md via `$SkillName` or `.codex/skills/<name>` patterns; their `SKILL.md` files are added to evidence for the LLM checks.
|
||||
|
||||
All checks run relative to the current working directory and are defined in `skills/codex-readiness-unit-test/references/checks/checks.json`, weighted equally by default. Each run writes outputs to `.codex-readiness-unit-test/<timestamp>/` and updates `.codex-readiness-unit-test/latest.json`.
|
||||
The helper scripts read `.codex-readiness-unit-test/latest.json` by default to locate the latest run directory.
|
||||
|
||||
### Strict JSON + Retry Loop (Required)
|
||||
For each LLM/HYBRID check:
|
||||
1) Run the specialized prompt expecting **strict JSON**.
|
||||
2) If JSON is invalid or missing keys, run `skills/codex-readiness-unit-test/references/json_fix.md` with the raw output.
|
||||
3) Retry up to **2 additional attempts** (max 3 total).
|
||||
4) If still invalid: mark the check as **WARN** with rationale: "Invalid JSON from evaluator after retries".
|
||||
|
||||
The JSON schema is:
|
||||
```json
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...","quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
```
|
||||
|
||||
### Single Confirmation (Required)
|
||||
Combine the command summary and execute plan into **one** concise confirmation step. Present:
|
||||
- The extracted build/test/dev loop commands (human-readable, labeled).
|
||||
- The planned execute details (cwd, ordered commands, soft timeout policy, env).
|
||||
Ask for a single confirmation to proceed. **Do not** paste raw JSON, full evidence, or the full `plan.json`. If declined, mark execute-required checks as `NOT_RUN`.
|
||||
|
||||
### Required Files
|
||||
- `.codex-readiness-unit-test/<timestamp>/evidence.json` (from `collect_evidence.py`)
|
||||
- `.codex-readiness-unit-test/<timestamp>/deterministic_results.json` (from `deterministic_rules.py`)
|
||||
- `.codex-readiness-unit-test/<timestamp>/llm_results.json` (from in-session references)
|
||||
- `.codex-readiness-unit-test/<timestamp>/execution_summary.json` (execute mode only)
|
||||
- `.codex-readiness-unit-test/<timestamp>/report.json` and `.codex-readiness-unit-test/<timestamp>/report.html` (from `scoring.py`)
|
||||
- `.codex-readiness-unit-test/<timestamp>/summary.json` (structured pass/fail summary from `scoring.py`)
|
||||
- `.codex-readiness-unit-test/latest.json` (stable pointer to the latest run directory)
|
||||
|
||||
### Prompt Mapping
|
||||
- #3 `project_context_specified` → `skills/codex-readiness-unit-test/references/project_context.md`
|
||||
- #4 `build_test_commands_exist` → `skills/codex-readiness-unit-test/references/commands.md`
|
||||
- #5 `dev_build_test_loops_documented` → `skills/codex-readiness-unit-test/references/loop_quality.md`
|
||||
- #6 `dev_build_test_loop_execution` → `skills/codex-readiness-unit-test/references/execution_explanation.md`
|
||||
|
||||
### plan.json schema (execute mode)
|
||||
```json
|
||||
{
|
||||
"project_dir": "relative/or/absolute/path (optional)",
|
||||
"cwd": "optional/absolute/path (defaults to current directory)",
|
||||
"commands": [
|
||||
{"label": "setup", "cmd": "npm install"},
|
||||
{"label": "build", "cmd": "npm run build"},
|
||||
{"label": "test", "cmd": "npm test"}
|
||||
],
|
||||
"env": {
|
||||
"EXAMPLE": "value"
|
||||
}
|
||||
}
|
||||
```
|
||||
Place `plan.json` inside the run directory (e.g., `.codex-readiness-unit-test/<timestamp>/plan.json`).
|
||||
|
||||
### llm_results.json schema
|
||||
```json
|
||||
{
|
||||
"project_context_specified": {"status":"PASS","rationale":"...","evidence_quotes":[],"recommendations":[],"confidence":0.7},
|
||||
"build_test_commands_exist": {"status":"PASS","rationale":"...","evidence_quotes":[],"recommendations":[],"confidence":0.7},
|
||||
"dev_build_test_loops_documented": {"status":"WARN","rationale":"...","evidence_quotes":[],"recommendations":[],"confidence":0.6},
|
||||
"dev_build_test_loop_execution": {"status":"PASS","rationale":"...","evidence_quotes":[],"recommendations":[],"confidence":0.6}
|
||||
}
|
||||
```
|
||||
|
||||
### Scoring Rules
|
||||
- PASS = 100% of weight
|
||||
- WARN = 50% of weight
|
||||
- FAIL/NOT_RUN = 0%
|
||||
- Overall status: FAIL if any FAIL; else WARN if any WARN or NOT_RUN; else PASS.
|
||||
|
||||
### Safety + Timeouts
|
||||
- Denylisted commands are **not executed** and marked FAIL.
|
||||
- Soft timeout defaults to 600s; hard cap defaults to 3x soft timeout.
|
||||
- Execution logs are written to `.codex-readiness-unit-test/<timestamp>/logs/`.
|
||||
@@ -1,119 +0,0 @@
|
||||
{
|
||||
"schema_version": "1.0",
|
||||
"checks": [
|
||||
{
|
||||
"id": "agents_md_exists",
|
||||
"title": "AGENTS.md exists (cwd)",
|
||||
"description": "Verify AGENTS.md exists in the current working directory.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "cwd",
|
||||
"execute_required": false,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "agents_exists",
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "plans_md_exists",
|
||||
"title": "Referenced planning markdown exists",
|
||||
"description": "Verify AGENTS.md references a planning markdown file (plan-named or with planning headings) and the referenced file exists.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "cwd",
|
||||
"execute_required": false,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "plans_reference_exists",
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "agents_md_under_300_lines",
|
||||
"title": "AGENTS.md is under 300 lines",
|
||||
"description": "AGENTS.md must be 300 lines or fewer.",
|
||||
"priority": 1,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "cwd",
|
||||
"execute_required": false,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "agents_line_count_under_300",
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "config_toml_exists",
|
||||
"title": "config.toml exists",
|
||||
"description": "config.toml exists at repo root, repo .codex/config.toml, or ~/.codex/config.toml.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "DETERMINISTIC",
|
||||
"scope": "cwd",
|
||||
"execute_required": false,
|
||||
"evaluator_prompt_id": null,
|
||||
"deterministic_rule_id": "config_toml_exists",
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "project_context_specified",
|
||||
"title": "Project directory context is clear",
|
||||
"description": "AGENTS docs include explicit project paths with short context for what each path is used for.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "LLM",
|
||||
"scope": "cwd",
|
||||
"execute_required": false,
|
||||
"evaluator_prompt_id": "project_context",
|
||||
"deterministic_rule_id": null,
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "build_test_commands_exist",
|
||||
"title": "Build/test commands are copy-pastable",
|
||||
"description": "AGENTS docs or referenced skills provide concrete build/test commands with no placeholders.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "LLM",
|
||||
"scope": "cwd",
|
||||
"execute_required": false,
|
||||
"evaluator_prompt_id": "commands",
|
||||
"deterministic_rule_id": null,
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "dev_build_test_loops_documented",
|
||||
"title": "Dev/build/test loop is documented",
|
||||
"description": "AGENTS docs describe ordering, when to run, and success criteria for one or more dev/build/test loops.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "LLM",
|
||||
"scope": "cwd",
|
||||
"execute_required": false,
|
||||
"evaluator_prompt_id": "loop_quality",
|
||||
"deterministic_rule_id": null,
|
||||
"deterministic_rule_params": {},
|
||||
"enabled_by_default": true
|
||||
},
|
||||
{
|
||||
"id": "dev_build_test_loop_execution",
|
||||
"title": "Dev/build/test loop executes",
|
||||
"description": "Execute documented loop (setup/dev/build/test) and validate outcomes.",
|
||||
"priority": 0,
|
||||
"weight": null,
|
||||
"type": "HYBRID",
|
||||
"scope": "cwd",
|
||||
"execute_required": true,
|
||||
"evaluator_prompt_id": "execution_explanation",
|
||||
"deterministic_rule_id": "execution_summary_status",
|
||||
"deterministic_rule_params": {
|
||||
"summary_path": "execution_summary.json"
|
||||
},
|
||||
"enabled_by_default": true
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,29 +0,0 @@
|
||||
Evaluate whether the AGENTS documentation provides concrete, copy-pastable build/test commands for the current working directory.
|
||||
|
||||
Definition of PASS:
|
||||
- Commands are explicit and copy-pastable (no placeholders like "run unit tests").
|
||||
- Commands appear runnable from the project directory unless AGENTS states otherwise.
|
||||
- Multiple commands are acceptable (unit/integration/build).
|
||||
|
||||
Definition of FAIL:
|
||||
- No build/test commands, or only vague prose without actual commands.
|
||||
|
||||
Use evidence snippets in the evidence JSON.
|
||||
|
||||
Return STRICT JSON only with this schema:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...", "quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- Evidence quotes must be exact excerpts from files; keep each quote short (<240 chars).
|
||||
- No patches or diffs in recommendations.
|
||||
- If commands exist but include placeholders or missing context (e.g., need extra args), use WARN.
|
||||
- If no AGENTS content is available, use FAIL.
|
||||
|
||||
Evidence:
|
||||
{{EVIDENCE_JSON}}
|
||||
@@ -1,19 +0,0 @@
|
||||
You are evaluating repository onboarding quality based on provided evidence.
|
||||
|
||||
Return STRICT JSON only with this schema:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...", "quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- Use only evidence provided (AGENTS/PLANS snippets and summaries).
|
||||
- Evidence quotes must be exact excerpts from files; keep each quote short (<240 chars).
|
||||
- No patches or diffs in recommendations.
|
||||
- If evidence is missing, use WARN or FAIL and explain why.
|
||||
|
||||
Evidence:
|
||||
{{EVIDENCE_JSON}}
|
||||
@@ -1,28 +0,0 @@
|
||||
You are summarizing execution results for a documented dev/build/test loop.
|
||||
|
||||
The deterministic runner has already decided the status below. You MUST copy the provided status verbatim.
|
||||
|
||||
Deterministic status:
|
||||
{{DETERMINISTIC_STATUS}}
|
||||
|
||||
Execution summary JSON:
|
||||
{{EXECUTION_SUMMARY_JSON}}
|
||||
|
||||
Return STRICT JSON only with this schema:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...", "quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- The status must equal the deterministic status shown above.
|
||||
- Evidence quotes should come from the execution summary file path, not from AGENTS.
|
||||
- Rationale must mention the executed command(s) from the execution summary (wrap them in backticks).
|
||||
- Keep quotes short (<240 chars).
|
||||
- No patches or diffs in recommendations.
|
||||
|
||||
Execution summary path:
|
||||
{{EXECUTION_SUMMARY_PATH}}
|
||||
@@ -1,18 +0,0 @@
|
||||
You are fixing invalid JSON from a prior evaluator.
|
||||
|
||||
Return ONLY valid JSON that matches this schema exactly:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...", "quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- Do not include any extra keys.
|
||||
- Do not include markdown, commentary, or code fences.
|
||||
- If the original content lacks evidence, keep evidence_quotes empty.
|
||||
|
||||
Invalid output to fix:
|
||||
{{RAW_OUTPUT}}
|
||||
@@ -1,30 +0,0 @@
|
||||
Evaluate whether the AGENTS documentation describes the dev/build/test loop(s) for the current working directory.
|
||||
|
||||
Definition of PASS:
|
||||
- Documentation includes ordering (what to run first, next, last).
|
||||
- It specifies when to run the loop (e.g., after each change, before PR).
|
||||
- It defines success criteria (what output indicates success).
|
||||
- Multiple loops are acceptable (fast vs full).
|
||||
|
||||
Definition of FAIL:
|
||||
- No loop guidance, or only vague statements with no ordering/criteria.
|
||||
|
||||
Use evidence snippets in the evidence JSON.
|
||||
|
||||
Return STRICT JSON only with this schema:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...", "quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- Evidence quotes must be exact excerpts from files; keep each quote short (<240 chars).
|
||||
- No patches or diffs in recommendations.
|
||||
- If ordering exists but missing when-to-run or success criteria, use WARN.
|
||||
- If no AGENTS content is available, use FAIL.
|
||||
|
||||
Evidence:
|
||||
{{EVIDENCE_JSON}}
|
||||
@@ -1,28 +0,0 @@
|
||||
Evaluate whether the AGENTS documentation in the current working directory provides clear context with explicit paths.
|
||||
|
||||
Definition of PASS:
|
||||
- The AGENTS docs include explicit paths to important directories/files (e.g., services/auth, src/, Makefile, pyproject.toml).
|
||||
- Each path has a short context for what it is used for.
|
||||
|
||||
Definition of FAIL:
|
||||
- No explicit paths, or only vague prose without concrete paths.
|
||||
|
||||
Use the evidence snippets in the evidence JSON.
|
||||
|
||||
Return STRICT JSON only with this schema:
|
||||
{
|
||||
"status": "PASS|WARN|FAIL|NOT_RUN",
|
||||
"rationale": "string",
|
||||
"evidence_quotes": [{"path":"...", "quote":"..."}],
|
||||
"recommendations": ["..."],
|
||||
"confidence": 0.0
|
||||
}
|
||||
|
||||
Rules:
|
||||
- Evidence quotes must be exact excerpts from files; keep each quote short (<240 chars).
|
||||
- No patches or diffs in recommendations.
|
||||
- If the docs contain some paths but lack context, use WARN.
|
||||
- If no AGENTS content is available, use FAIL.
|
||||
|
||||
Evidence:
|
||||
{{EVIDENCE_JSON}}
|
||||
@@ -1,324 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, TypedDict, cast
|
||||
|
||||
SKIP_DIRS = {
|
||||
".git",
|
||||
".codex-readiness-unit-test",
|
||||
"node_modules",
|
||||
"dist",
|
||||
"build",
|
||||
".venv",
|
||||
"venv",
|
||||
"__pycache__",
|
||||
}
|
||||
|
||||
BUILD_SIGNAL_FILES = [
|
||||
"package.json",
|
||||
"pnpm-workspace.yaml",
|
||||
"yarn.lock",
|
||||
"pnpm-lock.yaml",
|
||||
"package-lock.json",
|
||||
"pyproject.toml",
|
||||
"setup.py",
|
||||
"requirements.txt",
|
||||
"Pipfile",
|
||||
"poetry.lock",
|
||||
"Makefile",
|
||||
"CMakeLists.txt",
|
||||
"go.mod",
|
||||
"Cargo.toml",
|
||||
"pom.xml",
|
||||
"build.gradle",
|
||||
"build.gradle.kts",
|
||||
"Gemfile",
|
||||
"composer.json",
|
||||
"mix.exs",
|
||||
"gradlew",
|
||||
"tox.ini",
|
||||
"pytest.ini",
|
||||
"jest.config.js",
|
||||
"vitest.config.ts",
|
||||
]
|
||||
|
||||
COMMAND_KEYWORDS = [
|
||||
"npm ",
|
||||
"yarn ",
|
||||
"pnpm ",
|
||||
"make ",
|
||||
"pytest",
|
||||
"go test",
|
||||
"go build",
|
||||
"cargo ",
|
||||
"mvn ",
|
||||
"gradle ",
|
||||
"./gradlew",
|
||||
"bundle ",
|
||||
"rake ",
|
||||
"tox",
|
||||
"poetry ",
|
||||
"pip ",
|
||||
"pipenv ",
|
||||
"cmake ",
|
||||
]
|
||||
|
||||
SKILL_REF_PATTERN = re.compile(r"\$([A-Za-z0-9_.-]+)")
|
||||
SKILL_PATH_PATTERN = re.compile(
|
||||
r"(?:\.codex/skills|~/.codex/skills|/\.codex/skills|skills)/([A-Za-z0-9_.-]+)"
|
||||
)
|
||||
|
||||
|
||||
class SkillReference(TypedDict):
|
||||
name: str
|
||||
|
||||
|
||||
class SkillResolved(TypedDict):
|
||||
name: str
|
||||
path: str
|
||||
|
||||
|
||||
class SkillsInfo(TypedDict):
|
||||
roots: list[str]
|
||||
referenced: list[str]
|
||||
resolved: list[SkillResolved]
|
||||
missing: list[SkillReference]
|
||||
|
||||
|
||||
def read_text(path: Path) -> str:
|
||||
try:
|
||||
return path.read_text(encoding="utf-8")
|
||||
except Exception:
|
||||
try:
|
||||
return path.read_text(encoding="utf-8", errors="ignore")
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def extract_snippet(text: str, max_chars: int) -> str:
|
||||
if len(text) <= max_chars:
|
||||
return text
|
||||
return text[: max_chars - 3] + "..."
|
||||
|
||||
|
||||
def extract_candidate_commands(text: str) -> list[str]:
|
||||
commands = []
|
||||
in_code_block = False
|
||||
for line in text.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("```"):
|
||||
in_code_block = not in_code_block
|
||||
continue
|
||||
if in_code_block:
|
||||
if stripped and not stripped.startswith("#"):
|
||||
commands.append(stripped)
|
||||
continue
|
||||
inline = re.findall(r"`([^`]+)`", line)
|
||||
for cmd in inline:
|
||||
cmd_str = cmd.strip()
|
||||
if cmd_str:
|
||||
commands.append(cmd_str)
|
||||
if stripped.startswith("$"):
|
||||
cmd = stripped.lstrip("$ ")
|
||||
if cmd:
|
||||
commands.append(cmd)
|
||||
if any(keyword in stripped for keyword in COMMAND_KEYWORDS):
|
||||
commands.append(stripped)
|
||||
# Normalize and dedupe
|
||||
normalized = []
|
||||
seen = set()
|
||||
for cmd in commands:
|
||||
cleaned = cmd.strip()
|
||||
if not cleaned:
|
||||
continue
|
||||
if cleaned in seen:
|
||||
continue
|
||||
seen.add(cleaned)
|
||||
normalized.append(cleaned)
|
||||
return normalized
|
||||
|
||||
|
||||
def extract_skill_refs(text: str) -> list[str]:
|
||||
refs = set(SKILL_REF_PATTERN.findall(text))
|
||||
refs.update(SKILL_PATH_PATTERN.findall(text))
|
||||
return sorted(refs)
|
||||
|
||||
|
||||
def resolve_skills_roots(repo_root: Path) -> list[Path]:
|
||||
candidates = []
|
||||
codex_home = os.environ.get("CODEX_HOME")
|
||||
if codex_home:
|
||||
candidates.append(Path(codex_home) / "skills")
|
||||
candidates.append(repo_root / ".codex" / "skills")
|
||||
candidates.append(Path.home() / ".codex" / "skills")
|
||||
|
||||
roots = []
|
||||
seen = set()
|
||||
for candidate in candidates:
|
||||
try:
|
||||
resolved = candidate.expanduser().resolve()
|
||||
except Exception:
|
||||
resolved = candidate.expanduser()
|
||||
key = str(resolved)
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
if resolved.exists():
|
||||
roots.append(resolved)
|
||||
return roots
|
||||
|
||||
|
||||
def find_repo_signals(repo_root: Path, max_results: int = 200) -> list[dict]:
|
||||
results = []
|
||||
for dirpath, dirnames, filenames in os.walk(repo_root):
|
||||
dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS]
|
||||
for filename in filenames:
|
||||
if filename in BUILD_SIGNAL_FILES:
|
||||
path = Path(dirpath) / filename
|
||||
results.append(
|
||||
{
|
||||
"path": str(path),
|
||||
"type": "build_signal",
|
||||
}
|
||||
)
|
||||
if len(results) >= max_results:
|
||||
return results
|
||||
return results
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Collect deterministic evidence for codex-readiness-unit-test."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-unit-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--max-snippet-chars", type=int, default=2000, help="Max chars per snippet")
|
||||
args = parser.parse_args()
|
||||
|
||||
cwd = Path.cwd()
|
||||
repo_root = cwd
|
||||
agents_path = cwd / "AGENTS.md"
|
||||
plans_path = cwd / "PLANS.md"
|
||||
|
||||
snippets: list[dict[str, str]] = []
|
||||
candidate_commands: list[str] = []
|
||||
skill_refs: list[str] = []
|
||||
skills_info: SkillsInfo = cast(
|
||||
SkillsInfo,
|
||||
{
|
||||
"roots": [],
|
||||
"referenced": [],
|
||||
"resolved": [],
|
||||
"missing": [],
|
||||
},
|
||||
)
|
||||
if agents_path.exists():
|
||||
text = read_text(agents_path)
|
||||
if not text:
|
||||
text = ""
|
||||
snippets.append(
|
||||
{
|
||||
"path": str(agents_path),
|
||||
"snippet": extract_snippet(text, args.max_snippet_chars),
|
||||
}
|
||||
)
|
||||
if text:
|
||||
candidate_commands.extend(extract_candidate_commands(text))
|
||||
skill_refs = extract_skill_refs(text)
|
||||
|
||||
if skill_refs:
|
||||
skills_info["referenced"] = skill_refs
|
||||
skills_roots = resolve_skills_roots(repo_root)
|
||||
skills_info["roots"] = [str(root) for root in skills_roots]
|
||||
for skill_name in skill_refs:
|
||||
skill_path = None
|
||||
for root in skills_roots:
|
||||
candidate = root / skill_name / "SKILL.md"
|
||||
if candidate.exists():
|
||||
skill_path = candidate
|
||||
break
|
||||
if skill_path is None:
|
||||
skills_info["missing"].append({"name": skill_name})
|
||||
continue
|
||||
skills_info["resolved"].append(
|
||||
{
|
||||
"name": skill_name,
|
||||
"path": str(skill_path),
|
||||
}
|
||||
)
|
||||
skill_text = read_text(skill_path)
|
||||
snippets.append(
|
||||
{
|
||||
"path": str(skill_path),
|
||||
"snippet": extract_snippet(skill_text, args.max_snippet_chars),
|
||||
}
|
||||
)
|
||||
if skill_text:
|
||||
candidate_commands.extend(extract_candidate_commands(skill_text))
|
||||
|
||||
# De-duplicate candidate commands
|
||||
seen_cmds = set()
|
||||
unique_cmds = []
|
||||
for cmd in candidate_commands:
|
||||
if cmd in seen_cmds:
|
||||
continue
|
||||
seen_cmds.add(cmd)
|
||||
unique_cmds.append(cmd)
|
||||
|
||||
out_dir = Path(args.out_dir)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
run_time = datetime.now(timezone.utc)
|
||||
run_id = run_time.strftime("%Y-%m-%dT%H-%M-%SZ")
|
||||
run_dir = (out_dir / run_id).resolve()
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
evidence: dict[str, Any] = cast(
|
||||
dict[str, Any],
|
||||
{
|
||||
"run_context": {
|
||||
"cwd": str(cwd),
|
||||
"repo_root": str(repo_root),
|
||||
"run_dir": str(run_dir),
|
||||
"run_id": run_id,
|
||||
"timestamp": run_time.isoformat(),
|
||||
},
|
||||
"agents_md": {
|
||||
"exists": agents_path.exists(),
|
||||
"path": str(agents_path) if agents_path.exists() else None,
|
||||
},
|
||||
"plans_md": {
|
||||
"exists": plans_path.exists(),
|
||||
"path": str(plans_path) if plans_path.exists() else None,
|
||||
},
|
||||
"skills": skills_info,
|
||||
"snippets": snippets,
|
||||
"inferred": {
|
||||
"candidate_commands": unique_cmds[:50],
|
||||
"repo_signals": find_repo_signals(repo_root),
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
evidence_path = run_dir / "evidence.json"
|
||||
evidence_path.write_text(json.dumps(evidence, indent=2), encoding="utf-8")
|
||||
|
||||
latest_path = out_dir / "latest.json"
|
||||
latest_payload = {
|
||||
"run_dir": str(run_dir),
|
||||
"run_id": run_id,
|
||||
"timestamp": evidence["run_context"]["timestamp"],
|
||||
}
|
||||
latest_path.write_text(json.dumps(latest_payload, indent=2), encoding="utf-8")
|
||||
|
||||
print(str(evidence_path))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,396 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = load_json(latest_path)
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "evidence.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def result(
|
||||
status: str,
|
||||
rationale: str,
|
||||
evidence_path: str | None = None,
|
||||
quote: str | None = None,
|
||||
recommendations=None,
|
||||
confidence: float = 1.0,
|
||||
) -> dict:
|
||||
if recommendations is None:
|
||||
recommendations = []
|
||||
evidence_quotes = []
|
||||
if evidence_path and quote:
|
||||
evidence_quotes.append({"path": evidence_path, "quote": quote})
|
||||
return {
|
||||
"status": status,
|
||||
"rationale": rationale,
|
||||
"evidence_quotes": evidence_quotes,
|
||||
"recommendations": recommendations,
|
||||
"confidence": confidence,
|
||||
}
|
||||
|
||||
|
||||
def read_text(path: Path) -> str:
|
||||
try:
|
||||
return path.read_text(encoding="utf-8")
|
||||
except Exception:
|
||||
try:
|
||||
return path.read_text(encoding="utf-8", errors="ignore")
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def rule_agents_exists(evidence: dict) -> dict:
|
||||
agents = evidence.get("agents_md", {})
|
||||
path = agents.get("path")
|
||||
if agents.get("exists") and path and Path(path).exists():
|
||||
return result(
|
||||
"PASS",
|
||||
"AGENTS.md exists in the current directory.",
|
||||
recommendations=[],
|
||||
)
|
||||
return result(
|
||||
"FAIL",
|
||||
"AGENTS.md is missing in the current directory.",
|
||||
recommendations=["Add an AGENTS.md in the current directory with onboarding context."],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
|
||||
PLAN_HEADINGS = [
|
||||
"## Purpose / Big Picture",
|
||||
"## Progress",
|
||||
"## Decision Log",
|
||||
"## Outcomes & Retrospective",
|
||||
"## Surprises & Discoveries",
|
||||
]
|
||||
|
||||
|
||||
def _clean_markdown_ref(raw_ref: str) -> str:
|
||||
if not raw_ref:
|
||||
return ""
|
||||
cleaned = raw_ref.strip()
|
||||
if cleaned.startswith("<") and cleaned.endswith(">"):
|
||||
cleaned = cleaned[1:-1].strip()
|
||||
cleaned = re.split(r"[?#]", cleaned, maxsplit=1)[0]
|
||||
cleaned = cleaned.strip().strip("`'\"()[]{}<>.,:;")
|
||||
return cleaned
|
||||
|
||||
|
||||
def _is_markdown_path(ref: str) -> bool:
|
||||
if not ref:
|
||||
return False
|
||||
try:
|
||||
name = Path(ref).name.lower()
|
||||
except Exception:
|
||||
return False
|
||||
return name.endswith((".md", ".markdown"))
|
||||
|
||||
|
||||
def _extract_markdown_refs(text: str) -> list[tuple[str, str]]:
|
||||
references: list[tuple[str, str]] = []
|
||||
for match in re.finditer(r"\[[^\]]*\]\(([^)]+)\)", text):
|
||||
target = _clean_markdown_ref(match.group(1))
|
||||
if _is_markdown_path(target):
|
||||
start = text.rfind("\n", 0, match.start())
|
||||
end = text.find("\n", match.start())
|
||||
line = text[start + 1 : end if end != -1 else None]
|
||||
references.append((target, line.strip()))
|
||||
for line in text.splitlines():
|
||||
for token in re.findall(r"(?i)[A-Za-z0-9_./\\-]*\.(?:md|markdown)", line):
|
||||
cleaned = _clean_markdown_ref(token)
|
||||
if _is_markdown_path(cleaned):
|
||||
references.append((cleaned, line.strip()))
|
||||
return references
|
||||
|
||||
|
||||
def _is_plan_named(path: Path) -> bool:
|
||||
return "plan" in path.name.lower()
|
||||
|
||||
|
||||
def _has_planning_conventions(text: str) -> bool:
|
||||
lowered = text.lower()
|
||||
matches = sum(1 for heading in PLAN_HEADINGS if heading.lower() in lowered)
|
||||
return matches >= 3
|
||||
|
||||
|
||||
def rule_plans_reference_exists(evidence: dict) -> dict:
|
||||
agents = evidence.get("agents_md", {})
|
||||
agents_path = agents.get("path")
|
||||
if not (agents.get("exists") and agents_path and Path(agents_path).exists()):
|
||||
return result(
|
||||
"FAIL",
|
||||
"AGENTS.md is missing; cannot resolve referenced plans file.",
|
||||
recommendations=["Add an AGENTS.md that references a plans markdown file."],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
text = read_text(Path(agents_path))
|
||||
references = _extract_markdown_refs(text)
|
||||
if not references:
|
||||
return result(
|
||||
"FAIL",
|
||||
"No planning markdown file reference found in AGENTS.md.",
|
||||
recommendations=[
|
||||
"Reference a planning markdown file in AGENTS.md (e.g., PLANS.md or exec-plan.md)."
|
||||
],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
base_dir = Path(agents_path).parent
|
||||
resolved = {}
|
||||
for raw_ref, line in references:
|
||||
cleaned = _clean_markdown_ref(raw_ref)
|
||||
if not _is_markdown_path(cleaned):
|
||||
continue
|
||||
path = Path(cleaned)
|
||||
resolved_path = path if path.is_absolute() else (base_dir / path).resolve()
|
||||
if resolved_path not in resolved:
|
||||
resolved[resolved_path] = line
|
||||
|
||||
missing = [path for path in resolved if not path.exists()]
|
||||
evidence_quotes = []
|
||||
for line in resolved.values():
|
||||
if line:
|
||||
evidence_quotes.append({"path": agents_path, "quote": line[:240]})
|
||||
if len(evidence_quotes) >= 3:
|
||||
break
|
||||
|
||||
if missing:
|
||||
missing_list = ", ".join(str(path) for path in missing)
|
||||
return {
|
||||
"status": "FAIL",
|
||||
"rationale": f"Referenced planning markdown file(s) not found: {missing_list}.",
|
||||
"evidence_quotes": evidence_quotes,
|
||||
"recommendations": ["Create the referenced planning markdown file(s)."],
|
||||
"confidence": 1.0,
|
||||
}
|
||||
|
||||
qualifying_paths = []
|
||||
non_qualifying_paths = []
|
||||
for path in resolved:
|
||||
if _is_plan_named(path):
|
||||
qualifying_paths.append(path)
|
||||
continue
|
||||
try:
|
||||
content = read_text(path)
|
||||
except Exception:
|
||||
non_qualifying_paths.append(path)
|
||||
continue
|
||||
if _has_planning_conventions(content):
|
||||
qualifying_paths.append(path)
|
||||
else:
|
||||
non_qualifying_paths.append(path)
|
||||
|
||||
if qualifying_paths:
|
||||
return {
|
||||
"status": "PASS",
|
||||
"rationale": "Referenced planning markdown file(s) exist and follow planning conventions.",
|
||||
"evidence_quotes": evidence_quotes,
|
||||
"recommendations": [],
|
||||
"confidence": 1.0,
|
||||
}
|
||||
|
||||
missing_list = ", ".join(str(path) for path in non_qualifying_paths)
|
||||
return {
|
||||
"status": "FAIL",
|
||||
"rationale": f"Referenced markdown file(s) do not appear to be planning docs: {missing_list}.",
|
||||
"evidence_quotes": evidence_quotes,
|
||||
"recommendations": [
|
||||
"Reference a planning markdown file (name contains 'plan') or add planning headings."
|
||||
],
|
||||
"confidence": 1.0,
|
||||
}
|
||||
|
||||
|
||||
def rule_agents_line_count_under_300(evidence: dict) -> dict:
|
||||
agents = evidence.get("agents_md", {})
|
||||
path = agents.get("path")
|
||||
if not (agents.get("exists") and path and Path(path).exists()):
|
||||
return result(
|
||||
"FAIL",
|
||||
"AGENTS.md is missing; cannot verify line count.",
|
||||
recommendations=["Add an AGENTS.md in the current directory with onboarding context."],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
text = read_text(Path(path))
|
||||
line_count = len(text.splitlines())
|
||||
if line_count <= 300:
|
||||
return result(
|
||||
"PASS",
|
||||
f"AGENTS.md has {line_count} lines (<= 300).",
|
||||
recommendations=[],
|
||||
confidence=1.0,
|
||||
)
|
||||
return result(
|
||||
"FAIL",
|
||||
f"AGENTS.md has {line_count} lines (> 300).",
|
||||
recommendations=["Trim AGENTS.md to 300 lines or fewer."],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
|
||||
def rule_config_toml_exists(evidence: dict) -> dict:
|
||||
run_context = evidence.get("run_context", {})
|
||||
cwd = run_context.get("cwd")
|
||||
if not cwd:
|
||||
return result(
|
||||
"FAIL",
|
||||
"Run context missing; cannot resolve config.toml location.",
|
||||
recommendations=["Ensure evidence.json includes run_context.cwd."],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
repo_root = Path(cwd)
|
||||
repo_config = repo_root / "config.toml"
|
||||
codex_config = repo_root / ".codex" / "config.toml"
|
||||
user_codex_config = Path.home() / ".codex" / "config.toml"
|
||||
|
||||
found_paths = []
|
||||
if repo_config.exists():
|
||||
found_paths.append(str(repo_config))
|
||||
if codex_config.exists():
|
||||
found_paths.append(str(codex_config))
|
||||
if user_codex_config.exists():
|
||||
found_paths.append(str(user_codex_config))
|
||||
|
||||
if found_paths:
|
||||
return result(
|
||||
"PASS",
|
||||
f"config.toml found at: {', '.join(found_paths)}.",
|
||||
recommendations=[],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
return result(
|
||||
"FAIL",
|
||||
"config.toml not found in repo root, repo .codex/, or user .codex/.",
|
||||
recommendations=[
|
||||
"Add config.toml at the repo root, under .codex/config.toml, or in ~/.codex/config.toml."
|
||||
],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
|
||||
def rule_execution_summary_status(params: dict) -> dict:
|
||||
summary_path = Path(params.get("summary_path", "execution_summary.json"))
|
||||
if not summary_path.exists():
|
||||
return result(
|
||||
"NOT_RUN",
|
||||
"Execution summary not found; execution was not run.",
|
||||
recommendations=["Run execute mode to validate the documented dev/build/test loop."],
|
||||
confidence=1.0,
|
||||
)
|
||||
try:
|
||||
summary = load_json(summary_path)
|
||||
except Exception:
|
||||
return result(
|
||||
"WARN",
|
||||
"Execution summary exists but could not be parsed.",
|
||||
recommendations=["Review the execution summary JSON for corruption."],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
status = summary.get("overall_status", "WARN")
|
||||
rationale = f"Execution summary reports overall status: {status}."
|
||||
return result(
|
||||
status,
|
||||
rationale,
|
||||
recommendations=[],
|
||||
confidence=1.0,
|
||||
)
|
||||
|
||||
|
||||
RULES = {
|
||||
"agents_exists": rule_agents_exists,
|
||||
"plans_reference_exists": rule_plans_reference_exists,
|
||||
"agents_line_count_under_300": rule_agents_line_count_under_300,
|
||||
"config_toml_exists": rule_config_toml_exists,
|
||||
"execution_summary_status": rule_execution_summary_status,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run deterministic codex-readiness-unit-test rules."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-unit-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument("--evidence", default=None, help="Path to evidence.json (optional)")
|
||||
parser.add_argument(
|
||||
"--checks",
|
||||
default=str(
|
||||
Path(__file__).resolve().parents[1]
|
||||
/ "references"
|
||||
/ "checks"
|
||||
/ "checks.json"
|
||||
),
|
||||
help="Path to checks.json",
|
||||
)
|
||||
parser.add_argument("--out", default=None, help="Output path (optional)")
|
||||
args = parser.parse_args()
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
evidence_path = Path(args.evidence) if args.evidence else (run_dir / "evidence.json")
|
||||
checks_path = Path(args.checks)
|
||||
if not evidence_path.exists():
|
||||
raise SystemExit(f"Evidence file not found: {evidence_path}")
|
||||
if not checks_path.exists():
|
||||
raise SystemExit(f"Checks file not found: {checks_path}")
|
||||
|
||||
evidence = load_json(evidence_path)
|
||||
checks = load_json(checks_path)
|
||||
|
||||
results = {}
|
||||
for check in checks.get("checks", []):
|
||||
if not check.get("enabled_by_default", False):
|
||||
continue
|
||||
rule_id = check.get("deterministic_rule_id")
|
||||
if not rule_id:
|
||||
continue
|
||||
rule = RULES.get(rule_id)
|
||||
if not rule:
|
||||
continue
|
||||
params = check.get("deterministic_rule_params", {})
|
||||
if rule_id == "execution_summary_status":
|
||||
summary_path = params.get("summary_path", "execution_summary.json")
|
||||
summary_candidate = Path(summary_path)
|
||||
if not summary_candidate.is_absolute():
|
||||
summary_candidate = run_dir / summary_candidate
|
||||
params = {**params, "summary_path": str(summary_candidate)}
|
||||
results[check["id"]] = rule(params)
|
||||
else:
|
||||
results[check["id"]] = rule(evidence)
|
||||
|
||||
out_path = Path(args.out) if args.out else (run_dir / "deterministic_results.json")
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_text(json.dumps({"results": results}, indent=2), encoding="utf-8")
|
||||
print(str(out_path))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,344 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import selectors
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any, TypedDict
|
||||
|
||||
DENYLIST_PATTERNS = [
|
||||
r"\brm\s+-rf\b",
|
||||
r"\brm\s+-fr\b",
|
||||
r"\brm\s+-r\b",
|
||||
r"\bgit\s+clean\s+-xfd\b",
|
||||
r"\bmkfs\b",
|
||||
r"\bdd\s+if=",
|
||||
r"\bdiskutil\s+erase\b",
|
||||
r"\b:;\s*\b", # basic fork bomb patterns
|
||||
r"\bmkfs\.[a-z0-9]+\b",
|
||||
]
|
||||
TEST_KEYWORDS = [
|
||||
" test",
|
||||
"pytest",
|
||||
"node --test",
|
||||
"go test",
|
||||
"cargo test",
|
||||
"mvn test",
|
||||
"gradle test",
|
||||
"./gradlew test",
|
||||
]
|
||||
BUILD_KEYWORDS = [
|
||||
" build",
|
||||
"compile",
|
||||
"mvn package",
|
||||
"gradle build",
|
||||
"./gradlew build",
|
||||
"go build",
|
||||
"cargo build",
|
||||
]
|
||||
|
||||
|
||||
class PlanCommand(TypedDict, total=False):
|
||||
label: str
|
||||
cmd: str
|
||||
timeout_soft_seconds: int
|
||||
timeout_hard_seconds: int
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = load_json(latest_path)
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "evidence.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def is_denylisted(cmd: str) -> bool:
|
||||
lower = cmd.lower()
|
||||
return any(re.search(pattern, lower) for pattern in DENYLIST_PATTERNS)
|
||||
|
||||
|
||||
def normalize_plan(plan_data: dict[str, Any]) -> dict[str, Any]:
|
||||
if "commands" not in plan_data:
|
||||
plan_data["commands"] = []
|
||||
commands: list[PlanCommand] = []
|
||||
for entry in plan_data.get("commands", []):
|
||||
if isinstance(entry, str):
|
||||
command_str: PlanCommand = {"label": "step", "cmd": entry}
|
||||
commands.append(command_str)
|
||||
elif isinstance(entry, dict):
|
||||
cmd = entry.get("cmd")
|
||||
if not cmd:
|
||||
raise ValueError("Each command entry must include 'cmd'.")
|
||||
command: PlanCommand = {
|
||||
"label": entry.get("label") or "step",
|
||||
"cmd": cmd,
|
||||
}
|
||||
soft = entry.get("timeout_soft_seconds")
|
||||
hard = entry.get("timeout_hard_seconds")
|
||||
if soft is not None:
|
||||
command["timeout_soft_seconds"] = int(soft)
|
||||
if hard is not None:
|
||||
command["timeout_hard_seconds"] = int(hard)
|
||||
commands.append(command)
|
||||
else:
|
||||
raise ValueError("Commands must be strings or objects with 'cmd'.")
|
||||
plan_data["commands"] = commands
|
||||
return plan_data
|
||||
|
||||
|
||||
def classify_command(cmd: str) -> str | None:
|
||||
lower = cmd.lower()
|
||||
if any(keyword in lower for keyword in TEST_KEYWORDS):
|
||||
return "test"
|
||||
if any(keyword in lower for keyword in BUILD_KEYWORDS):
|
||||
return "build"
|
||||
return None
|
||||
|
||||
|
||||
def infer_plan_commands(run_dir: Path) -> list[PlanCommand]:
|
||||
evidence_path = run_dir / "evidence.json"
|
||||
if not evidence_path.exists():
|
||||
return []
|
||||
evidence = load_json(evidence_path)
|
||||
candidates = (
|
||||
evidence.get("inferred", {}).get("candidate_commands") if isinstance(evidence, dict) else []
|
||||
)
|
||||
if not isinstance(candidates, list):
|
||||
return []
|
||||
|
||||
build_cmd = None
|
||||
test_cmd = None
|
||||
for cmd in candidates:
|
||||
if not isinstance(cmd, str):
|
||||
continue
|
||||
label = classify_command(cmd)
|
||||
if label == "build" and build_cmd is None:
|
||||
build_cmd = cmd
|
||||
elif label == "test" and test_cmd is None:
|
||||
test_cmd = cmd
|
||||
if build_cmd and test_cmd:
|
||||
break
|
||||
|
||||
commands: list[PlanCommand] = []
|
||||
if build_cmd:
|
||||
commands.append({"label": "build", "cmd": build_cmd})
|
||||
if test_cmd:
|
||||
commands.append({"label": "test", "cmd": test_cmd})
|
||||
return commands
|
||||
|
||||
|
||||
def run_command(
|
||||
cmd: str, cwd: Path, env: dict, soft_timeout: int, hard_timeout: int, log_path: Path
|
||||
) -> dict:
|
||||
started_at = now_iso()
|
||||
start_time = time.time()
|
||||
soft_exceeded = False
|
||||
hard_exceeded = False
|
||||
exit_code = None
|
||||
|
||||
with log_path.open("w", encoding="utf-8") as log_file:
|
||||
proc = subprocess.Popen(
|
||||
cmd,
|
||||
shell=True,
|
||||
cwd=str(cwd),
|
||||
env=env,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
text=True,
|
||||
bufsize=1,
|
||||
)
|
||||
selector = selectors.DefaultSelector()
|
||||
if proc.stdout:
|
||||
selector.register(proc.stdout, selectors.EVENT_READ)
|
||||
|
||||
while True:
|
||||
now = time.time()
|
||||
if not soft_exceeded and now - start_time > soft_timeout:
|
||||
soft_exceeded = True
|
||||
if now - start_time > hard_timeout:
|
||||
hard_exceeded = True
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
break
|
||||
events = selector.select(timeout=0.2)
|
||||
for key, _ in events:
|
||||
file_obj = key.fileobj
|
||||
if isinstance(file_obj, io.TextIOBase):
|
||||
line = file_obj.readline()
|
||||
if line:
|
||||
log_file.write(line)
|
||||
if proc.poll() is not None:
|
||||
break
|
||||
|
||||
# Drain remaining output
|
||||
if proc.stdout:
|
||||
for line in proc.stdout:
|
||||
log_file.write(line)
|
||||
|
||||
exit_code = proc.returncode
|
||||
|
||||
ended_at = now_iso()
|
||||
duration = time.time() - start_time
|
||||
|
||||
if hard_exceeded:
|
||||
status = "FAIL"
|
||||
elif exit_code == 0:
|
||||
status = "WARN" if soft_exceeded else "PASS"
|
||||
else:
|
||||
status = "FAIL"
|
||||
|
||||
return {
|
||||
"cmd": cmd,
|
||||
"status": status,
|
||||
"exit_code": exit_code,
|
||||
"duration_seconds": round(duration, 2),
|
||||
"soft_timeout_seconds": soft_timeout,
|
||||
"hard_timeout_seconds": hard_timeout,
|
||||
"soft_timeout_exceeded": soft_exceeded,
|
||||
"hard_timeout_exceeded": hard_exceeded,
|
||||
"log_path": str(log_path),
|
||||
"started_at": started_at,
|
||||
"ended_at": ended_at,
|
||||
}
|
||||
|
||||
|
||||
def sanitize_label(label: str) -> str:
|
||||
cleaned = re.sub(r"[^a-zA-Z0-9_.-]+", "-", label.strip().lower())
|
||||
return cleaned.strip("-") or "step"
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Execute a documented dev/build/test plan.")
|
||||
parser.add_argument("--plan", required=True, help="Path to plan JSON")
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-unit-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument(
|
||||
"--soft-timeout-seconds", type=int, default=600, help="Soft timeout per command"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--hard-timeout-multiplier", type=int, default=3, help="Hard timeout multiplier"
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
plan_path = Path(args.plan)
|
||||
if not plan_path.exists():
|
||||
raise SystemExit(f"Plan file not found: {plan_path}")
|
||||
|
||||
plan_data = normalize_plan(load_json(plan_path))
|
||||
|
||||
cwd = Path(plan_data.get("cwd") or plan_data.get("project_dir") or Path.cwd())
|
||||
if not cwd.is_absolute():
|
||||
cwd = (Path.cwd() / cwd).resolve()
|
||||
|
||||
env = os.environ.copy()
|
||||
env.update(plan_data.get("env", {}))
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
base_dir.mkdir(parents=True, exist_ok=True)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
logs_dir = run_dir / "logs"
|
||||
logs_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
if not plan_data.get("commands"):
|
||||
inferred = infer_plan_commands(run_dir)
|
||||
if inferred:
|
||||
plan_data["commands"] = inferred
|
||||
else:
|
||||
raise ValueError(
|
||||
"Plan JSON has no commands and no build/test commands could be inferred from skills."
|
||||
)
|
||||
|
||||
steps = []
|
||||
for index, entry in enumerate(plan_data.get("commands", []), start=1):
|
||||
label = entry.get("label") or f"step-{index}"
|
||||
cmd = entry.get("cmd", "").strip()
|
||||
soft_timeout = entry.get("timeout_soft_seconds") or args.soft_timeout_seconds
|
||||
hard_timeout = entry.get("timeout_hard_seconds") or (
|
||||
soft_timeout * args.hard_timeout_multiplier
|
||||
)
|
||||
|
||||
log_path = logs_dir / f"{index:02d}-{sanitize_label(label)}.log"
|
||||
if is_denylisted(cmd):
|
||||
steps.append(
|
||||
{
|
||||
"label": label,
|
||||
"cmd": cmd,
|
||||
"status": "FAIL",
|
||||
"exit_code": None,
|
||||
"duration_seconds": 0,
|
||||
"soft_timeout_seconds": soft_timeout,
|
||||
"hard_timeout_seconds": hard_timeout,
|
||||
"soft_timeout_exceeded": False,
|
||||
"hard_timeout_exceeded": False,
|
||||
"denylisted": True,
|
||||
"log_path": str(log_path),
|
||||
"started_at": now_iso(),
|
||||
"ended_at": now_iso(),
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
result = run_command(cmd, cwd, env, soft_timeout, hard_timeout, log_path)
|
||||
result["label"] = label
|
||||
result["denylisted"] = False
|
||||
steps.append(result)
|
||||
|
||||
overall_status = "PASS"
|
||||
for step in steps:
|
||||
if step["status"] == "FAIL":
|
||||
overall_status = "FAIL"
|
||||
break
|
||||
if step["status"] == "WARN":
|
||||
overall_status = "WARN"
|
||||
|
||||
summary = {
|
||||
"plan_path": str(plan_path),
|
||||
"project_dir": str(plan_data.get("project_dir", cwd)),
|
||||
"cwd": str(cwd),
|
||||
"soft_timeout_seconds": args.soft_timeout_seconds,
|
||||
"hard_timeout_multiplier": args.hard_timeout_multiplier,
|
||||
"steps": steps,
|
||||
"overall_status": overall_status,
|
||||
"started_at": steps[0]["started_at"] if steps else now_iso(),
|
||||
"ended_at": steps[-1]["ended_at"] if steps else now_iso(),
|
||||
}
|
||||
|
||||
summary_path = run_dir / "execution_summary.json"
|
||||
summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
||||
print(str(summary_path))
|
||||
|
||||
return 0 if overall_status == "PASS" else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -1,452 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
from decimal import ROUND_HALF_UP, Decimal
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"}
|
||||
PRIORITY_MULTIPLIERS = {0: 4, 1: 3, 2: 2, 3: 1}
|
||||
|
||||
|
||||
def load_json(path: Path) -> dict:
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path:
|
||||
if run_dir_arg:
|
||||
return Path(run_dir_arg).resolve()
|
||||
latest_path = base_dir / "latest.json"
|
||||
if latest_path.exists():
|
||||
try:
|
||||
latest = load_json(latest_path)
|
||||
run_dir = latest.get("run_dir")
|
||||
if run_dir:
|
||||
return Path(run_dir)
|
||||
except Exception:
|
||||
pass
|
||||
if (base_dir / "evidence.json").exists():
|
||||
return base_dir.resolve()
|
||||
return base_dir.resolve()
|
||||
|
||||
|
||||
def round_half_up(value: float) -> int:
|
||||
return int(Decimal(value).quantize(Decimal("1"), rounding=ROUND_HALF_UP))
|
||||
|
||||
|
||||
def status_points(status: str) -> float:
|
||||
if status == "PASS":
|
||||
return 1.0
|
||||
if status == "WARN":
|
||||
return 0.5
|
||||
return 0.0
|
||||
|
||||
|
||||
def normalize_priority(value) -> int:
|
||||
try:
|
||||
priority = int(value)
|
||||
except (TypeError, ValueError):
|
||||
priority = 3
|
||||
if priority in PRIORITY_MULTIPLIERS:
|
||||
return priority
|
||||
return 3
|
||||
|
||||
|
||||
def priority_label(priority: int) -> str:
|
||||
return f"P{priority}"
|
||||
|
||||
|
||||
def validate_result(result: dict) -> dict | None:
|
||||
if not isinstance(result, dict):
|
||||
return None
|
||||
if result.get("status") not in VALID_STATUSES:
|
||||
return None
|
||||
if (
|
||||
"rationale" not in result
|
||||
or "evidence_quotes" not in result
|
||||
or "recommendations" not in result
|
||||
or "confidence" not in result
|
||||
):
|
||||
return None
|
||||
if not isinstance(result.get("evidence_quotes"), list):
|
||||
return None
|
||||
if not isinstance(result.get("recommendations"), list):
|
||||
return None
|
||||
return result
|
||||
|
||||
|
||||
def fallback_invalid_json() -> dict:
|
||||
return {
|
||||
"status": "WARN",
|
||||
"rationale": "Invalid JSON from evaluator after retries.",
|
||||
"evidence_quotes": [],
|
||||
"recommendations": ["Re-run the evaluator with the json_fix prompt."],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
|
||||
|
||||
def build_weights(checks: list[dict]) -> dict:
|
||||
enabled = [c for c in checks if c.get("enabled_by_default")]
|
||||
raw_weights = []
|
||||
for check in enabled:
|
||||
weight = check.get("weight")
|
||||
base_weight = weight if isinstance(weight, (int, float)) else 1.0
|
||||
priority = normalize_priority(check.get("priority"))
|
||||
raw_weights.append(base_weight * PRIORITY_MULTIPLIERS[priority])
|
||||
total = sum(raw_weights) if raw_weights else 1.0
|
||||
weights = {}
|
||||
for check, raw in zip(enabled, raw_weights):
|
||||
weights[check["id"]] = (raw / total) * 100.0
|
||||
return weights
|
||||
|
||||
|
||||
def build_results(
|
||||
checks: list[dict],
|
||||
deterministic_results: dict,
|
||||
llm_results: dict,
|
||||
execution_summary: dict | None,
|
||||
mode: str,
|
||||
) -> dict:
|
||||
results = {}
|
||||
execution_status = None
|
||||
if execution_summary:
|
||||
execution_status = execution_summary.get("overall_status")
|
||||
|
||||
for check in checks:
|
||||
if not check.get("enabled_by_default"):
|
||||
continue
|
||||
check_id = check["id"]
|
||||
check_type = check.get("type")
|
||||
|
||||
if check_type == "DETERMINISTIC":
|
||||
result = deterministic_results.get(check_id)
|
||||
if result:
|
||||
valid = validate_result(result)
|
||||
results[check_id] = valid if valid else fallback_invalid_json()
|
||||
else:
|
||||
results[check_id] = {
|
||||
"status": "FAIL",
|
||||
"rationale": "Deterministic result missing for this check.",
|
||||
"evidence_quotes": [],
|
||||
"recommendations": ["Run deterministic_rules.py to populate results."],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
continue
|
||||
|
||||
if check_type == "LLM":
|
||||
llm_result = llm_results.get(check_id)
|
||||
if llm_result:
|
||||
valid = validate_result(llm_result)
|
||||
results[check_id] = valid if valid else fallback_invalid_json()
|
||||
else:
|
||||
results[check_id] = {
|
||||
"status": "WARN",
|
||||
"rationale": "LLM evaluation missing for this check.",
|
||||
"evidence_quotes": [],
|
||||
"recommendations": ["Run the evaluator prompt for this check."],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
continue
|
||||
|
||||
if check_type == "HYBRID":
|
||||
if mode == "read-only":
|
||||
status = "NOT_RUN"
|
||||
else:
|
||||
status = (
|
||||
execution_status
|
||||
or deterministic_results.get(check_id, {}).get("status")
|
||||
or "NOT_RUN"
|
||||
)
|
||||
llm_result = llm_results.get(check_id)
|
||||
if llm_result:
|
||||
valid = validate_result(llm_result) or fallback_invalid_json()
|
||||
valid["status"] = status
|
||||
results[check_id] = valid
|
||||
else:
|
||||
results[check_id] = {
|
||||
"status": status,
|
||||
"rationale": "Execution summary present but LLM rationale missing."
|
||||
if status != "NOT_RUN"
|
||||
else "Execution not run.",
|
||||
"evidence_quotes": [],
|
||||
"recommendations": [
|
||||
"Provide execution rationale using the execution_explanation prompt."
|
||||
],
|
||||
"confidence": 0.0,
|
||||
}
|
||||
continue
|
||||
|
||||
return results
|
||||
|
||||
|
||||
def render_html(report: dict) -> str:
|
||||
score = report["scorecard"]["score_total"]
|
||||
status = report["scorecard"]["overall_status"]
|
||||
results = report.get("results", {})
|
||||
enabled_checks = report.get("enabled_checks", [])
|
||||
checks_by_id = {check["id"]: check for check in enabled_checks}
|
||||
run_context = report.get("run_context", {})
|
||||
file_path = run_context.get("repo_root") or run_context.get("cwd") or ""
|
||||
|
||||
def status_class(value: str) -> str:
|
||||
return value.lower()
|
||||
|
||||
html = [
|
||||
"<!doctype html>",
|
||||
"<html>",
|
||||
"<head>",
|
||||
"<meta charset='utf-8'>",
|
||||
"<title>Codex Readiness Unit Test Report</title>",
|
||||
"<style>",
|
||||
"body{font-family:Arial,sans-serif;margin:24px;color:#222;background:#fafafa;}",
|
||||
".badge{display:inline-block;padding:4px 10px;border-radius:12px;color:#fff;font-size:12px;text-transform:uppercase;}",
|
||||
".pass{background:#2e7d32;} .warn{background:#f9a825;} .fail{background:#c62828;} .not_run{background:#546e7a;}",
|
||||
"table{border-collapse:collapse;width:100%;margin:12px 0 20px;background:#fff;table-layout:fixed;}",
|
||||
"th,td{border:1px solid #ddd;padding:8px;font-size:13px;vertical-align:top;word-break:break-word;}",
|
||||
"th{background:#f1f1f1;text-align:left;}",
|
||||
"h3{margin:12px 0 6px;}",
|
||||
"h2{margin-top:24px;}",
|
||||
"</style>",
|
||||
"</head>",
|
||||
"<body>",
|
||||
"<h1>Codex Readiness Unit Test Report</h1>",
|
||||
f"<h2>{file_path}</h2>" if file_path else "",
|
||||
f"<p>Overall score: <strong>{score}</strong> <span class='badge {status_class(status)}'>{status}</span></p>",
|
||||
"<h2>Checks</h2>",
|
||||
]
|
||||
colgroup = (
|
||||
"<colgroup><col style='width:42%'><col style='width:12%'><col style='width:46%'></colgroup>"
|
||||
)
|
||||
grouped: dict[str, list[str]] = {
|
||||
f"P{priority}": [] for priority in sorted(PRIORITY_MULTIPLIERS.keys())
|
||||
}
|
||||
for check in enabled_checks:
|
||||
check_id = check["id"]
|
||||
label = check.get("priority_label") or priority_label(
|
||||
normalize_priority(check.get("priority"))
|
||||
)
|
||||
if label in grouped:
|
||||
grouped[label].append(check_id)
|
||||
|
||||
for label in ["P0", "P1", "P2", "P3"]:
|
||||
check_ids = grouped.get(label, [])
|
||||
if not check_ids:
|
||||
continue
|
||||
html.append(f"<h3>{label}</h3>")
|
||||
html.append("<table>")
|
||||
html.append(colgroup)
|
||||
html.append("<tr><th>Check</th><th>Status</th><th>Rationale</th></tr>")
|
||||
for check_id in check_ids:
|
||||
result = results.get(check_id, {})
|
||||
title = checks_by_id.get(check_id, {}).get("title") or check_id
|
||||
rationale = result.get("rationale", "")
|
||||
status_value = result.get("status", "WARN")
|
||||
html.append(
|
||||
f"<tr><td>{title}</td><td><span class='badge {status_class(status_value)}'>{status_value}</span></td><td>{rationale}</td></tr>"
|
||||
)
|
||||
html.append("</table>")
|
||||
html.append("</body></html>")
|
||||
return "\n".join(html)
|
||||
|
||||
|
||||
def build_summary(
|
||||
checks: list[dict], results: dict, overall_counts: dict, overall_status: str
|
||||
) -> dict:
|
||||
enabled_checks = [c for c in checks if c.get("enabled_by_default")]
|
||||
by_status: dict[str, list[dict[str, Any]]] = {"PASS": [], "FAIL": [], "WARN": [], "NOT_RUN": []}
|
||||
all_checks = []
|
||||
for check in enabled_checks:
|
||||
check_id = check["id"]
|
||||
result = results.get(check_id, {})
|
||||
status = result.get("status", "WARN")
|
||||
priority = normalize_priority(check.get("priority"))
|
||||
entry = {
|
||||
"id": check_id,
|
||||
"title": check.get("title") or check_id,
|
||||
"status": status,
|
||||
"priority": priority,
|
||||
"priority_label": priority_label(priority),
|
||||
}
|
||||
all_checks.append(entry)
|
||||
if status == "PASS":
|
||||
by_status["PASS"].append(entry)
|
||||
continue
|
||||
detail = {
|
||||
**entry,
|
||||
"rationale": result.get("rationale", ""),
|
||||
"recommendations": result.get("recommendations", []),
|
||||
}
|
||||
by_status.get(status, by_status["WARN"]).append(detail)
|
||||
|
||||
return {
|
||||
"overall_status": overall_status,
|
||||
"counts": overall_counts,
|
||||
"checks": all_checks,
|
||||
"passed": by_status["PASS"],
|
||||
"failed": by_status["FAIL"],
|
||||
"warned": by_status["WARN"],
|
||||
"not_run": by_status["NOT_RUN"],
|
||||
}
|
||||
|
||||
|
||||
def render_summary_text(summary: dict) -> str:
|
||||
counts = summary.get("counts", {})
|
||||
lines = [
|
||||
"Summary",
|
||||
f"Overall status: {summary.get('overall_status')}",
|
||||
f"Counts: PASS={counts.get('PASS', 0)} WARN={counts.get('WARN', 0)} FAIL={counts.get('FAIL', 0)} NOT_RUN={counts.get('NOT_RUN', 0)}",
|
||||
]
|
||||
|
||||
def render_section(label: str, items: list[dict], include_details: bool = False) -> None:
|
||||
if not items:
|
||||
return
|
||||
lines.append(f"{label}:")
|
||||
for item in items:
|
||||
title = item.get("title") or item.get("id")
|
||||
lines.append(f"- {item.get('priority_label')} {title} ({item.get('id')})")
|
||||
if include_details:
|
||||
rationale = item.get("rationale", "")
|
||||
if rationale:
|
||||
lines.append(f" Rationale: {rationale}")
|
||||
recommendations = item.get("recommendations", [])
|
||||
if recommendations:
|
||||
lines.append(f" Recommendations: {', '.join(recommendations)}")
|
||||
|
||||
render_section("Failed", summary.get("failed", []), include_details=True)
|
||||
render_section("Warned", summary.get("warned", []), include_details=True)
|
||||
render_section("Not run", summary.get("not_run", []), include_details=True)
|
||||
render_section("Passed", summary.get("passed", []), include_details=False)
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Compute scorecard and render report outputs.")
|
||||
parser.add_argument("--mode", choices=["read-only", "execute"], required=True, help="Run mode")
|
||||
parser.add_argument(
|
||||
"--out-dir", default=".codex-readiness-unit-test", help="Base output directory"
|
||||
)
|
||||
parser.add_argument("--run-dir", default=None, help="Specific run directory to use")
|
||||
parser.add_argument(
|
||||
"--checks",
|
||||
default=str(
|
||||
Path(__file__).resolve().parents[1]
|
||||
/ "references"
|
||||
/ "checks"
|
||||
/ "checks.json"
|
||||
),
|
||||
help="Path to checks.json",
|
||||
)
|
||||
parser.add_argument("--evidence", default=None, help="Path to evidence.json (optional)")
|
||||
parser.add_argument(
|
||||
"--deterministic", default=None, help="Path to deterministic results (optional)"
|
||||
)
|
||||
parser.add_argument("--llm", default=None, help="Path to LLM results (optional)")
|
||||
args = parser.parse_args()
|
||||
|
||||
base_dir = Path(args.out_dir)
|
||||
base_dir.mkdir(parents=True, exist_ok=True)
|
||||
run_dir = resolve_run_dir(base_dir, args.run_dir)
|
||||
run_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
checks = load_json(Path(args.checks)).get("checks", [])
|
||||
evidence_path = Path(args.evidence) if args.evidence else (run_dir / "evidence.json")
|
||||
deterministic_path = (
|
||||
Path(args.deterministic) if args.deterministic else (run_dir / "deterministic_results.json")
|
||||
)
|
||||
llm_path = Path(args.llm) if args.llm else (run_dir / "llm_results.json")
|
||||
evidence = load_json(evidence_path)
|
||||
deterministic_results = (
|
||||
load_json(deterministic_path).get("results", {}) if deterministic_path.exists() else {}
|
||||
)
|
||||
llm_results = load_json(llm_path) if llm_path.exists() else {}
|
||||
|
||||
execution_summary_path = run_dir / "execution_summary.json"
|
||||
execution_summary = (
|
||||
load_json(execution_summary_path) if execution_summary_path.exists() else None
|
||||
)
|
||||
|
||||
weights = build_weights(checks)
|
||||
results = build_results(
|
||||
checks, deterministic_results, llm_results, execution_summary, args.mode
|
||||
)
|
||||
|
||||
overall_counts = {"PASS": 0, "WARN": 0, "FAIL": 0, "NOT_RUN": 0}
|
||||
per_check_contributions = {}
|
||||
total = 0.0
|
||||
for check in checks:
|
||||
if not check.get("enabled_by_default"):
|
||||
continue
|
||||
check_id = check["id"]
|
||||
result = results.get(check_id, {})
|
||||
status = result.get("status", "WARN")
|
||||
overall_counts[status] = overall_counts.get(status, 0) + 1
|
||||
weight = weights.get(check_id, 0.0)
|
||||
contribution = weight * status_points(status)
|
||||
total += contribution
|
||||
per_check_contributions[check_id] = {
|
||||
"status": status,
|
||||
"weight": round(weight, 2),
|
||||
"contribution": round(contribution, 2),
|
||||
"priority": normalize_priority(check.get("priority")),
|
||||
"priority_label": priority_label(normalize_priority(check.get("priority"))),
|
||||
}
|
||||
|
||||
overall_status = "PASS"
|
||||
if overall_counts.get("FAIL"):
|
||||
overall_status = "FAIL"
|
||||
elif overall_counts.get("WARN") or overall_counts.get("NOT_RUN"):
|
||||
overall_status = "WARN"
|
||||
|
||||
enabled_checks = []
|
||||
for check in checks:
|
||||
if not check.get("enabled_by_default"):
|
||||
continue
|
||||
priority = normalize_priority(check.get("priority"))
|
||||
enabled_checks.append(
|
||||
{
|
||||
**check,
|
||||
"priority": priority,
|
||||
"priority_label": priority_label(priority),
|
||||
"priority_multiplier": PRIORITY_MULTIPLIERS[priority],
|
||||
"weight": round(weights.get(check["id"], 0.0), 2),
|
||||
}
|
||||
)
|
||||
|
||||
report = {
|
||||
"schema_version": "1.0",
|
||||
"tool_name": "codex-readiness-unit-test",
|
||||
"tool_version": "0.1.0",
|
||||
"run_context": evidence.get("run_context", {}),
|
||||
"enabled_checks": enabled_checks,
|
||||
"results": results,
|
||||
"execution_summary": execution_summary if args.mode == "execute" else None,
|
||||
"scorecard": {
|
||||
"score_total_raw": round(total, 2),
|
||||
"score_total": round_half_up(total),
|
||||
"per_check_contributions": per_check_contributions,
|
||||
"counts": overall_counts,
|
||||
"overall_status": overall_status,
|
||||
},
|
||||
}
|
||||
|
||||
report_path = run_dir / "report.json"
|
||||
report_path.write_text(json.dumps(report, indent=2), encoding="utf-8")
|
||||
|
||||
html = render_html(report)
|
||||
html_path = run_dir / "report.html"
|
||||
html_path.write_text(html, encoding="utf-8")
|
||||
|
||||
summary = build_summary(checks, results, overall_counts, overall_status)
|
||||
summary_path = run_dir / "summary.json"
|
||||
summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8")
|
||||
|
||||
print(str(report_path))
|
||||
print(str(html_path))
|
||||
print(str(summary_path))
|
||||
print(render_summary_text(summary))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,202 +0,0 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -1,74 +0,0 @@
|
||||
---
|
||||
name: create-plan
|
||||
description: Create a concise plan. Use when a user explicitly asks for a plan related to a coding task.
|
||||
metadata:
|
||||
short-description: Create a plan
|
||||
---
|
||||
|
||||
# Create Plan
|
||||
|
||||
## Goal
|
||||
|
||||
Turn a user prompt into a **single, actionable plan** delivered in the final assistant message.
|
||||
|
||||
## Minimal workflow
|
||||
|
||||
Throughout the entire workflow, operate in read-only mode. Do not write or update files.
|
||||
|
||||
1. **Scan context quickly**
|
||||
- Read `README.md` and any obvious docs (`docs/`, `CONTRIBUTING.md`, `ARCHITECTURE.md`).
|
||||
- Skim relevant files (the ones most likely touched).
|
||||
- Identify constraints (language, frameworks, CI/test commands, deployment shape).
|
||||
|
||||
2. **Ask follow-ups only if blocking**
|
||||
- Ask **at most 1–2 questions**.
|
||||
- Only ask if you cannot responsibly plan without the answer; prefer multiple-choice.
|
||||
- If unsure but not blocked, make a reasonable assumption and proceed.
|
||||
|
||||
3. **Create a plan using the template below**
|
||||
- Start with **1 short paragraph** describing the intent and approach.
|
||||
- Clearly call out what is **in scope** and what is **not in scope** in short.
|
||||
- Then provide a **small checklist** of action items (default 6–10 items).
|
||||
- Each checklist item should be a concrete action and, when helpful, mention files/commands.
|
||||
- **Make items atomic and ordered**: discovery → changes → tests → rollout.
|
||||
- **Verb-first**: “Add…”, “Refactor…”, “Verify…”, “Ship…”.
|
||||
- Include at least one item for **tests/validation** and one for **edge cases/risk** when applicable.
|
||||
- If there are unknowns, include a tiny **Open questions** section (max 3).
|
||||
|
||||
4. **Do not preface the plan with meta explanations; output only the plan as per template**
|
||||
|
||||
## Plan template (follow exactly)
|
||||
|
||||
```markdown
|
||||
# Plan
|
||||
|
||||
<1–3 sentences: what we’re doing, why, and the high-level approach.>
|
||||
|
||||
## Scope
|
||||
- In:
|
||||
- Out:
|
||||
|
||||
## Action items
|
||||
[ ] <Step 1>
|
||||
[ ] <Step 2>
|
||||
[ ] <Step 3>
|
||||
[ ] <Step 4>
|
||||
[ ] <Step 5>
|
||||
[ ] <Step 6>
|
||||
|
||||
## Open questions
|
||||
- <Question 1>
|
||||
- <Question 2>
|
||||
- <Question 3>
|
||||
```
|
||||
|
||||
## Checklist item guidance
|
||||
Good checklist items:
|
||||
- Point to likely files/modules: src/..., app/..., services/...
|
||||
- Name concrete validation: “Run npm test”, “Add unit tests for X”
|
||||
- Include safe rollout when relevant: feature flag, migration plan, rollback note
|
||||
|
||||
Avoid:
|
||||
- Vague steps (“handle backend”, “do auth”)
|
||||
- Too many micro-steps
|
||||
- Writing code snippets (keep the plan implementation-agnostic)
|
||||
@@ -1,202 +0,0 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -1,59 +0,0 @@
|
||||
---
|
||||
name: gitlab-address-comments
|
||||
description: Help address review/issue comments on the open GitLab MR for the current branch using glab CLI. Use when the user wants help addressing review/issue comments on an open GitLab MR
|
||||
metadata:
|
||||
short-description: Address comments in a GitLab MR review
|
||||
---
|
||||
|
||||
# MR Comment Handler
|
||||
|
||||
Find the open MR for the current branch and address its review threads using `glab`. Run all `glab` commands with elevated network access.
|
||||
|
||||
## Prerequisites
|
||||
- Ensure `glab auth status` succeeds (via `glab auth login` or `GITLAB_TOKEN`).
|
||||
- Ensure `glab` is at least v1.80.4.
|
||||
- When sandboxing blocks network calls, rerun with `sandbox_permissions=require_escalated`.
|
||||
- Sanity check auth up front:
|
||||
```bash
|
||||
glab auth status
|
||||
```
|
||||
|
||||
## 1) Resolve the MR for the current branch
|
||||
Do a quick check so we know which MR we are about to operate on:
|
||||
```bash
|
||||
branch="$(git rev-parse --abbrev-ref HEAD)"
|
||||
glab mr view "$branch" --output json
|
||||
```
|
||||
If this fails, the fetch script below will still try to locate the MR by `source_branch`.
|
||||
|
||||
## 2) Fetch unresolved discussions to `/tmp`
|
||||
Use the local script to fetch MR discussions via `glab api`. This filters out bot/system-only threads and returns unresolved discussions when `--open-comments` is set.
|
||||
```bash
|
||||
skill_dir="<path-to-skill>"
|
||||
branch="$(git rev-parse --abbrev-ref HEAD)"
|
||||
safe_branch="${branch//\//_}"
|
||||
out="/tmp/${safe_branch}_mr_open_discussions.json"
|
||||
python "$skill_dir/scripts/fetch_comments.py" --open-comments --output "$out"
|
||||
```
|
||||
If you want the full payload (including resolved discussions), drop `--open-comments`.
|
||||
|
||||
## 3) Summarize, triage, and ask once
|
||||
- Load the JSON and number each unresolved discussion.
|
||||
- Start with a compact summary list instead of dumping full threads.
|
||||
- Sort by: unresolved first (already filtered), then most recently updated.
|
||||
- When possible, group or label by file path to reduce context switching.
|
||||
- In the summary list, show: number, discussion id, author, file:line (if present), and a one-line summary.
|
||||
- Ask for a batch selection in one shot. Accept: `1,3,5-7`, `all`, `none`, or `top N`.
|
||||
- If the user does not choose, suggest a small default set (for example `top 3`) with a short rationale.
|
||||
- Only after selection, show the full thread and code context for the selected numbers.
|
||||
- When showing code context, include 3 lines before and after and clearly mark the referenced line(s).
|
||||
|
||||
## 4) Implement fixes for the selected discussions
|
||||
- Apply focused fixes that address the selected threads.
|
||||
- Run the most relevant tests or checks you can in-repo.
|
||||
- Report back with: what changed, which discussion numbers were addressed, and any follow-ups.
|
||||
|
||||
Notes:
|
||||
- If `glab` hits auth or rate issues, prompt the user to run `glab auth login` or re-export `GITLAB_TOKEN`, then retry.
|
||||
- If no open MR is found for the branch, say so clearly and ask for the MR URL or IID.
|
||||
- Do not prompt “address or skip?” one comment at a time unless the user explicitly asks for that mode.
|
||||
@@ -1,274 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Fetch GitLab merge request discussions (including inline threads) for the MR
|
||||
associated with the current git branch, by shelling out to:
|
||||
|
||||
glab api
|
||||
|
||||
Requires:
|
||||
- `glab auth status` succeeds (uses glab config/keyring or GITLAB_TOKEN)
|
||||
- current branch has an associated open MR
|
||||
|
||||
Usage:
|
||||
python scripts/fetch_comments.py > /tmp/mr_comments.json
|
||||
python scripts/fetch_comments.py --open-comments > /tmp/open_threads.json
|
||||
python scripts/fetch_comments.py --output /tmp/mr_comments.json
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any, Iterable
|
||||
from urllib.parse import quote
|
||||
|
||||
|
||||
def _run(cmd: list[str]) -> str:
|
||||
p = subprocess.run(cmd, capture_output=True, text=True)
|
||||
if p.returncode != 0:
|
||||
raise RuntimeError(f"Command failed: {' '.join(cmd)}\n{p.stderr}")
|
||||
return p.stdout
|
||||
|
||||
|
||||
def _run_json(cmd: list[str]) -> Any:
|
||||
out = _run(cmd)
|
||||
try:
|
||||
return json.loads(out)
|
||||
except json.JSONDecodeError as e:
|
||||
raise RuntimeError(f"Failed to parse JSON from command output: {e}\nRaw:\n{out}") from e
|
||||
|
||||
|
||||
def _ensure_glab_authenticated() -> None:
|
||||
try:
|
||||
_run(["glab", "auth", "status"])
|
||||
except RuntimeError as exc:
|
||||
raise RuntimeError(
|
||||
"glab auth status failed; run `glab auth login` or set GITLAB_TOKEN"
|
||||
) from exc
|
||||
|
||||
|
||||
def _git_current_branch() -> str:
|
||||
return _run(["git", "rev-parse", "--abbrev-ref", "HEAD"]).strip()
|
||||
|
||||
|
||||
def _git_origin_url() -> str:
|
||||
return _run(["git", "remote", "get-url", "origin"]).strip()
|
||||
|
||||
|
||||
def _strip_dot_git(path: str) -> str:
|
||||
return path[:-4] if path.endswith(".git") else path
|
||||
|
||||
|
||||
def _parse_project_path(remote_url: str) -> str:
|
||||
"""
|
||||
Convert a git remote URL into a GitLab project path (group/subgroup/project).
|
||||
Supports common SSH and HTTPS formats.
|
||||
"""
|
||||
# HTTPS: https://gitlab.example.com/group/project.git
|
||||
https_match = re.match(r"^https?://[^/]+/(.+)$", remote_url)
|
||||
if https_match:
|
||||
return _strip_dot_git(https_match.group(1))
|
||||
|
||||
# SSH scp-like: git@gitlab.example.com:group/project.git
|
||||
ssh_match = re.match(r"^(?:ssh://)?git@[^:/]+[:/](.+)$", remote_url)
|
||||
if ssh_match:
|
||||
return _strip_dot_git(ssh_match.group(1))
|
||||
|
||||
raise RuntimeError(f"Unable to parse GitLab project path from origin URL: {remote_url}")
|
||||
|
||||
|
||||
def _glab_api_get(endpoint: str, params: dict[str, Any] | None = None) -> Any:
|
||||
cmd = ["glab", "api", endpoint, "-X", "GET"]
|
||||
for key, value in (params or {}).items():
|
||||
if value is None:
|
||||
continue
|
||||
cmd += ["-F", f"{key}={value}"]
|
||||
return _run_json(cmd)
|
||||
|
||||
|
||||
def _paginate(endpoint: str, base_params: dict[str, Any], per_page: int = 100, max_pages: int = 20) -> list[Any]:
|
||||
results: list[Any] = []
|
||||
page = 1
|
||||
|
||||
while page <= max_pages:
|
||||
params = dict(base_params)
|
||||
params.update({"per_page": per_page, "page": page})
|
||||
chunk = _glab_api_get(endpoint, params=params)
|
||||
|
||||
if not isinstance(chunk, list) or not chunk:
|
||||
break
|
||||
|
||||
results.extend(chunk)
|
||||
|
||||
if len(chunk) < per_page:
|
||||
break
|
||||
page += 1
|
||||
|
||||
return results
|
||||
|
||||
|
||||
def _encode_project_path(project_path: str) -> str:
|
||||
# GitLab API accepts URL-encoded project paths in place of numeric IDs.
|
||||
return quote(project_path, safe="")
|
||||
|
||||
|
||||
def _find_open_mr_for_branch(project: str, branch: str) -> dict[str, Any]:
|
||||
endpoint = f"/projects/{project}/merge_requests"
|
||||
mrs = _paginate(
|
||||
endpoint,
|
||||
base_params={
|
||||
"state": "opened",
|
||||
"source_branch": branch,
|
||||
"order_by": "updated_at",
|
||||
"sort": "desc",
|
||||
},
|
||||
)
|
||||
if not mrs:
|
||||
raise RuntimeError(f"No open merge request found for source branch: {branch}")
|
||||
# Prefer the most recently updated MR for this branch.
|
||||
return mrs[0]
|
||||
|
||||
|
||||
def _get_mr(project: str, mr_iid: int) -> dict[str, Any]:
|
||||
endpoint = f"/projects/{project}/merge_requests/{mr_iid}"
|
||||
mr = _glab_api_get(endpoint)
|
||||
if not isinstance(mr, dict):
|
||||
raise RuntimeError("Unexpected response when fetching merge request details")
|
||||
return mr
|
||||
|
||||
|
||||
def _discussion_notes(discussion: dict[str, Any]) -> list[dict[str, Any]]:
|
||||
notes = discussion.get("notes")
|
||||
return notes if isinstance(notes, list) else []
|
||||
|
||||
|
||||
def _is_bot_or_system_note(note: dict[str, Any]) -> bool:
|
||||
if note.get("system"):
|
||||
return True
|
||||
author = note.get("author") or {}
|
||||
if author.get("bot"):
|
||||
return True
|
||||
username = str(author.get("username") or "").lower()
|
||||
return username.endswith("[bot]") or username.endswith("-bot")
|
||||
|
||||
|
||||
def _filter_bot_notes(notes: Iterable[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
return [n for n in notes if not _is_bot_or_system_note(n)]
|
||||
|
||||
|
||||
def _discussion_is_open(discussion: dict[str, Any]) -> bool:
|
||||
# Prefer the top-level resolved flag if present.
|
||||
resolved = discussion.get("resolved")
|
||||
if resolved is True:
|
||||
return False
|
||||
|
||||
notes = _discussion_notes(discussion)
|
||||
resolvable_notes = [n for n in notes if n.get("resolvable")]
|
||||
if resolvable_notes:
|
||||
# If any resolvable note is still unresolved, treat the discussion as open.
|
||||
return any(not bool(n.get("resolved")) for n in resolvable_notes)
|
||||
|
||||
# Fallback: treat unresolved/unknown as open.
|
||||
return not bool(resolved)
|
||||
|
||||
|
||||
def _discussion_has_non_bot_content(discussion: dict[str, Any]) -> bool:
|
||||
notes = _discussion_notes(discussion)
|
||||
return len(_filter_bot_notes(notes)) > 0
|
||||
|
||||
|
||||
def fetch_all(project_path: str, branch: str) -> dict[str, Any]:
|
||||
encoded_project = _encode_project_path(project_path)
|
||||
|
||||
mr_hint = _find_open_mr_for_branch(encoded_project, branch)
|
||||
mr_iid = int(mr_hint["iid"])
|
||||
mr = _get_mr(encoded_project, mr_iid)
|
||||
|
||||
discussions_endpoint = f"/projects/{encoded_project}/merge_requests/{mr_iid}/discussions"
|
||||
discussions = _paginate(discussions_endpoint, base_params={})
|
||||
|
||||
# Drop pure bot/system discussions.
|
||||
discussions = [d for d in discussions if _discussion_has_non_bot_content(d)]
|
||||
|
||||
open_discussions = [d for d in discussions if _discussion_is_open(d)]
|
||||
|
||||
mr_meta = {
|
||||
"iid": mr.get("iid"),
|
||||
"project_id": mr.get("project_id"),
|
||||
"web_url": mr.get("web_url"),
|
||||
"title": mr.get("title"),
|
||||
"state": mr.get("state"),
|
||||
"source_branch": mr.get("source_branch"),
|
||||
"target_branch": mr.get("target_branch"),
|
||||
"updated_at": mr.get("updated_at"),
|
||||
}
|
||||
|
||||
return {
|
||||
"merge_request": mr_meta,
|
||||
"project": {
|
||||
"path": project_path,
|
||||
"encoded_path": encoded_project,
|
||||
"branch": branch,
|
||||
},
|
||||
"discussions": discussions,
|
||||
"open_discussions": open_discussions,
|
||||
}
|
||||
|
||||
|
||||
def _build_arg_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description="Fetch GitLab MR discussions for the current branch")
|
||||
parser.add_argument(
|
||||
"--open-comments",
|
||||
action="store_true",
|
||||
help="emit only unresolved/open discussions (still includes merge_request metadata)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output",
|
||||
help="optional path to write JSON output; defaults to stdout",
|
||||
)
|
||||
return parser
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = _build_arg_parser()
|
||||
args = parser.parse_args()
|
||||
|
||||
_ensure_glab_authenticated()
|
||||
|
||||
branch = _git_current_branch()
|
||||
origin = _git_origin_url()
|
||||
project_path = _parse_project_path(origin)
|
||||
|
||||
result = fetch_all(project_path=project_path, branch=branch)
|
||||
|
||||
if args.open_comments:
|
||||
payload: Any = {
|
||||
"merge_request": result["merge_request"],
|
||||
"project": result["project"],
|
||||
"open_discussions": result["open_discussions"],
|
||||
}
|
||||
else:
|
||||
payload = result
|
||||
|
||||
output = json.dumps(payload, indent=2)
|
||||
|
||||
if args.output:
|
||||
out_path = Path(args.output)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(out_path, "w", encoding="utf-8") as f:
|
||||
f.write(output)
|
||||
f.write("\n")
|
||||
else:
|
||||
print(output)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
main()
|
||||
except RuntimeError as exc:
|
||||
print(str(exc), file=sys.stderr)
|
||||
sys.exit(2)
|
||||
@@ -1,202 +0,0 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -1,40 +0,0 @@
|
||||
---
|
||||
name: "codex-wrapped"
|
||||
description: "Generate a Codex Wrapped usage recap from local Codex logs, including last 30 days, last 7 days, and an all-time focus-hours callout. Use when the user asks for a usage summary, activity recap, or Codex Wrapped report."
|
||||
---
|
||||
|
||||
|
||||
# Codex Wrapped
|
||||
|
||||
Use this skill whenever the user wants a Codex Wrapped report or usage insights. Render text-only output (no image generation).
|
||||
|
||||
The report must be year-agnostic and should highlight last 30 days and last 7 days, while still calling out all-time focus hours.
|
||||
|
||||
## Quick Commands (run in order)
|
||||
|
||||
1) **Compute stats**
|
||||
```bash
|
||||
python3 .codex/skills/codex-wrapped/scripts/get_codex_stats.py \
|
||||
--output /tmp/wrapped_stats.json
|
||||
```
|
||||
(Defaults to the system timezone; override `--timezone` only if the user requests it.)
|
||||
|
||||
2) **Render text report**
|
||||
```bash
|
||||
.codex/skills/codex-wrapped/scripts/report.sh \
|
||||
--stats-file /tmp/wrapped_stats.json
|
||||
```
|
||||
This prints the report directly to stdout.
|
||||
|
||||
## Files
|
||||
- `scripts/get_codex_stats.py` -- computes rolling-window stats to `/tmp/wrapped_stats.json`.
|
||||
- `scripts/report.sh` -- text report renderer.
|
||||
|
||||
## Responding to the user
|
||||
- Paste the report text exactly as printed, wrapped in triple backticks (```), to preserve spacing/box drawing.
|
||||
- If something fails, state what you ran and the error.
|
||||
|
||||
## Notes
|
||||
- Keep `/tmp/wrapped_stats.json` unless sensitive; rerun stats if outdated.
|
||||
- The report adapts to terminal width. Set `WRAPPED_WIDTH=120` (or similar) to force a wider layout.
|
||||
- Layout options: default is `columns` (two-column). Use `--layout table` (or `WRAPPED_LAYOUT=table`) to switch back to the compact grid.
|
||||
@@ -1,4 +0,0 @@
|
||||
interface:
|
||||
display_name: "Wrapped"
|
||||
short_description: "Create a Codex activity report from local usage data"
|
||||
default_prompt: "Generate a Codex Wrapped usage report from my local Codex logs."
|
||||
@@ -1,451 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Aggregate Codex usage metrics for the Wrapped report.
|
||||
|
||||
Outputs JSON with rolling windows:
|
||||
- all_time
|
||||
- last_30_days
|
||||
- last_7_days
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
from typing import Iterable
|
||||
from zoneinfo import ZoneInfo, ZoneInfoNotFoundError
|
||||
|
||||
CODEX_HOME = Path.home() / ".codex"
|
||||
SESSION_DIRS = ["sessions", "archived_sessions"]
|
||||
DEFAULT_TIMEZONE = None
|
||||
DEFAULT_OUTPUT_PATH = Path(__file__).with_name("wrapped_stats.json")
|
||||
WINDOW_DELTAS = {
|
||||
"all_time": None,
|
||||
"last_30_days": 30,
|
||||
"last_7_days": 7,
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class WindowAccumulator:
|
||||
name: str
|
||||
start: datetime | None
|
||||
session_count: int = 0
|
||||
total_assistant_messages: int = 0
|
||||
total_user_messages: int = 0
|
||||
turn_usage_seconds: float = 0.0
|
||||
session_span_seconds: float = 0.0
|
||||
day_tokens: dict[datetime.date, int] = field(default_factory=lambda: defaultdict(int))
|
||||
active_days: set[datetime.date] = field(default_factory=set)
|
||||
hour_usage: dict[int, int] = field(default_factory=lambda: defaultdict(int))
|
||||
repo_usage: dict[str, int] = field(default_factory=lambda: defaultdict(int))
|
||||
longest_turn_duration: float = 0.0
|
||||
longest_turn_timestamp: datetime | None = None
|
||||
longest_turn_session: str | None = None
|
||||
|
||||
|
||||
def parse_timestamp(ts: str) -> datetime:
|
||||
if not ts:
|
||||
raise ValueError("Empty timestamp")
|
||||
if ts.endswith("Z"):
|
||||
ts = ts[:-1] + "+00:00"
|
||||
return datetime.fromisoformat(ts)
|
||||
|
||||
|
||||
def iter_session_files() -> Iterable[Path]:
|
||||
for rel in SESSION_DIRS:
|
||||
root = CODEX_HOME / rel
|
||||
if not root.exists():
|
||||
continue
|
||||
yield from root.rglob("*.jsonl")
|
||||
|
||||
|
||||
def format_tokens(value: int) -> str:
|
||||
if value <= 0:
|
||||
return "0"
|
||||
if value >= 1_000_000:
|
||||
scaled = value / 1_000_000
|
||||
return f"{scaled:.1f}M".replace(".0M", "M")
|
||||
if value >= 1_000:
|
||||
scaled = value / 1_000
|
||||
return f"{scaled:.1f}k".replace(".0k", "k")
|
||||
return f"{value:,}"
|
||||
|
||||
|
||||
def render_usage_hours(seconds: float) -> str:
|
||||
if seconds <= 0:
|
||||
return "0 minutes"
|
||||
hours = seconds / 3600
|
||||
if hours < 1:
|
||||
minutes = int(round(seconds / 60))
|
||||
minutes = max(minutes, 1)
|
||||
return f"{minutes} minute{'s' if minutes != 1 else ''}"
|
||||
rounded_hours = int(round(hours))
|
||||
rounded_hours = max(rounded_hours, 1)
|
||||
return f"{rounded_hours} hour{'s' if rounded_hours != 1 else ''}"
|
||||
|
||||
|
||||
def build_contrib_lines(active_days: set[datetime.date]) -> list[str]:
|
||||
if not active_days:
|
||||
return []
|
||||
|
||||
day_set = set(active_days)
|
||||
first_date = min(day_set)
|
||||
last_date = max(day_set)
|
||||
|
||||
def to_sunday(dt: datetime.date) -> datetime.date:
|
||||
offset = (dt.weekday() + 1) % 7
|
||||
return dt - timedelta(days=offset)
|
||||
|
||||
def to_saturday(dt: datetime.date) -> datetime.date:
|
||||
offset = 6 - ((dt.weekday() + 1) % 7)
|
||||
return dt + timedelta(days=offset)
|
||||
|
||||
start = to_sunday(first_date)
|
||||
end = to_saturday(last_date)
|
||||
total_days = (end - start).days + 1
|
||||
total_weeks = total_days // 7
|
||||
|
||||
weekday_labels = ["Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat"]
|
||||
line_chars: list[list[str]] = [[] for _ in range(7)]
|
||||
|
||||
for week_index in range(total_weeks):
|
||||
week_start = start + timedelta(days=week_index * 7)
|
||||
for day_offset in range(7):
|
||||
current_day = week_start + timedelta(days=day_offset)
|
||||
char = "•" if current_day in day_set else "◦"
|
||||
line_chars[day_offset].append(char)
|
||||
|
||||
contrib_lines = [
|
||||
f"{weekday_labels[idx]} {''.join(chars)}" for idx, chars in enumerate(line_chars) if chars
|
||||
]
|
||||
return contrib_lines
|
||||
|
||||
|
||||
def local_timezone_name() -> str:
|
||||
local_tz = datetime.now().astimezone().tzinfo
|
||||
if hasattr(local_tz, "key") and local_tz.key:
|
||||
return local_tz.key
|
||||
if local_tz:
|
||||
return str(local_tz)
|
||||
return "UTC"
|
||||
|
||||
|
||||
def get_timezone(tz_name: str) -> ZoneInfo:
|
||||
try:
|
||||
return ZoneInfo(tz_name)
|
||||
except ZoneInfoNotFoundError:
|
||||
fallback = local_timezone_name()
|
||||
try:
|
||||
return ZoneInfo(fallback)
|
||||
except ZoneInfoNotFoundError:
|
||||
return ZoneInfo("UTC")
|
||||
|
||||
|
||||
def classify_hour(hour: int | None) -> str:
|
||||
if hour is None:
|
||||
return "unknown"
|
||||
if 22 <= hour or hour <= 3:
|
||||
return "owl"
|
||||
if 4 <= hour <= 9:
|
||||
return "bird"
|
||||
if 10 <= hour <= 16:
|
||||
return "day"
|
||||
return "eve"
|
||||
|
||||
|
||||
def longest_streak(active_days: set[datetime.date]) -> int:
|
||||
if not active_days:
|
||||
return 0
|
||||
sorted_days = sorted(active_days)
|
||||
longest = 1
|
||||
current = 1
|
||||
prev_day = sorted_days[0]
|
||||
for day in sorted_days[1:]:
|
||||
if day == prev_day + timedelta(days=1):
|
||||
current += 1
|
||||
elif day == prev_day:
|
||||
pass
|
||||
else:
|
||||
longest = max(longest, current)
|
||||
current = 1
|
||||
prev_day = day
|
||||
longest = max(longest, current)
|
||||
return longest
|
||||
|
||||
|
||||
def window_stats(window: WindowAccumulator, tz_abbrev: str) -> dict[str, object]:
|
||||
total_tokens = sum(window.day_tokens.values())
|
||||
active_days_count = len(window.active_days)
|
||||
peak_hour = None
|
||||
if window.hour_usage:
|
||||
peak_hour = max(window.hour_usage.items(), key=lambda item: item[1])[0]
|
||||
|
||||
peak_hour_display = f"{peak_hour:02d}:00" if peak_hour is not None else "unknown"
|
||||
peak_hour_label = classify_hour(peak_hour)
|
||||
if peak_hour is not None:
|
||||
peak_hour_slot = f"{peak_hour_display} {peak_hour_label}"
|
||||
else:
|
||||
peak_hour_slot = "unknown"
|
||||
|
||||
biggest_day_date = None
|
||||
biggest_day_tokens = 0
|
||||
if window.day_tokens:
|
||||
biggest_day_date, biggest_day_tokens = max(
|
||||
window.day_tokens.items(), key=lambda item: item[1]
|
||||
)
|
||||
if biggest_day_date:
|
||||
biggest_day_display = f"{biggest_day_date.day} {biggest_day_date.strftime('%b')}"
|
||||
else:
|
||||
biggest_day_display = "unknown"
|
||||
|
||||
top_repo = None
|
||||
if window.repo_usage:
|
||||
top_repo = max(window.repo_usage.items(), key=lambda item: item[1])[0]
|
||||
if top_repo:
|
||||
repo_path = Path(top_repo)
|
||||
top_repo_label = repo_path.name or str(repo_path)
|
||||
else:
|
||||
top_repo_label = "unknown"
|
||||
|
||||
usage_seconds = max(window.session_span_seconds, window.turn_usage_seconds)
|
||||
usage_hours_display = render_usage_hours(usage_seconds)
|
||||
|
||||
streak_days = longest_streak(window.active_days)
|
||||
usage_streak_display = (
|
||||
f"{streak_days} day" if streak_days == 1 else f"{streak_days} days" if streak_days else "unknown"
|
||||
)
|
||||
|
||||
longest_turn_display = "unknown"
|
||||
if window.longest_turn_timestamp is not None:
|
||||
minutes = int(window.longest_turn_duration // 60)
|
||||
seconds = int(round(window.longest_turn_duration % 60))
|
||||
longest_turn_display = f"{minutes}m {seconds}s"
|
||||
|
||||
return {
|
||||
"sessions": window.session_count,
|
||||
"sessions_display": f"{window.session_count:,}",
|
||||
"assistant_messages": window.total_assistant_messages,
|
||||
"assistant_messages_display": f"{window.total_assistant_messages:,}",
|
||||
"user_messages": window.total_user_messages,
|
||||
"user_messages_display": f"{window.total_user_messages:,}",
|
||||
"active_days": active_days_count,
|
||||
"active_days_display": f"{active_days_count:,}",
|
||||
"total_tokens": total_tokens,
|
||||
"total_tokens_display": format_tokens(total_tokens),
|
||||
"usage_hours_display": usage_hours_display,
|
||||
"peak_hour": peak_hour,
|
||||
"peak_hour_display": peak_hour_display,
|
||||
"peak_hour_label": peak_hour_label,
|
||||
"peak_hour_slot": peak_hour_slot,
|
||||
"timezone_abbrev": tz_abbrev,
|
||||
"usage_streak_days": streak_days,
|
||||
"usage_streak_display": usage_streak_display,
|
||||
"biggest_day_display": biggest_day_display,
|
||||
"biggest_day_tokens": biggest_day_tokens,
|
||||
"top_repo_display": top_repo_label,
|
||||
"longest_turn_display": longest_turn_display,
|
||||
}
|
||||
|
||||
|
||||
def gather_metrics(tz_name: str) -> dict[str, object]:
|
||||
target_tz = get_timezone(tz_name)
|
||||
now_local = datetime.now(target_tz)
|
||||
|
||||
windows: dict[str, WindowAccumulator] = {}
|
||||
for name, delta in WINDOW_DELTAS.items():
|
||||
start = None if delta is None else now_local - timedelta(days=delta)
|
||||
windows[name] = WindowAccumulator(name=name, start=start)
|
||||
|
||||
joined_at: datetime | None = None
|
||||
|
||||
for session_path in iter_session_files():
|
||||
current_turn_start: datetime | None = None
|
||||
session_start: datetime | None = None
|
||||
session_end: datetime | None = None
|
||||
session_workspace: str | None = None
|
||||
|
||||
try:
|
||||
with session_path.open("r", encoding="utf-8") as handle:
|
||||
for raw_line in handle:
|
||||
raw_line = raw_line.strip()
|
||||
if not raw_line:
|
||||
continue
|
||||
try:
|
||||
record = json.loads(raw_line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
|
||||
ts_str = record.get("timestamp")
|
||||
if not ts_str:
|
||||
ts_str = record.get("payload", {}).get("timestamp")
|
||||
if not ts_str:
|
||||
continue
|
||||
try:
|
||||
ts = parse_timestamp(ts_str)
|
||||
except ValueError:
|
||||
continue
|
||||
|
||||
local_ts = ts.astimezone(target_tz)
|
||||
if session_start is None:
|
||||
session_start = local_ts
|
||||
session_end = local_ts
|
||||
|
||||
rec_type = record.get("type")
|
||||
payload = record.get("payload", {})
|
||||
|
||||
for window in windows.values():
|
||||
if window.start is None or local_ts >= window.start:
|
||||
window.active_days.add(local_ts.date())
|
||||
|
||||
if rec_type == "response_item" and payload.get("type") == "message":
|
||||
role = payload.get("role")
|
||||
for window in windows.values():
|
||||
if window.start is None or local_ts >= window.start:
|
||||
if role == "assistant":
|
||||
window.total_assistant_messages += 1
|
||||
elif role == "user":
|
||||
window.total_user_messages += 1
|
||||
window.hour_usage[local_ts.hour] += 1
|
||||
|
||||
if rec_type == "session_meta":
|
||||
session_workspace = (
|
||||
payload.get("workspacePath")
|
||||
or payload.get("workspace_path")
|
||||
or payload.get("cwd")
|
||||
or session_workspace
|
||||
)
|
||||
|
||||
if rec_type == "turn_context":
|
||||
session_workspace = (
|
||||
payload.get("workspacePath")
|
||||
or payload.get("workspace_path")
|
||||
or payload.get("cwd")
|
||||
or session_workspace
|
||||
)
|
||||
current_turn_start = ts
|
||||
continue
|
||||
|
||||
if rec_type == "event_msg" and payload.get("type") == "token_count":
|
||||
info = payload.get("info")
|
||||
if not info or current_turn_start is None:
|
||||
continue
|
||||
duration = (ts - current_turn_start).total_seconds()
|
||||
if duration < 0:
|
||||
duration = 0
|
||||
usage = info.get("last_token_usage") or info.get("total_token_usage")
|
||||
tokens = 0
|
||||
if usage and usage.get("total_tokens"):
|
||||
tokens = usage["total_tokens"]
|
||||
|
||||
for window in windows.values():
|
||||
if window.start is None or local_ts >= window.start:
|
||||
window.turn_usage_seconds += duration
|
||||
if duration > window.longest_turn_duration:
|
||||
window.longest_turn_duration = duration
|
||||
window.longest_turn_timestamp = local_ts
|
||||
window.longest_turn_session = session_path.name
|
||||
if tokens:
|
||||
window.day_tokens[local_ts.date()] += tokens
|
||||
current_turn_start = None
|
||||
|
||||
except OSError:
|
||||
continue
|
||||
|
||||
if session_start and (joined_at is None or session_start < joined_at):
|
||||
joined_at = session_start
|
||||
|
||||
if session_start and session_end:
|
||||
for window in windows.values():
|
||||
if window.start is None:
|
||||
window_start = session_start
|
||||
else:
|
||||
window_start = max(window.start, session_start)
|
||||
window_end = min(session_end, now_local)
|
||||
overlap = (window_end - window_start).total_seconds()
|
||||
if overlap > 0:
|
||||
window.session_span_seconds += overlap
|
||||
window.session_count += 1
|
||||
if session_workspace:
|
||||
window.repo_usage[session_workspace] += 1
|
||||
|
||||
tz_abbrev = now_local.strftime("%Z")
|
||||
if joined_at:
|
||||
joined_str = joined_at.date().isoformat()
|
||||
joined_label = joined_at.strftime("%b %d")
|
||||
days_diff = (now_local.date() - joined_at.date()).days
|
||||
days_ago = f"{days_diff} day{'s' if days_diff != 1 else ''} ago"
|
||||
joined_display = f"{joined_label} ({days_ago})"
|
||||
else:
|
||||
joined_str = "unknown"
|
||||
days_ago = "unknown"
|
||||
joined_display = "unknown"
|
||||
|
||||
metrics: dict[str, object] = {
|
||||
"timezone": tz_name,
|
||||
"timezone_abbrev": tz_abbrev,
|
||||
"generated_at": now_local.isoformat(),
|
||||
"joined_date": joined_str,
|
||||
"joined_days_ago": days_ago,
|
||||
"joined_display": joined_display,
|
||||
"joined_relative_display": days_ago,
|
||||
"windows": {name: window_stats(win, tz_abbrev) for name, win in windows.items()},
|
||||
"contrib_lines": build_contrib_lines(windows["all_time"].active_days),
|
||||
}
|
||||
|
||||
return metrics
|
||||
|
||||
|
||||
def print_text(metrics: dict[str, object]) -> None:
|
||||
all_time = metrics.get("windows", {}).get("all_time", {})
|
||||
print(f"Joined Codex: {metrics.get('joined_display', 'unknown')}")
|
||||
print(f"Sessions: {all_time.get('sessions_display', '0')}")
|
||||
print(f"Assistant messages: {all_time.get('assistant_messages_display', '0')}")
|
||||
print(f"Prompts: {all_time.get('user_messages_display', '0')}")
|
||||
print(f"Tokens: {all_time.get('total_tokens_display', '0')}")
|
||||
print(f"Usage time: {all_time.get('usage_hours_display', 'unknown')}")
|
||||
print(f"Peak hour: {all_time.get('peak_hour_slot', 'unknown')}")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Compute Codex Wrapped usage metrics.")
|
||||
parser.add_argument(
|
||||
"--json",
|
||||
action="store_true",
|
||||
help="Emit metrics as JSON for scripting.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--timezone",
|
||||
default=local_timezone_name(),
|
||||
help="IANA timezone name for local stats (defaults to system timezone).",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output",
|
||||
default=str(DEFAULT_OUTPUT_PATH),
|
||||
help=f"Path to save metrics JSON (default: {DEFAULT_OUTPUT_PATH}).",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
metrics = gather_metrics(args.timezone)
|
||||
|
||||
output_path = Path(args.output).expanduser()
|
||||
with output_path.open("w", encoding="utf-8") as handle:
|
||||
json.dump(metrics, handle, indent=2)
|
||||
handle.write("\n")
|
||||
|
||||
if args.json:
|
||||
json.dump(metrics, fp=sys.stdout, indent=2)
|
||||
print()
|
||||
return
|
||||
|
||||
print(f"Wrote stats to {output_path}")
|
||||
print_text(metrics)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,488 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
# Config knobs
|
||||
LABEL_MIN=5
|
||||
LABEL_MAX=15
|
||||
MIN_PANEL_WIDTH=80
|
||||
DEFAULT_PANEL_WIDTH=96
|
||||
FRAME_BUFFER=7
|
||||
|
||||
detect_term_width() {
|
||||
local cols=""
|
||||
if [[ -t 1 ]]; then
|
||||
if [[ -n "${COLUMNS:-}" ]]; then
|
||||
cols="$COLUMNS"
|
||||
fi
|
||||
if command -v tput >/dev/null 2>&1; then
|
||||
local tcols
|
||||
tcols=$(tput cols 2>/dev/null || true)
|
||||
if [[ "$tcols" =~ ^[0-9]+$ ]]; then
|
||||
cols="$tcols"
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
if [[ "$cols" =~ ^[0-9]+$ && "$cols" -gt 0 ]]; then
|
||||
echo "$cols"
|
||||
fi
|
||||
}
|
||||
|
||||
TERM_WIDTH=$(detect_term_width)
|
||||
if [[ -n "${WRAPPED_WIDTH:-}" && "${WRAPPED_WIDTH}" =~ ^[0-9]+$ ]]; then
|
||||
TERM_WIDTH="$WRAPPED_WIDTH"
|
||||
fi
|
||||
if [[ -z "$TERM_WIDTH" ]]; then
|
||||
TERM_WIDTH=$DEFAULT_PANEL_WIDTH
|
||||
fi
|
||||
DATA_LIMIT=$((TERM_WIDTH - 2))
|
||||
if (( DATA_LIMIT < MIN_PANEL_WIDTH )); then
|
||||
DATA_LIMIT=$MIN_PANEL_WIDTH
|
||||
fi
|
||||
|
||||
DECOR_CHUNK=".:*~*:._.:*~*:._.:*~*:._.:*~*:._.:*~*:."
|
||||
TITLE="OpenAI Codex Wrapped"
|
||||
SUBTITLE=""
|
||||
|
||||
# Defaults (overwritten by stats JSON)
|
||||
joined_line="Member since unknown"
|
||||
timezone_line="Local time: unknown"
|
||||
table_rows=()
|
||||
hero_line=""
|
||||
highlight_lines=()
|
||||
activity_lines=()
|
||||
column_rows=()
|
||||
column_head_left="Last 30 days"
|
||||
column_head_right="Last 7 days"
|
||||
|
||||
die() { echo "$*" >&2; exit 1; }
|
||||
|
||||
show_help() {
|
||||
cat <<'EOF'
|
||||
Usage: report.sh --stats-file <path> [--layout table|columns]
|
||||
|
||||
--stats-file Path to the metrics JSON produced by get_codex_stats.py.
|
||||
--layout Layout style (table or columns). Default: columns.
|
||||
-h, --help Show this message and exit.
|
||||
EOF
|
||||
}
|
||||
|
||||
strip_ansi_len() {
|
||||
local raw="$1"
|
||||
local clean
|
||||
clean=$(printf '%s' "$raw" | perl -pe 's/\e\[[0-9;]*[a-zA-Z]//g')
|
||||
printf '%s' "${#clean}"
|
||||
}
|
||||
|
||||
calc_panel_width() {
|
||||
local max_seen=0
|
||||
for entry in "$@"; do
|
||||
local span
|
||||
span=$(strip_ansi_len "$entry")
|
||||
(( span > max_seen )) && max_seen=$span
|
||||
done
|
||||
max_seen=$((max_seen + FRAME_BUFFER))
|
||||
if (( max_seen < MIN_PANEL_WIDTH )); then
|
||||
max_seen=$MIN_PANEL_WIDTH
|
||||
fi
|
||||
if (( max_seen > DATA_LIMIT )); then
|
||||
max_seen=$DATA_LIMIT
|
||||
fi
|
||||
echo "$max_seen"
|
||||
}
|
||||
|
||||
center_line() {
|
||||
local text="$1"
|
||||
local width="$2"
|
||||
local text_len
|
||||
text_len=$(strip_ansi_len "$text")
|
||||
local left_pad=$(( (width - text_len) / 2 ))
|
||||
local right_pad=$(( width - text_len - left_pad ))
|
||||
printf "│%*s%s%*s│\n" "$left_pad" "" "$text" "$right_pad" ""
|
||||
}
|
||||
|
||||
divider_line() {
|
||||
local width="$1"
|
||||
local left="$2"
|
||||
local mid="$3"
|
||||
local right="$4"
|
||||
local bar="$left"
|
||||
for ((i=0; i<width; i++)); do bar+="─"; done
|
||||
bar+="$right"
|
||||
printf '%s\n' "$bar"
|
||||
}
|
||||
|
||||
split_row() {
|
||||
local label="$1"
|
||||
local value="$2"
|
||||
local l_width="$3"
|
||||
local r_width="$4"
|
||||
|
||||
if [[ -z "$label" ]]; then
|
||||
label=$(printf "%-${l_width}s" "")
|
||||
elif (( ${#label} < LABEL_MIN )); then
|
||||
label=$(printf "%-${LABEL_MIN}s" "$label")
|
||||
elif (( ${#label} > l_width )); then
|
||||
label=$(echo "$label" | cut -c1-$((l_width-3)))...
|
||||
else
|
||||
label=$(printf "%-${l_width}s" "$label")
|
||||
fi
|
||||
|
||||
if (( ${#value} > r_width )); then
|
||||
value=$(echo "$value" | cut -c1-$((r_width-3)))...
|
||||
else
|
||||
value=$(printf "%-${r_width}s" "$value")
|
||||
fi
|
||||
|
||||
printf "│ %s │ %s │\n" "$label" "$value"
|
||||
}
|
||||
|
||||
left_line() {
|
||||
local text="$1"
|
||||
local width="$2"
|
||||
local content=" $text"
|
||||
content=$(fit_cell "$content" "$width")
|
||||
printf "│%s│\n" "$content"
|
||||
}
|
||||
|
||||
fit_cell() {
|
||||
local text="$1"
|
||||
local width="$2"
|
||||
local len
|
||||
len=$(strip_ansi_len "$text")
|
||||
if (( len > width )); then
|
||||
if (( width > 3 )); then
|
||||
text=$(echo "$text" | cut -c1-$((width-3)))...
|
||||
else
|
||||
text=$(echo "$text" | cut -c1-"$width")
|
||||
fi
|
||||
fi
|
||||
printf "%-${width}s" "$text"
|
||||
}
|
||||
|
||||
center_cell() {
|
||||
local text="$1"
|
||||
local width="$2"
|
||||
local text_len
|
||||
text_len=$(strip_ansi_len "$text")
|
||||
if (( text_len >= width )); then
|
||||
printf "%s" "$(fit_cell "$text" "$width")"
|
||||
return
|
||||
fi
|
||||
local left_pad=$(( (width - text_len) / 2 ))
|
||||
local right_pad=$(( width - text_len - left_pad ))
|
||||
printf "%*s%s%*s" "$left_pad" "" "$text" "$right_pad" ""
|
||||
}
|
||||
|
||||
trim_ws() {
|
||||
local s="$1"
|
||||
s="${s#"${s%%[![:space:]]*}"}"
|
||||
s="${s%"${s##*[![:space:]]}"}"
|
||||
printf "%s" "$s"
|
||||
}
|
||||
|
||||
load_stats() {
|
||||
local src="$1"
|
||||
[[ -r "$src" ]] || die "Unable to read stats file: $src"
|
||||
local tsv
|
||||
tsv=$(python3 - "$src" <<'PY'
|
||||
import json, sys
|
||||
from pathlib import Path
|
||||
|
||||
data = json.loads(Path(sys.argv[1]).read_text())
|
||||
windows = data.get("windows", {})
|
||||
all_time = windows.get("all_time", {})
|
||||
last_30 = windows.get("last_30_days", {})
|
||||
last_7 = windows.get("last_7_days", {})
|
||||
|
||||
def triple(key, default="unknown"):
|
||||
return (
|
||||
all_time.get(key, default),
|
||||
last_30.get(key, default),
|
||||
last_7.get(key, default),
|
||||
)
|
||||
|
||||
def compress_time(value: str) -> str:
|
||||
if not isinstance(value, str):
|
||||
return str(value)
|
||||
value = value.replace(" hours", "h").replace(" hour", "h")
|
||||
value = value.replace(" minutes", "m").replace(" minute", "m")
|
||||
return value
|
||||
|
||||
def compress_days(value: str) -> str:
|
||||
if not isinstance(value, str):
|
||||
return str(value)
|
||||
return value.replace(" days", "d").replace(" day", "d")
|
||||
|
||||
def compress_peak(value: str) -> str:
|
||||
if not isinstance(value, str):
|
||||
return str(value)
|
||||
return (
|
||||
value.replace(" bird", " b")
|
||||
.replace(" owl", " o")
|
||||
.replace(" day", " d")
|
||||
.replace(" eve", " e")
|
||||
)
|
||||
|
||||
def double_str(key, default="unknown", compress=None, collapse=False):
|
||||
_, m, w = triple(key, default)
|
||||
if compress:
|
||||
m = compress(m)
|
||||
w = compress(w)
|
||||
if collapse and m == w:
|
||||
return str(m)
|
||||
return f"30d {m} | 7d {w}"
|
||||
|
||||
rows = [
|
||||
("Window", "Last 30 days | Last 7 days"),
|
||||
("Sessions", double_str("sessions_display", "0")),
|
||||
("Assistant messages", double_str("assistant_messages_display", "0")),
|
||||
("Tokens", double_str("total_tokens_display", "0")),
|
||||
("Usage time", double_str("usage_hours_display", "0")),
|
||||
("Biggest day", double_str("biggest_day_display", "unknown")),
|
||||
("Top repo", double_str("top_repo_display", "unknown", collapse=True)),
|
||||
("Longest turn", double_str("longest_turn_display", "unknown")),
|
||||
]
|
||||
|
||||
column_metrics = [
|
||||
("Sessions", "sessions_display"),
|
||||
("Prompts", "user_messages_display"),
|
||||
("Tokens", "total_tokens_display"),
|
||||
("Usage time", "usage_hours_display"),
|
||||
("Biggest day", "biggest_day_display"),
|
||||
]
|
||||
stack_sections = [
|
||||
("Last 30 days", "last_30_days"),
|
||||
("Last 7 days", "last_7_days"),
|
||||
]
|
||||
|
||||
joined_display = data.get("joined_display", "unknown")
|
||||
timezone = data.get("timezone_abbrev") or data.get("timezone", "unknown")
|
||||
def compress_focus(value: str) -> str:
|
||||
return str(value)
|
||||
|
||||
print(f"joined_line\tMember since {joined_display}")
|
||||
print(f"timezone_line\tLocal time: {timezone}")
|
||||
for label, value in rows:
|
||||
print(f"row\t{label}\t{value}")
|
||||
|
||||
def format_stack(stat_key: str, value: str) -> str:
|
||||
return str(value)
|
||||
|
||||
def bar(active: int, total: int, width: int = 10) -> str:
|
||||
if total <= 0:
|
||||
return "." * width
|
||||
ratio = max(0.0, min(1.0, active / total))
|
||||
filled = int(round(ratio * width))
|
||||
return "#" * filled + "." * (width - filled)
|
||||
|
||||
def percent(active: int, total: int) -> str:
|
||||
if total <= 0:
|
||||
return "0%"
|
||||
return f"{int(round(active / total * 100))}%"
|
||||
|
||||
all_time = windows.get("all_time", {})
|
||||
last_30 = windows.get("last_30_days", {})
|
||||
last_7 = windows.get("last_7_days", {})
|
||||
|
||||
high_30_prompts = last_30.get("user_messages_display", "0")
|
||||
high_30_active = last_30.get("active_days_display", "0")
|
||||
high_peak = last_30.get("peak_hour_slot", "unknown")
|
||||
high_streak = last_30.get("usage_streak_display", "unknown")
|
||||
high_focus = all_time.get("usage_hours_display", "unknown")
|
||||
|
||||
print(f"hero_line\t{high_30_prompts} prompts in 30d — peak {high_peak}")
|
||||
print(f"highlight_line\t{joined_display} | {timezone}")
|
||||
print(f"highlight_line\tActive days: {high_30_active} in 30d | Streak: {high_streak}")
|
||||
print(f"highlight_line\tAll-time focus: {high_focus}")
|
||||
|
||||
act_30 = last_30.get("active_days", 0)
|
||||
act_7 = last_7.get("active_days", 0)
|
||||
bar_30 = bar(int(act_30), 30)
|
||||
bar_7 = bar(int(act_7), 7)
|
||||
print(f"activity_line\t30d activity: {bar_30} ({int(act_30)}/30)")
|
||||
print(f"activity_line\t7d activity: {bar_7} ({int(act_7)}/7)")
|
||||
|
||||
print("column_head\tLast 30 days\tLast 7 days")
|
||||
for label, stat_key in column_metrics:
|
||||
left = format_stack(stat_key, last_30.get(stat_key, "unknown"))
|
||||
right = format_stack(stat_key, last_7.get(stat_key, "unknown"))
|
||||
print(f"column_row\t{label}\t{left}\t{right}")
|
||||
for line in data.get("contrib_lines", []):
|
||||
print(f"contrib_line\t{line}")
|
||||
PY
|
||||
) || die "Failed to parse stats JSON: $src"
|
||||
|
||||
while IFS=$'\t' read -r key label value extra; do
|
||||
case "$key" in
|
||||
joined_line) joined_line="$label" ;;
|
||||
timezone_line) timezone_line="$label" ;;
|
||||
focus_line) ;;
|
||||
row) table_rows+=("${label}"$'\t'"${value}") ;;
|
||||
stack_section) stack_rows+=("SECTION"$'\t'"${label}") ;;
|
||||
stack_item) stack_rows+=("ITEM"$'\t'"${label}"$'\t'"${value}") ;;
|
||||
stack_blank) stack_rows+=("BLANK") ;;
|
||||
hero_line) hero_line="$label" ;;
|
||||
highlight_line) highlight_lines+=("$label") ;;
|
||||
activity_line) activity_lines+=("$label") ;;
|
||||
column_head)
|
||||
column_head_left="$label"
|
||||
column_head_right="$value"
|
||||
;;
|
||||
column_row)
|
||||
column_rows+=("${label}"$'\t'"${value}"$'\t'"${extra}") ;;
|
||||
esac
|
||||
done <<<"$tsv"
|
||||
}
|
||||
|
||||
render() {
|
||||
local stats_path="$1"
|
||||
local layout_override="${2:-}"
|
||||
load_stats "$stats_path"
|
||||
if ((${#contrib_lines[@]} == 0)); then
|
||||
contrib_lines=("Sun " "Mon " "Tue " "Wed " "Thu " "Fri " "Sat ")
|
||||
fi
|
||||
|
||||
holiday_lines=()
|
||||
|
||||
local layout="${WRAPPED_LAYOUT:-columns}"
|
||||
if [[ -n "$layout_override" ]]; then
|
||||
layout="$layout_override"
|
||||
elif [[ -n "${REPORT_LAYOUT:-}" ]]; then
|
||||
layout="${REPORT_LAYOUT}"
|
||||
fi
|
||||
|
||||
local width_input=("$DECOR_CHUNK" "$TITLE" "$SUBTITLE" "$hero_line")
|
||||
width_input+=("${highlight_lines[@]}")
|
||||
width_input+=("${activity_lines[@]}")
|
||||
if [[ "$layout" == "table" ]]; then
|
||||
for row in "${table_rows[@]}"; do
|
||||
IFS=$'\t' read -r label value <<<"$row"
|
||||
width_input+=("$label" "$value")
|
||||
done
|
||||
else
|
||||
width_input+=("$column_head_left" "$column_head_right")
|
||||
for row in "${column_rows[@]}"; do
|
||||
IFS=$'\t' read -r label value_left value_right <<<"$row"
|
||||
width_input+=("$label" "$value_left" "$value_right")
|
||||
done
|
||||
fi
|
||||
PANEL_WIDTH=$(calc_panel_width "${width_input[@]}")
|
||||
|
||||
local max_label=0
|
||||
for row in "${table_rows[@]}"; do
|
||||
IFS=$'\t' read -r label value <<<"$row"
|
||||
local span
|
||||
span=$(strip_ansi_len "$label")
|
||||
(( span > max_label )) && max_label=$span
|
||||
done
|
||||
local left_span=$max_label
|
||||
(( left_span < LABEL_MIN )) && left_span=$LABEL_MIN
|
||||
(( left_span > LABEL_MAX )) && left_span=$LABEL_MAX
|
||||
local right_span=$((PANEL_WIDTH - 5 - left_span))
|
||||
if (( right_span < 20 )); then
|
||||
right_span=20
|
||||
left_span=$((PANEL_WIDTH - 5 - right_span))
|
||||
(( left_span < LABEL_MIN )) && left_span=$LABEL_MIN
|
||||
fi
|
||||
local stack_rows_mode=0
|
||||
if (( right_span < 40 )); then
|
||||
stack_rows_mode=1
|
||||
fi
|
||||
|
||||
divider_line "$((PANEL_WIDTH))" "┌" "─" "┐"
|
||||
local deco_line="$DECOR_CHUNK"
|
||||
while (( $(strip_ansi_len "$deco_line") < PANEL_WIDTH )); do
|
||||
deco_line+="$DECOR_CHUNK"
|
||||
done
|
||||
deco_line=${deco_line:0:PANEL_WIDTH}
|
||||
printf "│%s│\n" "$deco_line"
|
||||
center_line "$TITLE" "$PANEL_WIDTH"
|
||||
if [[ -n "$SUBTITLE" ]]; then
|
||||
center_line "$SUBTITLE" "$PANEL_WIDTH"
|
||||
fi
|
||||
if [[ -n "$hero_line" ]]; then
|
||||
center_line "$hero_line" "$PANEL_WIDTH"
|
||||
fi
|
||||
printf "│%s│\n" "$deco_line"
|
||||
|
||||
divider_line "$((PANEL_WIDTH))" "├" "─" "┤"
|
||||
center_line "Highlights" "$PANEL_WIDTH"
|
||||
for ln in "${highlight_lines[@]}"; do
|
||||
left_line "$ln" "$PANEL_WIDTH"
|
||||
done
|
||||
for ln in "${activity_lines[@]}"; do
|
||||
left_line "$ln" "$PANEL_WIDTH"
|
||||
done
|
||||
center_line "" "$PANEL_WIDTH"
|
||||
|
||||
if [[ "$layout" == "table" ]]; then
|
||||
local horiz_left="├"; local horiz_mid="┬"; local horiz_right="┤"
|
||||
local lbar; printf -v lbar '%*s' $((left_span + 2)) ""; lbar=${lbar// /─}
|
||||
local rbar; printf -v rbar '%*s' $((right_span + 2)) ""; rbar=${rbar// /─}
|
||||
printf "%s%s%s%s%s\n" "$horiz_left" "$lbar" "$horiz_mid" "$rbar" "$horiz_right"
|
||||
|
||||
for row in "${table_rows[@]}"; do
|
||||
IFS=$'\t' read -r label value <<<"$row"
|
||||
if (( stack_rows_mode == 1 )) && [[ "$label" == "Window" ]]; then
|
||||
continue
|
||||
fi
|
||||
if (( stack_rows_mode == 1 )) && [[ "$value" == *"|"* ]]; then
|
||||
IFS='|' read -r part_a part_b part_c <<<"$value"
|
||||
part_a=$(trim_ws "$part_a")
|
||||
part_b=$(trim_ws "$part_b")
|
||||
part_c=$(trim_ws "$part_c")
|
||||
split_row "$label" "$part_a" "$left_span" "$right_span"
|
||||
if [[ -n "$part_b" ]]; then
|
||||
split_row "" "$part_b" "$left_span" "$right_span"
|
||||
fi
|
||||
if [[ -n "$part_c" ]]; then
|
||||
split_row "" "$part_c" "$left_span" "$right_span"
|
||||
fi
|
||||
else
|
||||
split_row "$label" "$value" "$left_span" "$right_span"
|
||||
fi
|
||||
done
|
||||
|
||||
local tail_left="└"; local tail_mid="┴"; local tail_right="┘"
|
||||
printf "%s%s%s%s%s\n" "$tail_left" "$lbar" "$tail_mid" "$rbar" "$tail_right"
|
||||
else
|
||||
divider_line "$((PANEL_WIDTH))" "├" "─" "┤"
|
||||
local inner_width=$((PANEL_WIDTH))
|
||||
local gap=" │ "
|
||||
local gap_len=3
|
||||
local left_width=$(( (inner_width - gap_len) / 2 ))
|
||||
local right_width=$(( inner_width - gap_len - left_width ))
|
||||
local head_left
|
||||
local head_right
|
||||
head_left=$(center_cell "$column_head_left" "$left_width")
|
||||
head_right=$(center_cell "$column_head_right" "$right_width")
|
||||
printf "│%s%s%s│\n" "$head_left" "$gap" "$head_right"
|
||||
|
||||
for row in "${column_rows[@]}"; do
|
||||
IFS=$'\t' read -r label value_left value_right <<<"$row"
|
||||
local ltext=" ${label}: ${value_left}"
|
||||
local rtext=" ${label}: ${value_right}"
|
||||
ltext=$(fit_cell "$ltext" "$left_width")
|
||||
rtext=$(fit_cell "$rtext" "$right_width")
|
||||
printf "│%s%s%s│\n" "$ltext" "$gap" "$rtext"
|
||||
done
|
||||
|
||||
divider_line "$((PANEL_WIDTH))" "└" "─" "┘"
|
||||
fi
|
||||
}
|
||||
|
||||
main() {
|
||||
local stats_path=""
|
||||
local layout_arg=""
|
||||
while [[ $# -gt 0 ]]; do
|
||||
case "$1" in
|
||||
--stats-file) stats_path="$2"; shift 2 ;;
|
||||
--layout) layout_arg="$2"; shift 2 ;;
|
||||
-h|--help) show_help; exit 0 ;;
|
||||
*) show_help; exit 1 ;;
|
||||
esac
|
||||
done
|
||||
[[ -n "$stats_path" ]] || die "--stats-file is required."
|
||||
render "$stats_path" "$layout_arg"
|
||||
}
|
||||
|
||||
main "$@"
|
||||
Reference in New Issue
Block a user