diff --git a/skills/.experimental/codex-readiness-integration-test/LICENSE.txt b/skills/.experimental/codex-readiness-integration-test/LICENSE.txt deleted file mode 100644 index d645695..0000000 --- a/skills/.experimental/codex-readiness-integration-test/LICENSE.txt +++ /dev/null @@ -1,202 +0,0 @@ - - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright [yyyy] [name of copyright owner] - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. diff --git a/skills/.experimental/codex-readiness-integration-test/SKILL.md b/skills/.experimental/codex-readiness-integration-test/SKILL.md deleted file mode 100644 index 3e82d6f..0000000 --- a/skills/.experimental/codex-readiness-integration-test/SKILL.md +++ /dev/null @@ -1,77 +0,0 @@ ---- -name: codex-readiness-integration-test -description: Run the Codex Readiness integration test. Use when you need an end-to-end agentic loop with build/test scoring. -metadata: - short-description: Run Codex Readiness integration test ---- - -# LLM Codex Readiness Integration Test - -This skill runs a multi-stage integration test to validate agentic execution quality. It always runs in execute mode (no read-only mode). - -## Outputs - -Each run writes to `.codex-readiness-integration-test//` and updates `.codex-readiness-integration-test/latest.json`. - -New outputs per run: -- `agentic_summary.json` and `logs/agentic.log` (agentic loop execution) -- `llm_results.json` (automatic LLM evaluation) -- `summary.txt` (human-readable summary) - -## Pre-conditions (Required) - -- Authenticate with the Codex CLI using the repo-local HOME before running the test. - Run these in your own terminal (not via the integration test): - HOME=$PWD/.codex-home XDG_CACHE_HOME=$PWD/.codex-home/.cache codex login - HOME=$PWD/.codex-home XDG_CACHE_HOME=$PWD/.codex-home/.cache codex login status -- The integration test creates {repo_root}/.codex-home and {repo_root}/.codex-home/.cache/codex as its first step. - -## Workflow - -0) Ask the user how to source the task. - - Offer two explicit options: (a) user provides a custom task/prompt, or (b) auto-generate a task. - - Do not run the entry point until the user chooses one option. -1) Generate or load `{out_dir}/prompt.pending.json`. - - Use the integration test's expected prompt path, not `prompt.json` at the repo root. - - With the default out dir, this path is `.codex-readiness-integration-test/prompt.pending.json`. - - If `--seed-task` is provided, it is used as the starting task. - - If not provided, generate a task with `skills/codex-readiness-integration-test/references/generate_prompt.md` and save the JSON to `{out_dir}/prompt.pending.json`. - - The user must approve the prompt before execution (no auto-approve mode). Make sure to output a summary of the prompt when asking the user to approve. -2) Execute the agentic loop via Codex CLI (uses `AGENTS.md` and `change_prompt`). -3) Run build/test commands from the prompt plan via `skills/codex-readiness-integration-test/scripts/run_plan.py`. -4) Collect evidence (`evidence.json`), deterministic checks, and run automatic LLM evals via Codex CLI. -5) Score and write the report + summary output. - -## Configuration - -Optional fields in `{out_dir}/prompt.pending.json`: -- `agentic_loop`: configure Codex CLI invocation for the agentic loop. -- `llm_eval`: configure Codex CLI invocation for automatic evals. - -If these fields are omitted, defaults are used. - -## Requirements - -- The LLM evaluator must fail if evidence mentions the phrase `Context compaction enabled`. -- Use qualitative context-usage evaluation (no strict thresholds). - - -## What this test covers well - -- Runs Codex CLI against the real repo root, producing real filesystem edits and git diffs. -- Executes the approved change prompt and then runs the build/test plan in-repo. -- Captures evidence, deterministic checks, and LLM eval artifacts for review. - -## What this test does not represent - -- The agentic loop may use non-default flags (e.g., bypass approvals/sandbox), so interactive guardrails differ. -- Uses a dedicated HOME (`.codex-home`), which can change auth/config/cache vs normal CLI use. -- Auto-generated prompts and one-shot execution do not simulate interactive guidance. -- MCP servers/tools are not exercised unless explicitly configured. - -## Notes - -- The prompts in `skills/codex-readiness-integration-test/references/` expect strict JSON. -- Use `skills/codex-readiness-integration-test/references/json_fix.md` to repair invalid JSON output. -- This skill calls the `codex` CLI. Ensure it is installed and available on PATH, or override the command in `{out_dir}/prompt.pending.json`. -- If the agentic loop detects sandbox-blocked tool access, it now writes `requires_escalation: true` to `{run_dir}/agentic_summary.json` and exits with code `3`. Re-run the integration test with escalated permissions in that case. diff --git a/skills/.experimental/codex-readiness-integration-test/references/agentic_loop_eval.md b/skills/.experimental/codex-readiness-integration-test/references/agentic_loop_eval.md deleted file mode 100644 index ecb9f9a..0000000 --- a/skills/.experimental/codex-readiness-integration-test/references/agentic_loop_eval.md +++ /dev/null @@ -1,25 +0,0 @@ -You are evaluating whether the agentic loop executed successfully. You will be given a JSON input payload with: -- prompt: prompt.json content -- evidence: evidence.json content (includes agents_md and logs_index) -- git_diff -- execution_summary -- agentic_summary - -Hard fails: -- If any evidence or logs mention the phrase "Context compaction enabled", status MUST be FAIL. - -Prefer PASS only if agentic_summary.status is PASS and exit_code is 0. - -Return strict JSON with this schema: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...","quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- Use WARN for partial success (e.g., agentic_summary exists but status is FAIL). -- Reference concrete evidence paths in evidence_quotes. -- Output JSON only. No markdown. diff --git a/skills/.experimental/codex-readiness-integration-test/references/change_quality.md b/skills/.experimental/codex-readiness-integration-test/references/change_quality.md deleted file mode 100644 index 144ef0b..0000000 --- a/skills/.experimental/codex-readiness-integration-test/references/change_quality.md +++ /dev/null @@ -1,28 +0,0 @@ -Evaluate code change quality relative to prompt.json and the actual diff. You will be given a JSON input payload with: -- prompt: prompt.json content -- evidence: evidence.json content -- git_diff -- execution_summary -- agentic_summary - -Score quality across these dimensions: -- correctness vs change_prompt -- context usage (qualitative) -- maintainability/readability -- risk/regression assessment -- builds/tests passing (from execution_summary) - -Return strict JSON with this schema: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...","quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- If builds/tests FAIL, status should be FAIL unless the prompt explicitly allows it. -- Use WARN for partial correctness or limited context usage. -- Call out mismatches between change_prompt and git_diff. -- Output JSON only. No markdown. diff --git a/skills/.experimental/codex-readiness-integration-test/references/checks.json b/skills/.experimental/codex-readiness-integration-test/references/checks.json deleted file mode 100644 index 659ea45..0000000 --- a/skills/.experimental/codex-readiness-integration-test/references/checks.json +++ /dev/null @@ -1,129 +0,0 @@ -{ - "schema_version": "1.0", - "checks": [ - { - "id": "agentic_run_success", - "title": "P0: Agentic loop execution", - "description": "Agentic loop ran and exited with code 0.", - "priority": 0, - "weight": null, - "type": "DETERMINISTIC", - "scope": "run_dir", - "execute_required": true, - "evaluator_prompt_id": null, - "deterministic_rule_id": "agentic_run_success", - "deterministic_rule_params": { - "path": "agentic_summary.json" - }, - "enabled_by_default": true - }, - { - "id": "exec_plan_before_code_changes", - "title": "P0: Planning signal before code changes", - "description": "A planning signal (for example: update_plan or \"Plan:\") appears in agentic.log before any non-doc, non-.codex file changes.", - "priority": 0, - "weight": null, - "type": "DETERMINISTIC", - "scope": "run_dir", - "execute_required": true, - "evaluator_prompt_id": null, - "deterministic_rule_id": "exec_plan_before_code_changes", - "deterministic_rule_params": { - "prompt_path": "prompt.json", - "agentic_log_path": "logs/agentic.log" - }, - "enabled_by_default": true - }, - { - "id": "verification_after_code_changes", - "title": "P0: Verification after code changes", - "description": "A build, test, or lint verification command appears in agentic.log after the first non-doc, non-.codex file change.", - "priority": 0, - "weight": null, - "type": "DETERMINISTIC", - "scope": "run_dir", - "execute_required": true, - "evaluator_prompt_id": null, - "deterministic_rule_id": "verification_after_code_changes", - "deterministic_rule_params": { - "prompt_path": "prompt.json", - "agentic_log_path": "logs/agentic.log" - }, - "enabled_by_default": true - }, - { - "id": "repo_root_only_changes", - "title": "P1: Repo-root-only changes", - "description": "Git diff paths resolve under repo root.", - "priority": 1, - "weight": null, - "type": "DETERMINISTIC", - "scope": "run_dir", - "execute_required": true, - "evaluator_prompt_id": null, - "deterministic_rule_id": "repo_root_only_changes", - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "execution_summary_status", - "title": "P0: Build/test execution status", - "description": "Execution summary is present and overall status is PASS or WARN.", - "priority": 0, - "weight": null, - "type": "DETERMINISTIC", - "scope": "run_dir", - "execute_required": true, - "evaluator_prompt_id": null, - "deterministic_rule_id": "execution_summary_status", - "deterministic_rule_params": { - "summary_path": "execution_summary.json" - }, - "enabled_by_default": true - }, - { - "id": "execution_logs_no_errors", - "title": "P2: Execution logs show no obvious errors", - "description": "Execution logs do not contain obvious error markers.", - "priority": 2, - "weight": null, - "type": "DETERMINISTIC", - "scope": "run_dir", - "execute_required": true, - "evaluator_prompt_id": null, - "deterministic_rule_id": "execution_logs_no_errors", - "deterministic_rule_params": { - "logs_dir": "logs" - }, - "enabled_by_default": true - }, - { - "id": "agentic_loop_eval", - "title": "P0: Agentic loop success", - "description": "LLM evaluation of agentic loop success, AGENTS reference, and compaction avoidance.", - "priority": 0, - "weight": null, - "type": "LLM", - "scope": "run_dir", - "execute_required": true, - "evaluator_prompt_id": "agentic_loop_eval", - "deterministic_rule_id": null, - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "change_quality_eval", - "title": "P0: Change quality evaluation", - "description": "LLM evaluation of correctness, context usage, maintainability, and risk.", - "priority": 0, - "weight": null, - "type": "LLM", - "scope": "run_dir", - "execute_required": true, - "evaluator_prompt_id": "change_quality", - "deterministic_rule_id": null, - "deterministic_rule_params": {}, - "enabled_by_default": true - } - ] -} diff --git a/skills/.experimental/codex-readiness-integration-test/references/generate_prompt.md b/skills/.experimental/codex-readiness-integration-test/references/generate_prompt.md deleted file mode 100644 index c80f096..0000000 --- a/skills/.experimental/codex-readiness-integration-test/references/generate_prompt.md +++ /dev/null @@ -1,38 +0,0 @@ -You are generating a change prompt for an integration test. First, check whether AGENTS.md exists at the repo root and incorporate any build/test guidance from it. If AGENTS.md is missing, note that in your rationale field. - -You must generate a plan-worthy task. Plan-worthy means the change should be complex enough to justify a PLANS.md-style ExecPlan: at least two repository files are likely to be edited, validation requires tests or explicit verification steps, and there is non-trivial sequencing or reasoning involved. Keep scope small and realistic for this repo, but not trivial. Examples of plan-worthy prompts: refactor a subsystem to improve clarity or reduce duplication, or implement a sizeable feature slice that touches multiple layers (API, data model, and UI). - -Return strict JSON with this schema: -{ - "seed_task": "string or null", - "prompt_origin": "auto", - "change_prompt": "string", - "acceptance_criteria": ["..."], - "build_test_plan": [ -{"label": "build", "cmd": "..."}, -{"label": "test", "cmd": "..."} - ], - "scoring_focus": ["correctness", "context_usage", "builds_tests_pass", "maintainability", "risk"], - "agentic_loop": { -"cmd": "codex", -"args": ["exec", "--full-auto", "-C", "{repo_root}", "{change_prompt}"], -"timeout_seconds": 1800 - }, - "llm_eval": { -"cmd": "codex", -"args": ["exec", "--output-schema", "{eval_schema_path}", "--output-last-message", "{eval_output_path}", "--color", "never", "--sandbox", "read-only", "-C", "{repo_root}", "-"], -"timeout_seconds": 600 - }, - "rationale": "string" -} - -Notes: -- agentic_loop and llm_eval are optional; defaults will be applied if omitted. - -Rules: -- If a seed task is provided, use it as the primary direction and set seed_task accordingly. -- change_prompt must be actionable and scoped to a small but real code change. -- change_prompt must not include meta-instructions like “follow AGENTS.md”, “run tests”, or other process guidance; keep those in build_test_plan or rationale. -- acceptance_criteria must be testable and concrete. -- build_test_plan must include at least one build or test command if such commands are documented. -- Output JSON only. No markdown. diff --git a/skills/.experimental/codex-readiness-integration-test/references/json_fix.md b/skills/.experimental/codex-readiness-integration-test/references/json_fix.md deleted file mode 100644 index d963315..0000000 --- a/skills/.experimental/codex-readiness-integration-test/references/json_fix.md +++ /dev/null @@ -1,18 +0,0 @@ -You are fixing invalid JSON from a prior evaluator. - -Return ONLY valid JSON that matches this schema exactly: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...", "quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- Do not include any extra keys. -- Do not include markdown, commentary, or code fences. -- If the original content lacks evidence, keep evidence_quotes empty. - -Invalid output to fix: -{{RAW_OUTPUT}} diff --git a/skills/.experimental/codex-readiness-integration-test/references/llm_eval_schema.json b/skills/.experimental/codex-readiness-integration-test/references/llm_eval_schema.json deleted file mode 100644 index 2767309..0000000 --- a/skills/.experimental/codex-readiness-integration-test/references/llm_eval_schema.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "$schema": "http://json-schema.org/draft-07/schema#", - "type": "object", - "additionalProperties": false, - "properties": { - "status": { - "type": "string", - "enum": ["PASS", "WARN", "FAIL", "NOT_RUN"] - }, - "rationale": {"type": "string"}, - "evidence_quotes": { - "type": "array", - "items": { - "type": "object", - "additionalProperties": false, - "properties": { - "path": {"type": "string"}, - "quote": {"type": "string"} - }, - "required": ["path", "quote"] - } - }, - "recommendations": { - "type": "array", - "items": {"type": "string"} - }, - "confidence": {"type": "number"} - }, - "required": ["status", "rationale", "evidence_quotes", "recommendations", "confidence"] -} diff --git a/skills/.experimental/codex-readiness-integration-test/scripts/collect_evidence.py b/skills/.experimental/codex-readiness-integration-test/scripts/collect_evidence.py deleted file mode 100644 index 1d91281..0000000 --- a/skills/.experimental/codex-readiness-integration-test/scripts/collect_evidence.py +++ /dev/null @@ -1,191 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -import subprocess -from datetime import datetime, timezone -from pathlib import Path - -SKIP_DIRS = { - ".git", - ".codex-readiness-integration-test", - "node_modules", - "dist", - "build", - ".venv", - "venv", - "__pycache__", -} - - -def now_iso() -> str: - return datetime.now(timezone.utc).isoformat() - - -def read_text(path: Path) -> str: - try: - return path.read_text(encoding="utf-8") - except Exception: - try: - return path.read_text(encoding="utf-8", errors="ignore") - except Exception: - return "" - - -def extract_snippet(text: str, max_chars: int) -> str: - if len(text) <= max_chars: - return text - return text[: max_chars - 3] + "..." - - -def run_cmd(cmd: list[str]) -> str: - try: - output = subprocess.check_output(cmd, stderr=subprocess.STDOUT, text=True) - return output.strip() - except Exception as exc: - return f" {exc}" - - -def run_cmd_allow_failure(cmd: list[str]) -> str: - try: - result = subprocess.run( - cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, check=False - ) - return result.stdout.strip() - except Exception as exc: - return f" {exc}" - - -def should_include_untracked(path: Path) -> bool: - if path.name == ".DS_Store": - return False - return all(not part.startswith(".codex") for part in path.parts) - - -def build_untracked_diff() -> str: - raw = run_cmd(["git", "ls-files", "--others", "--exclude-standard"]) - if raw.startswith(""): - return "" - diffs = [] - for line in raw.splitlines(): - line = line.strip() - if not line: - continue - path = Path(line) - if not should_include_untracked(path): - continue - diff = run_cmd_allow_failure(["git", "diff", "--no-index", "/dev/null", line]) - if diff and not diff.startswith(""): - diffs.append(diff) - return "\n".join(diffs) - - -def load_json_if_exists(path: Path) -> dict | None: - if not path.exists(): - return None - try: - return json.loads(path.read_text(encoding="utf-8")) - except Exception: - return None - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = json.loads(latest_path.read_text(encoding="utf-8")) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "prompt.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def main() -> int: - parser = argparse.ArgumentParser(description="Collect evidence for integration test.") - parser.add_argument( - "--out-dir", default=".codex-readiness-integration-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument("--max-snippet-chars", type=int, default=2000) - args = parser.parse_args() - - cwd = Path.cwd() - base_dir = Path(args.out_dir) - run_dir = resolve_run_dir(base_dir, args.run_dir) - run_dir.mkdir(parents=True, exist_ok=True) - - agents_path = cwd / "AGENTS.md" - agents_text = read_text(agents_path) if agents_path.exists() else "" - - prompt_path = run_dir / "prompt.json" - plan_path = run_dir / "plan.json" - execution_summary_path = run_dir / "execution_summary.json" - agentic_summary_path = run_dir / "agentic_summary.json" - agentic_log_path = run_dir / "logs" / "agentic.log" - - logs_dir = run_dir / "logs" - logs_index = [str(path) for path in sorted(logs_dir.glob("*.log"))] if logs_dir.exists() else [] - - tracked_diff = run_cmd(["git", "diff"]) - untracked_diff = build_untracked_diff() - if untracked_diff: - if tracked_diff: - combined_diff = f"{tracked_diff}\n{untracked_diff}" - else: - combined_diff = untracked_diff - else: - combined_diff = tracked_diff - - evidence = { - "timestamp": now_iso(), - "repo_root": str(cwd), - "agents_md": { - "path": str(agents_path), - "exists": agents_path.exists(), - "snippet": extract_snippet(agents_text, args.max_snippet_chars), - }, - "prompt_json": { - "path": str(prompt_path), - "exists": prompt_path.exists(), - "content": load_json_if_exists(prompt_path), - }, - "plan_json": { - "path": str(plan_path), - "exists": plan_path.exists(), - "content": load_json_if_exists(plan_path), - }, - "execution_summary": { - "path": str(execution_summary_path), - "exists": execution_summary_path.exists(), - "content": load_json_if_exists(execution_summary_path), - }, - "agentic_summary": { - "path": str(agentic_summary_path), - "exists": agentic_summary_path.exists(), - "content": load_json_if_exists(agentic_summary_path), - }, - "agentic_log": { - "path": str(agentic_log_path), - "exists": agentic_log_path.exists(), - "snippet": extract_snippet(read_text(agentic_log_path), args.max_snippet_chars) - if agentic_log_path.exists() - else "", - }, - "logs_index": logs_index, - "git_status": run_cmd(["git", "status", "--porcelain"]), - "git_diff": combined_diff, - } - - output_path = run_dir / "evidence.json" - output_path.write_text(json.dumps(evidence, indent=2), encoding="utf-8") - print(str(output_path)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/skills/.experimental/codex-readiness-integration-test/scripts/deterministic_rules.py b/skills/.experimental/codex-readiness-integration-test/scripts/deterministic_rules.py deleted file mode 100644 index f71b5e9..0000000 --- a/skills/.experimental/codex-readiness-integration-test/scripts/deterministic_rules.py +++ /dev/null @@ -1,632 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -import re -import shlex -import subprocess -from pathlib import Path -from typing import Any - -VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"} -ANSI_ESCAPE_PATTERN = re.compile(r"\x1B(?:[@-Z\\-_]|\[[0-?]*[ -/]*[@-~])") -ERROR_MARKERS = [ - "error:", - "failed", - "exception", - "traceback", - "segmentation fault", -] -PLANNING_SIGNAL_PATTERNS = [ - re.compile(r"\bupdate_plan\b", re.IGNORECASE), - re.compile(r"^\s*plan\s*[:\-]", re.IGNORECASE), - re.compile(r"^\s*plan\s+update\b", re.IGNORECASE), - re.compile(r"^\s*\*\*planning\b", re.IGNORECASE), - re.compile(r"^\s*steps?\s*[:\-]", re.IGNORECASE), - re.compile(r"^\s*approach\s*[:\-]", re.IGNORECASE), - re.compile(r"\bhere(?:'s| is)\s+(?:the\s+)?plan\b", re.IGNORECASE), -] -COMMAND_LINE_PATTERNS = [ - re.compile(r"^\s*\$\s+(.+)$"), - re.compile(r"^\s*!\s*(.+)$"), - re.compile(r"^\s*running(?: command)?\s*:\s+(.+)$", re.IGNORECASE), - re.compile(r"^\s*cmd\s*:\s+(.+)$", re.IGNORECASE), -] -# Match shell "-lc ''" forms even when prefixed by a path like /bin/zsh. -SHELL_LC_PATTERN = re.compile(r"(?:^|\s)-lc\s+(?P['\"])(?P.+?)(?P=quote)") -VERIFICATION_KEYWORDS = [ - " test", - "pytest", - "npm test", - "pnpm test", - "yarn test", - "node --test", - "go test", - "cargo test", - "mvn test", - "gradle test", - "./gradlew test", - "lint", - "eslint", - "ruff", - "flake8", - "black --check", - "prettier --check", - "typecheck", - "tsc", - " build", - "compile", - "mvn package", - "gradle build", - "./gradlew build", - "go build", - "cargo build", - "make build", - "make test", - "make lint", - "make verify", - "verify", -] - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = load_json(latest_path) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "prompt.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def normalize_priority(value) -> int: - try: - priority = int(value) - except (TypeError, ValueError): - return 3 - return priority if priority in {0, 1, 2, 3} else 3 - - -def sort_checks_by_priority(checks: list[dict]) -> list[dict]: - return sorted( - checks, - key=lambda check: ( - normalize_priority(check.get("priority")), - check.get("id", ""), - ), - ) - - -def run_cmd(cmd: list[str]) -> str: - try: - output = subprocess.check_output(cmd, stderr=subprocess.STDOUT, text=True) - return output.strip() - except Exception as exc: - return f" {exc}" - - -def result( - status: str, - rationale: str, - evidence: list[dict] | None = None, - recs: list[str] | None = None, - confidence: float = 0.7, -) -> dict: - return { - "status": status, - "rationale": rationale, - "evidence_quotes": evidence or [], - "recommendations": recs or [], - "confidence": confidence, - } - - -def check_prompt_json_present(run_dir: Path, params: dict[str, Any]) -> dict: - path = run_dir / params.get("path", "prompt.json") - if path.exists(): - return result("PASS", "prompt.json exists.", [{"path": str(path), "quote": "present"}]) - return result( - "FAIL", - "prompt.json is missing.", - recs=["Generate prompt.json before running deterministic checks."], - ) - - -def check_build_test_plan_present(run_dir: Path, params: dict[str, Any]) -> dict: - path = run_dir / params.get("path", "prompt.json") - if not path.exists(): - return result("FAIL", "prompt.json is missing.") - try: - prompt = load_json(path) - except Exception: - return result("FAIL", "prompt.json could not be parsed.") - plan = prompt.get("build_test_plan") - if not isinstance(plan, list) or not plan: - return result( - "FAIL", - "build_test_plan is missing or empty.", - [{"path": str(path), "quote": "build_test_plan"}], - ) - commands = [entry.get("cmd") for entry in plan if isinstance(entry, dict)] - commands = [cmd for cmd in commands if isinstance(cmd, str) and cmd.strip()] - if not commands: - return result( - "WARN", - "build_test_plan has no valid commands.", - [{"path": str(path), "quote": "build_test_plan"}], - ) - return result( - "PASS", - "build_test_plan contains commands.", - [{"path": str(path), "quote": "build_test_plan"}], - ) - - -def check_execution_summary_status(run_dir: Path, params: dict[str, Any]) -> dict: - summary_path = run_dir / params.get("summary_path", "execution_summary.json") - if not summary_path.exists(): - return result("FAIL", "execution_summary.json is missing.") - try: - summary = load_json(summary_path) - except Exception: - return result("FAIL", "execution_summary.json could not be parsed.") - status = summary.get("overall_status") - if status in {"PASS", "WARN"}: - return result( - "PASS" if status == "PASS" else "WARN", - f"execution summary status is {status}.", - [{"path": str(summary_path), "quote": status}], - ) - return result( - "FAIL", - f"execution summary status is {status}.", - [{"path": str(summary_path), "quote": str(status)}], - ) - - -def check_execution_logs_no_errors(run_dir: Path, params: dict[str, Any]) -> dict: - logs_dir = run_dir / params.get("logs_dir", "logs") - if not logs_dir.exists(): - return result("WARN", "logs directory is missing.") - matches = [] - log_paths = sorted(logs_dir.glob("[0-9][0-9]-*.log")) - if not log_paths: - log_paths = sorted(logs_dir.glob("*.log")) - for log_path in log_paths: - try: - content = log_path.read_text(encoding="utf-8", errors="ignore") - except Exception: - continue - lower = content.lower() - for marker in ERROR_MARKERS: - if marker in lower: - snippet_index = lower.find(marker) - snippet = content[snippet_index : snippet_index + 200].splitlines()[0] - matches.append({"path": str(log_path), "quote": snippet}) - break - if matches: - return result( - "WARN", - "Execution logs contain error markers.", - matches, - ["Review build/test logs for failures."], - ) - return result("PASS", "No obvious error markers found in execution logs.") - - -def parse_agentic_file_update_events(log_text: str) -> list[dict]: - events = [] - lines = log_text.splitlines() - for idx, line in enumerate(lines): - if line.strip() != "file update": - continue - next_line = lines[idx + 1] if idx + 1 < len(lines) else "" - next_line = next_line.strip() - if not next_line: - continue - parts = next_line.split(" ", 1) - path = parts[1].strip() if len(parts) == 2 else parts[0].strip() - line_index = idx + 1 if idx + 1 < len(lines) else idx - events.append({"path": path, "line_index": line_index}) - return events - - -def resolve_repo_root() -> Path | None: - repo_root_raw = run_cmd(["git", "rev-parse", "--show-toplevel"]) - if repo_root_raw.startswith(""): - return None - return Path(repo_root_raw).resolve() - - -def is_doc_path(path: str) -> bool: - return path.lower().endswith(".md") - - -def strip_ansi(text: str) -> str: - return ANSI_ESCAPE_PATTERN.sub("", text) - - -def command_binary(cmd: str) -> str: - try: - parts = shlex.split(cmd) - except ValueError: - parts = cmd.split() - if not parts: - return "" - return Path(parts[0]).name.lower() - - -def is_codex_invocation(cmd: str) -> bool: - binary = command_binary(cmd) - return binary in {"codex", "codex.exe"} - - -def extract_command_events(log_text: str) -> list[dict[str, Any]]: - events: list[dict[str, Any]] = [] - for idx, raw_line in enumerate(log_text.splitlines()): - clean_line = strip_ansi(raw_line).strip() - if not clean_line: - continue - - # First handle common "command-like" prefixes such as "$ npm test". - for pattern in COMMAND_LINE_PATTERNS: - match = pattern.match(clean_line) - if not match: - continue - cmd = match.group(1).strip() - if not cmd: - continue - if is_codex_invocation(cmd): - # Ignore runner-level codex invocations; they are not verification steps. - continue - events.append( - { - "line_index": idx, - "cmd": cmd, - "raw_line": clean_line, - } - ) - break - - else: - # Fall back to extracting the inner command from shell "-lc" invocations - # such as: /bin/zsh -lc 'npm test' ... succeeded in 64ms - lc_match = SHELL_LC_PATTERN.search(clean_line) - if not lc_match: - continue - cmd = lc_match.group("cmd").strip() - if not cmd: - continue - if is_codex_invocation(cmd): - continue - events.append( - { - "line_index": idx, - "cmd": cmd, - "raw_line": clean_line, - } - ) - return events - - -def prompt_command_candidates(prompt: dict[str, Any]) -> list[str]: - candidates: list[str] = [] - plan = prompt.get("build_test_plan") - if not isinstance(plan, list): - return candidates - for entry in plan: - if isinstance(entry, dict): - cmd = entry.get("cmd") - elif isinstance(entry, str): - cmd = entry - else: - cmd = None - if isinstance(cmd, str) and cmd.strip(): - candidates.append(cmd.strip().lower()) - return candidates - - -def is_verification_command(cmd: str, prompt_cmds: list[str]) -> bool: - lower = cmd.lower() - if any(keyword in lower for keyword in VERIFICATION_KEYWORDS): - return True - return any(prompt_cmd and prompt_cmd in lower for prompt_cmd in prompt_cmds) - - -def first_code_change_event(events: list[dict[str, Any]]) -> dict[str, Any] | None: - code_events: list[dict[str, Any]] = [] - for event in events: - path = str(event.get("path", "")) - try: - line_index = int(event.get("line_index")) - except Exception: - continue - if not path: - continue - if path.startswith(".codex/"): - continue - if is_doc_path(path): - continue - code_events.append({"path": path, "line_index": line_index}) - if not code_events: - return None - return min(code_events, key=lambda item: item["line_index"]) - - -def file_update_line_indexes(events: list[dict[str, Any]]) -> set[int]: - indexes: set[int] = set() - for event in events: - try: - indexes.add(int(event.get("line_index"))) - except Exception: - continue - return indexes - - -def find_first_planning_signal_index(lines: list[str], skip_indexes: set[int]) -> int | None: - for idx, line in enumerate(lines): - if idx in skip_indexes: - continue - for pattern in PLANNING_SIGNAL_PATTERNS: - if pattern.search(line): - return idx - return None - - -def check_exec_plan_before_code_changes(run_dir: Path, params: dict[str, Any]) -> dict: - prompt_path = run_dir / params.get("prompt_path", "prompt.json") - if not prompt_path.exists(): - return result("FAIL", "prompt.json is missing.") - try: - _prompt = load_json(prompt_path) - except Exception: - return result("FAIL", "prompt.json could not be parsed.") - - log_path = run_dir / params.get("agentic_log_path", "logs/agentic.log") - if not log_path.exists(): - return result("FAIL", "agentic.log is missing.") - - log_text = log_path.read_text(encoding="utf-8", errors="ignore") - lines = log_text.splitlines() - events = parse_agentic_file_update_events(log_text) - if not events: - return result("WARN", "No file update entries found in agentic.log.") - - code_event = first_code_change_event(events) - if code_event is None: - return result( - "WARN", - "No non-doc, non-.codex file changes found; ordering not evaluated.", - [{"path": str(log_path), "quote": "no code updates"}], - ) - code_line = int(code_event["line_index"]) - - skip_indexes = file_update_line_indexes(events) - plan_line = find_first_planning_signal_index(lines, skip_indexes) - if plan_line is None: - return result( - "FAIL", - "No planning signal detected in agentic.log before code changes.", - [ - {"path": str(log_path), "quote": "planning signal not detected"}, - { - "path": str(log_path), - "quote": lines[code_line] if code_line < len(lines) else "", - }, - ], - ["Emit a short plan (for example: 'Plan:' or use update_plan) before code edits."], - ) - - if plan_line <= code_line: - return result( - "PASS", - "Planning signal appears before code changes.", - [ - { - "path": str(log_path), - "quote": lines[plan_line] if plan_line < len(lines) else "", - }, - { - "path": str(log_path), - "quote": lines[code_line] if code_line < len(lines) else "", - }, - ], - ) - - return result( - "FAIL", - "Planning signal appears after code changes.", - [ - {"path": str(log_path), "quote": lines[code_line] if code_line < len(lines) else ""}, - {"path": str(log_path), "quote": lines[plan_line] if plan_line < len(lines) else ""}, - ], - ["Emit a short plan (for example: 'Plan:' or use update_plan) before code edits."], - ) - - -def check_verification_after_code_changes(run_dir: Path, params: dict[str, Any]) -> dict: - prompt_path = run_dir / params.get("prompt_path", "prompt.json") - if not prompt_path.exists(): - return result("FAIL", "prompt.json is missing.") - try: - prompt = load_json(prompt_path) - except Exception: - return result("FAIL", "prompt.json could not be parsed.") - - log_path = run_dir / params.get("agentic_log_path", "logs/agentic.log") - if not log_path.exists(): - return result("FAIL", "agentic.log is missing.") - - log_text = log_path.read_text(encoding="utf-8", errors="ignore") - lines = [strip_ansi(line) for line in log_text.splitlines()] - file_events = parse_agentic_file_update_events(log_text) - if not file_events: - return result("WARN", "No file update entries found in agentic.log.") - - code_event = first_code_change_event(file_events) - if code_event is None: - return result( - "WARN", - "No non-doc, non-.codex file changes found; verification ordering not evaluated.", - [{"path": str(log_path), "quote": "no code updates"}], - ) - code_line = int(code_event["line_index"]) - - command_events = extract_command_events(log_text) - prompt_cmds = prompt_command_candidates(prompt) - verification_events = [ - event - for event in command_events - if int(event["line_index"]) > code_line - and is_verification_command(str(event.get("cmd", "")), prompt_cmds) - ] - - code_quote = lines[code_line] if code_line < len(lines) else str(code_event.get("path", "")) - if not verification_events: - return result( - "FAIL", - "No build/test/lint verification command detected after code changes in agentic.log.", - [{"path": str(log_path), "quote": code_quote}], - [ - "Run at least one build, test, or lint command after code changes within the agentic loop." - ], - ) - - evidence: list[dict[str, str]] = [{"path": str(log_path), "quote": code_quote}] - for event in verification_events[:2]: - idx = int(event["line_index"]) - quote = lines[idx] if idx < len(lines) else str(event.get("raw_line", event.get("cmd", ""))) - evidence.append({"path": str(log_path), "quote": quote}) - - return result( - "PASS", - "Verification command(s) appear after code changes in agentic.log.", - evidence, - ) - - -def check_agentic_run_success(run_dir: Path, params: dict[str, Any]) -> dict: - summary_path = run_dir / params.get("path", "agentic_summary.json") - if not summary_path.exists(): - return result("FAIL", "agentic_summary.json is missing.") - try: - summary = load_json(summary_path) - except Exception: - return result("FAIL", "agentic_summary.json could not be parsed.") - status = summary.get("status") - exit_code = summary.get("exit_code") - if status == "PASS" and exit_code == 0: - return result( - "PASS", - "Agentic loop completed successfully.", - [{"path": str(summary_path), "quote": "PASS"}], - ) - return result( - "FAIL", - "Agentic loop did not complete successfully.", - [{"path": str(summary_path), "quote": str(status)}], - ) - - -def check_repo_root_only_changes(run_dir: Path, params: dict[str, Any]) -> dict: - repo_root_raw = run_cmd(["git", "rev-parse", "--show-toplevel"]) - if repo_root_raw.startswith(""): - return result("FAIL", f"Unable to resolve repo root: {repo_root_raw}") - repo_root = Path(repo_root_raw).resolve() - - files_raw = run_cmd(["git", "diff", "--name-only"]) - if files_raw.startswith(""): - return result("FAIL", f"Unable to list git diff files: {files_raw}") - files = [line.strip() for line in files_raw.splitlines() if line.strip()] - if not files: - return result("WARN", "No git diff detected; path policy not evaluated.") - - bad_paths = [] - for rel in files: - if rel.startswith(("/", "..")): - bad_paths.append(rel) - continue - abs_path = (repo_root / rel).resolve() - try: - abs_path.relative_to(repo_root) - except ValueError: - bad_paths.append(rel) - - if bad_paths: - evidence = [{"path": str(repo_root), "quote": ", ".join(bad_paths[:5])}] - return result( - "FAIL", - "Git diff includes paths outside repo root.", - evidence, - ["Ensure all changes stay under repo root."], - ) - - return result( - "PASS", - "All git diff paths resolve under repo root.", - [{"path": str(repo_root), "quote": "repo root"}], - ) - - -RULES = { - "prompt_json_present": check_prompt_json_present, - "build_test_plan_present": check_build_test_plan_present, - "execution_summary_status": check_execution_summary_status, - "execution_logs_no_errors": check_execution_logs_no_errors, - "agentic_run_success": check_agentic_run_success, - "exec_plan_before_code_changes": check_exec_plan_before_code_changes, - "verification_after_code_changes": check_verification_after_code_changes, - "repo_root_only_changes": check_repo_root_only_changes, -} - - -def main() -> int: - parser = argparse.ArgumentParser(description="Run deterministic checks for integration test.") - parser.add_argument( - "--out-dir", default=".codex-readiness-integration-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument( - "--checks", - default=str(Path(__file__).resolve().parents[1] / "references" / "checks.json"), - ) - args = parser.parse_args() - - base_dir = Path(args.out_dir) - run_dir = resolve_run_dir(base_dir, args.run_dir) - - checks_path = Path(args.checks) - checks_data = load_json(checks_path) - results: dict[str, dict] = {} - - for check in sort_checks_by_priority(checks_data.get("checks", [])): - if not check.get("enabled_by_default"): - continue - rule_id = check.get("deterministic_rule_id") - if not rule_id: - continue - rule = RULES.get(rule_id) - if not rule: - results[check["id"]] = result("FAIL", f"Unknown deterministic rule: {rule_id}.") - continue - params = check.get("deterministic_rule_params", {}) - results[check["id"]] = rule(run_dir, params) - - output_path = run_dir / "deterministic_results.json" - output_path.write_text(json.dumps(results, indent=2), encoding="utf-8") - print(str(output_path)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/skills/.experimental/codex-readiness-integration-test/scripts/run_agentic_loop.py b/skills/.experimental/codex-readiness-integration-test/scripts/run_agentic_loop.py deleted file mode 100644 index 1eed630..0000000 --- a/skills/.experimental/codex-readiness-integration-test/scripts/run_agentic_loop.py +++ /dev/null @@ -1,869 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -import os -import pty -import re -import selectors -import shlex -import subprocess -import sys -import time -from datetime import datetime, timezone -from pathlib import Path -from typing import Any - -SESSION_ID_PATTERN = re.compile(r"session id:\s*([0-9a-fA-F-]{8,})") -ANSI_ESCAPE_PATTERN = re.compile(r"\x1B(?:[@-Z\\-_]|\[[0-?]*[ -/]*[@-~])") -QUESTION_PREFIXES = ( - "please ", - "could you", - "can you", - "would you", - "which ", - "what ", - "where ", -) -QUESTION_SUBSTRINGS = ( - "please provide", - "please paste", - "please point me", - "provide ", - "paste ", - "point me", - "clarify", - "i can't find", - "i can’t find", -) -IGNORE_LINE_PREFIXES = ( - "openai codex", - "--------", - "workdir:", - "model:", - "provider:", - "approval:", - "sandbox:", - "reasoning", - "session id:", - "mcp startup:", - "thinking", - "exec", - "tokens used", -) -MAX_FOLLOWUP_ROUNDS = 5 -TAIL_LINE_LIMIT = 400 -QUESTION_TERMINATE_GRACE_SECONDS = 2.0 - -SANDBOX_BLOCK_SUBSTRINGS = [ - "sandbox-blocked", - "shell tool is sandbox-blocked", - "sandbox_apply: operation not permitted", - "operation not permitted", -] - - -def now_iso() -> str: - return datetime.now(timezone.utc).isoformat() - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = load_json(latest_path) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "prompt.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def normalize_args(raw_args: Any) -> list[str]: - if raw_args is None: - return [] - if isinstance(raw_args, str): - return shlex.split(raw_args) - if isinstance(raw_args, list): - return [str(arg) for arg in raw_args] - return [] - - -def substitute_args(args: list[str], mapping: dict[str, str]) -> list[str]: - resolved = [] - for arg in args: - updated = arg - for key, value in mapping.items(): - updated = updated.replace(key, value) - resolved.append(updated) - return resolved - - -def sanitize_agentic_args(args: list[str]) -> list[str]: - """Remove unsafe or runner-managed flags from prompt-supplied args.""" - sanitized: list[str] = [] - skip_next = False - # Flags that should not be controlled by the prompt in this runner. - deny_flags_with_value = {"--sandbox", "--ask-for-approval", "-C", "--cd"} - deny_flags = { - "--dangerously-bypass-approvals-and-sandbox", - "--full-auto", - "exec", - "resume", - "{change_prompt}", - } - for arg in args: - if skip_next: - skip_next = False - continue - if arg in deny_flags_with_value: - skip_next = True - continue - if arg in deny_flags: - continue - sanitized.append(arg) - return sanitized - - -def build_command( - prompt: dict[str, Any], agents_path: Path, prompt_path: Path, repo_root: Path -) -> tuple[list[str], int]: - agentic_config = prompt.get("agentic_loop") - config: dict[str, Any] = agentic_config if isinstance(agentic_config, dict) else {} - cmd = config.get("cmd") or "codex" - prompt_args = sanitize_agentic_args(normalize_args(config.get("args"))) - # Hardcode a safe, broadly supported permission model at the runner level, - # while allowing other prompt-supplied flags (e.g., model selection). - raw_args = ["exec", "--full-auto"] + prompt_args + ["-C", "{repo_root}", "{change_prompt}"] - args = normalize_args(raw_args) - change_prompt = str(prompt.get("change_prompt") or "").strip() - plan_instruction = str(prompt.get("plan_instruction") or "").strip() - if plan_instruction: - change_prompt = f"{plan_instruction} {change_prompt}".strip() - mapping = { - "{agents_path}": str(agents_path), - "{change_prompt}": change_prompt, - "{prompt_path}": str(prompt_path), - "{run_dir}": str(prompt_path.parent), - "{repo_root}": str(repo_root), - } - args = substitute_args(args, mapping) - timeout = int(config.get("timeout_seconds") or 1800) - return [cmd] + args, timeout - - -def filter_resume_args(args: list[str]) -> list[str]: - filtered: list[str] = [] - skip_next = False - for arg in args: - if skip_next: - skip_next = False - continue - if arg == "exec": - continue - if arg in {"-C", "--cd"}: - skip_next = True - continue - if arg == "{change_prompt}": - continue - filtered.append(arg) - return filtered - - -def build_resume_command( - prompt: dict[str, Any], - session_id: str, - resume_prompt: str, - agents_path: Path, - prompt_path: Path, - repo_root: Path, -) -> tuple[list[str], int]: - agentic_config = prompt.get("agentic_loop") - config: dict[str, Any] = agentic_config if isinstance(agentic_config, dict) else {} - cmd = config.get("cmd") or "codex" - prompt_args = sanitize_agentic_args(normalize_args(config.get("args"))) - raw_args = ["exec", "--full-auto"] + prompt_args - args = normalize_args(raw_args) - mapping = { - "{agents_path}": str(agents_path), - "{prompt_path}": str(prompt_path), - "{run_dir}": str(prompt_path.parent), - "{repo_root}": str(repo_root), - } - filtered_args = substitute_args(filter_resume_args(args), mapping) - timeout = int(config.get("timeout_seconds") or 1800) - resume_args = ["exec", "resume"] + filtered_args + [session_id, resume_prompt] - return [cmd] + resume_args, timeout - - -def run_command( - cmd: list[str], - cwd: Path, - env: dict, - timeout: int, - log_path: Path, - *, - append: bool, - attempt_label: str, -) -> dict: - started_at = now_iso() - start_time = time.time() - exit_code = None - status = "FAIL" - mode = "a" if append else "w" - timed_out = False - error_message = None - master_fd = None - cmd_display = " ".join(shlex.quote(part) for part in cmd) - - try: - with log_path.open(mode, encoding="utf-8") as log_file: - if append: - log_file.write("\n\n") - log_file.write(f"===== {attempt_label} =====\n") - log_file.write(f"$ {cmd_display}\n") - log_file.flush() - - if not sys.stdin.isatty(): - error_message = "Interactive mode requires a TTY on stdin." - raise RuntimeError(error_message) - - master_fd, slave_fd = pty.openpty() - proc = subprocess.Popen( - cmd, - cwd=str(cwd), - env=env, - stdin=slave_fd, - stdout=slave_fd, - stderr=slave_fd, - close_fds=True, - ) - os.close(slave_fd) - - sel = selectors.DefaultSelector() - sel.register(master_fd, selectors.EVENT_READ) - sel.register(sys.stdin, selectors.EVENT_READ) - stdin_fd = sys.stdin.fileno() - old_tty = termios.tcgetattr(stdin_fd) - deadline = time.time() + timeout - - try: - tty.setraw(stdin_fd) - master_closed = False - while True: - if proc.poll() is not None: - break - if time.time() > deadline: - timed_out = True - proc.terminate() - break - events = sel.select(timeout=0.1) - for key, _ in events: - if key.fileobj == master_fd: - data = os.read(master_fd, 1024) - if data: - os.write(sys.stdout.fileno(), data) - log_file.buffer.write(data) - log_file.flush() - else: - master_closed = True - break - else: - data = os.read(stdin_fd, 1024) - if data: - os.write(master_fd, data) - if master_closed: - break - if timed_out: - try: - proc.wait(timeout=5) - except subprocess.TimeoutExpired: - proc.kill() - proc.wait(timeout=5) - exit_code = proc.wait() - finally: - termios.tcsetattr(stdin_fd, termios.TCSADRAIN, old_tty) - sel.close() - except FileNotFoundError as exc: - error_message = f"Command not found: {exc}" - except subprocess.TimeoutExpired: - error_message = f"Command timed out after {timeout} seconds." - except RuntimeError as exc: - error_message = str(exc) - except KeyboardInterrupt: - error_message = "Interrupted by user." - finally: - if master_fd is not None: - os.close(master_fd) - if error_message: - with log_path.open("a", encoding="utf-8") as log_file: - log_file.write(f"{error_message}\n") - if timed_out: - status = "FAIL" - else: - if exit_code == 0: - status = "PASS" - else: - status = "FAIL" - - ended_at = now_iso() - duration = round(time.time() - start_time, 2) - - return { - "cmd": cmd_display, - "status": status, - "exit_code": exit_code, - "duration_seconds": duration, - "started_at": started_at, - "ended_at": ended_at, - } - - -def prepare_codex_env(repo_root: Path, env: dict) -> dict: - codex_home = repo_root / ".codex-home" - cache_root = codex_home / ".cache" / "codex" - cache_root.mkdir(parents=True, exist_ok=True) - env = dict(env) - env["HOME"] = str(codex_home) - env.setdefault("XDG_CACHE_HOME", str(codex_home / ".cache")) - env.setdefault("CODEX_NO_UPDATE", "1") - return env - - -def extract_session_id(log_text: str) -> str | None: - matches = SESSION_ID_PATTERN.findall(log_text) - if not matches: - return None - return matches[-1] - - -def strip_ansi(text: str) -> str: - return ANSI_ESCAPE_PATTERN.sub("", text) - - -def is_question_line(line: str) -> bool: - lower = line.strip().lower() - if not lower: - return False - # Codex often ends with friendly follow-up headings like 'what changed:' - # or 'what i verified:'. Treat these as non-blocking in non-interactive runs. - if lower.startswith("what changed"): - return False - if lower.startswith("what ") and lower.endswith(":"): - return False - if lower.endswith("?"): - return True - if lower.startswith(QUESTION_PREFIXES): - return True - return any(fragment in lower for fragment in QUESTION_SUBSTRINGS) - - -def extract_last_attempt_log(log_text: str) -> str: - sections = re.split(r"^===== .* =====$", log_text, flags=re.MULTILINE) - if not sections: - return log_text - return sections[-1] - - -def extract_clarifying_question(log_text: str) -> str | None: - lines = [line.strip() for line in log_text.splitlines()] - for line in reversed(lines): - if not line: - continue - lower = line.lower() - if lower.startswith(IGNORE_LINE_PREFIXES): - continue - if is_question_line(line): - return line - return None - - -def prompt_for_answer(question: str) -> str: - print("\nCodex asked:") - print(question) - if not sys.stdin.isatty(): - auto = os.environ.get( - "CODEX_INTEGRATION_AUTOANSWER", - "Proceed with best effort using the repository context. Do not ask follow-up questions.", - ).strip() - print(f"Auto-answering (non-interactive): {auto}") - return auto - while True: - answer = input("Answer: ").strip() - if answer: - return answer - print("Please provide an answer to continue.") - - -def detect_sandbox_block(log_text: str) -> str | None: - for raw_line in log_text.splitlines(): - line = raw_line.strip() - if not line: - continue - lower = line.lower() - for marker in SANDBOX_BLOCK_SUBSTRINGS: - if marker in lower: - return line - return None - - -def append_tail_lines(tail_lines: list[str], line: str) -> None: - tail_lines.append(line) - if len(tail_lines) > TAIL_LINE_LIMIT: - del tail_lines[: len(tail_lines) - TAIL_LINE_LIMIT] - - -def run_non_interactive( - cmd: list[str], - cwd: Path, - env: dict, - timeout: int, - log_path: Path, - *, - append: bool, - attempt_label: str, -) -> dict: - started_at = now_iso() - start_time = time.time() - exit_code = None - status = "FAIL" - mode = "a" if append else "w" - timed_out = False - error_message = None - question_detected = None - session_id = None - terminated_for_question = False - log_buffer = "" - tail_lines: list[str] = [] - cmd_display = " ".join(shlex.quote(part) for part in cmd) - - try: - with log_path.open(mode, encoding="utf-8") as log_file: - if append: - log_file.write("\n\n") - log_file.write(f"===== {attempt_label} =====\n") - log_file.write(f"$ {cmd_display}\n") - log_file.flush() - - proc = subprocess.Popen( - cmd, - cwd=str(cwd), - env=env, - stdin=subprocess.DEVNULL, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - close_fds=True, - ) - - if proc.stdout is None: - raise RuntimeError("Failed to capture stdout for non-interactive mode.") - - sel = selectors.DefaultSelector() - sel.register(proc.stdout, selectors.EVENT_READ) - deadline = time.time() + timeout - - try: - while True: - if proc.poll() is not None: - break - if time.time() > deadline: - timed_out = True - proc.terminate() - break - events = sel.select(timeout=0.1) - for key, _ in events: - data = os.read(key.fileobj.fileno(), 1024) - if not data: - continue - log_file.buffer.write(data) - log_file.flush() - - text = data.decode("utf-8", errors="ignore") - log_buffer += text - while "\n" in log_buffer: - line, log_buffer = log_buffer.split("\n", 1) - clean = strip_ansi(line) - append_tail_lines(tail_lines, clean) - if session_id is None: - match = SESSION_ID_PATTERN.search(clean) - if match: - session_id = match.group(1) - if question_detected is None and is_question_line(clean): - question_detected = clean.strip() - if session_id: - terminated_for_question = True - proc.terminate() - break - if terminated_for_question: - break - if terminated_for_question: - break - - if log_buffer: - clean = strip_ansi(log_buffer) - append_tail_lines(tail_lines, clean) - if session_id is None: - match = SESSION_ID_PATTERN.search(clean) - if match: - session_id = match.group(1) - if question_detected is None and is_question_line(clean): - question_detected = clean.strip() - if terminated_for_question: - deadline = time.time() + QUESTION_TERMINATE_GRACE_SECONDS - while time.time() < deadline and proc.poll() is None: - time.sleep(0.05) - if timed_out: - try: - proc.wait(timeout=5) - except subprocess.TimeoutExpired: - proc.kill() - proc.wait(timeout=5) - else: - exit_code = proc.wait() - finally: - sel.close() - except FileNotFoundError as exc: - error_message = f"Command not found: {exc}" - except subprocess.TimeoutExpired: - error_message = f"Command timed out after {timeout} seconds." - except RuntimeError as exc: - error_message = str(exc) - except KeyboardInterrupt: - error_message = "Interrupted by user." - finally: - if error_message: - with log_path.open("a", encoding="utf-8") as log_file: - log_file.write(f"{error_message}\n") - if timed_out: - status = "FAIL" - else: - if exit_code == 0 or terminated_for_question: - status = "PASS" - else: - status = "FAIL" - - ended_at = now_iso() - duration = round(time.time() - start_time, 2) - - return { - "cmd": cmd_display, - "status": status, - "exit_code": exit_code, - "duration_seconds": duration, - "started_at": started_at, - "ended_at": ended_at, - "mode": "non_interactive", - "question_detected": question_detected, - "question_handled": False, - "session_id": session_id, - "terminated_for_question": terminated_for_question, - } - - -def run_safe_interactive( - cmd: list[str], - cwd: Path, - env: dict, - timeout: int, - log_path: Path, - *, - append: bool, - attempt_label: str, -) -> dict: - started_at = now_iso() - start_time = time.time() - exit_code = None - status = "FAIL" - mode = "a" if append else "w" - timed_out = False - error_message = None - master_fd = None - question_detected = None - session_id = None - questions_handled: list[str] = [] - log_buffer = "" - tail_lines: list[str] = [] - cmd_display = " ".join(shlex.quote(part) for part in cmd) - - try: - with log_path.open(mode, encoding="utf-8") as log_file: - if append: - log_file.write("\n\n") - log_file.write(f"===== {attempt_label} =====\n") - log_file.write(f"$ {cmd_display}\n") - log_file.flush() - - if not sys.stdin.isatty(): - error_message = "Safe-interactive mode requires a TTY on stdin." - raise RuntimeError(error_message) - - master_fd, slave_fd = pty.openpty() - proc = subprocess.Popen( - cmd, - cwd=str(cwd), - env=env, - stdin=slave_fd, - stdout=slave_fd, - stderr=slave_fd, - close_fds=True, - ) - os.close(slave_fd) - - sel = selectors.DefaultSelector() - sel.register(master_fd, selectors.EVENT_READ) - deadline = time.time() + timeout - - try: - master_closed = False - while True: - if proc.poll() is not None: - break - if time.time() > deadline: - timed_out = True - proc.terminate() - break - events = sel.select(timeout=0.1) - for key, _ in events: - if key.fileobj == master_fd: - data = os.read(master_fd, 1024) - if data: - log_file.buffer.write(data) - log_file.flush() - - text = data.decode("utf-8", errors="ignore") - log_buffer += text - while "\n" in log_buffer: - line, log_buffer = log_buffer.split("\n", 1) - clean = strip_ansi(line) - append_tail_lines(tail_lines, clean) - if session_id is None: - match = SESSION_ID_PATTERN.search(clean) - if match: - session_id = match.group(1) - if is_question_line(clean): - question_detected = clean.strip() - if ( - question_detected - and question_detected not in questions_handled - ): - answer = prompt_for_answer(question_detected) - os.write(master_fd, (answer + "\n").encode()) - questions_handled.append(question_detected) - else: - master_closed = True - break - if master_closed: - break - if timed_out: - try: - proc.wait(timeout=5) - except subprocess.TimeoutExpired: - proc.kill() - proc.wait(timeout=5) - exit_code = proc.wait() - finally: - sel.close() - except FileNotFoundError as exc: - error_message = f"Command not found: {exc}" - except subprocess.TimeoutExpired: - error_message = f"Command timed out after {timeout} seconds." - except RuntimeError as exc: - error_message = str(exc) - except KeyboardInterrupt: - error_message = "Interrupted by user." - finally: - if master_fd is not None: - os.close(master_fd) - if error_message: - with log_path.open("a", encoding="utf-8") as log_file: - log_file.write(f"{error_message}\n") - if timed_out: - status = "FAIL" - else: - if exit_code == 0: - status = "PASS" - else: - status = "FAIL" - - ended_at = now_iso() - duration = round(time.time() - start_time, 2) - - return { - "cmd": cmd_display, - "status": status, - "exit_code": exit_code, - "duration_seconds": duration, - "started_at": started_at, - "ended_at": ended_at, - "mode": "safe_interactive", - "question_detected": question_detected, - "question_handled": bool(questions_handled), - "session_id": session_id, - } - - -def main() -> int: - parser = argparse.ArgumentParser(description="Run the agentic loop via Codex CLI.") - parser.add_argument( - "--out-dir", default=".codex-readiness-integration-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - args = parser.parse_args() - - base_dir = Path(args.out_dir) - run_dir = resolve_run_dir(base_dir, args.run_dir) - run_dir.mkdir(parents=True, exist_ok=True) - logs_dir = run_dir / "logs" - logs_dir.mkdir(parents=True, exist_ok=True) - - prompt_path = run_dir / "prompt.json" - if not prompt_path.exists(): - summary = { - "cmd": "", - "status": "FAIL", - "exit_code": None, - "duration_seconds": 0, - "started_at": now_iso(), - "ended_at": now_iso(), - "error": "prompt.json missing", - } - summary_path = run_dir / "agentic_summary.json" - summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8") - print(str(summary_path)) - return 2 - - prompt = load_json(prompt_path) - repo_root = Path.cwd() - agents_path = repo_root / "AGENTS.md" - - log_path = logs_dir / "agentic.log" - env = prepare_codex_env(repo_root, os.environ.copy()) - - questions: list[str] = [] - attempts: list[dict] = [] - session_id: str | None = None - resume_prompt: str | None = None - append_log = False - summary: dict | None = None - auto_answer_count = 0 - last_auto_answer_text: str | None = None - - for attempt in range(1, MAX_FOLLOWUP_ROUNDS + 1): - if resume_prompt: - if session_id is None: - break - cmd, timeout = build_resume_command( - prompt, - session_id, - resume_prompt, - agents_path, - prompt_path, - repo_root, - ) - attempt_label = f"resume-attempt-{attempt}" - else: - cmd, timeout = build_command(prompt, agents_path, prompt_path, repo_root) - attempt_label = f"agentic-attempt-{attempt}" - - if resume_prompt and sys.stdin.isatty(): - summary = run_safe_interactive( - cmd, - repo_root, - env, - timeout, - log_path, - append=append_log, - attempt_label=attempt_label, - ) - else: - summary = run_non_interactive( - cmd, - repo_root, - env, - timeout, - log_path, - append=append_log, - attempt_label=attempt_label, - ) - attempts.append(dict(summary)) - append_log = True - - if summary.get("question_handled"): - break - - log_text = log_path.read_text(encoding="utf-8", errors="ignore") - if session_id is None: - session_id = summary.get("session_id") or extract_session_id(log_text) - attempt_log = extract_last_attempt_log(log_text) - question = summary.get("question_detected") or extract_clarifying_question(attempt_log) - if question: - questions.append(question) - if not sys.stdin.isatty(): - if session_id is None: - summary["status"] = "FAIL" - summary["error"] = "session id missing for auto-answer resume" - break - resume_prompt = prompt_for_answer(question) - auto_answer_count += 1 - last_auto_answer_text = resume_prompt - summary["question_detected"] = question - summary["auto_answer_used"] = True - summary["auto_answer_text"] = resume_prompt - summary["auto_answer_count"] = auto_answer_count - summary["non_interactive_question_ignored"] = False - continue - if session_id is None: - summary["status"] = "FAIL" - summary["error"] = "session id missing for resume" - break - resume_prompt = prompt_for_answer(question) - continue - break - - if summary is None: - summary = { - "cmd": "", - "status": "FAIL", - "exit_code": None, - "duration_seconds": 0, - "started_at": now_iso(), - "ended_at": now_iso(), - "error": "agentic loop did not run", - } - - log_text = log_path.read_text(encoding="utf-8", errors="ignore") - sandbox_block_evidence = detect_sandbox_block(log_text) - if sandbox_block_evidence: - summary["status"] = "FAIL" - summary["error"] = ( - "Codex tool access appears to be sandbox-blocked. " - "Re-run the integration test with escalated permissions." - ) - summary["sandbox_blocked"] = True - summary["sandbox_block_evidence"] = sandbox_block_evidence - summary["requires_escalation"] = True - print("Detected sandbox-blocked tool access; escalate permissions and re-run.") - - if auto_answer_count: - summary["auto_answer_count"] = auto_answer_count - if last_auto_answer_text: - summary["auto_answer_text"] = last_auto_answer_text - - summary_path = run_dir / "agentic_summary.json" - summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8") - print(str(summary_path)) - if summary.get("requires_escalation"): - return 3 - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/skills/.experimental/codex-readiness-integration-test/scripts/run_integration_test.py b/skills/.experimental/codex-readiness-integration-test/scripts/run_integration_test.py deleted file mode 100644 index 2d71e9c..0000000 --- a/skills/.experimental/codex-readiness-integration-test/scripts/run_integration_test.py +++ /dev/null @@ -1,592 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -import os -import re -import subprocess -import sys -from datetime import datetime, timezone -from pathlib import Path - -SCORING_FOCUS_DEFAULT = [ - "correctness", - "context_usage", - "builds_tests_pass", - "maintainability", - "risk", -] - -VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"} -SKILL_REF_PATTERN = re.compile(r"\$([A-Za-z0-9_.-]+)") -SKILL_PATH_PATTERN = re.compile( - r"(?:\.codex/skills|~/.codex/skills|/\.codex/skills|skills)/([A-Za-z0-9_.-]+)" -) -TEST_KEYWORDS = [ - " test", - "pytest", - "node --test", - "go test", - "cargo test", - "mvn test", - "gradle test", - "./gradlew test", -] -BUILD_KEYWORDS = [ - " build", - "compile", - "mvn package", - "gradle build", - "./gradlew build", - "go build", - "cargo build", -] - - -def now_stamp() -> str: - return datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ") - - -def now_iso() -> str: - return datetime.now(timezone.utc).isoformat() - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def write_json(path: Path, data: dict) -> None: - path.write_text(json.dumps(data, indent=2), encoding="utf-8") - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - return base_dir / now_stamp() - - -def ensure_prompt_template(prompt_path: Path, seed_task: str | None) -> Path: - if prompt_path.exists(): - return prompt_path - prompt = { - "seed_task": seed_task, - "prompt_origin": "manual", - "change_prompt": "", - "acceptance_criteria": [], - "build_test_plan": [], - "scoring_focus": SCORING_FOCUS_DEFAULT, - "approved": False, - "generated_at": now_iso(), - } - write_json(prompt_path, prompt) - return prompt_path - - -def approve_prompt(prompt_path: Path) -> dict: - prompt = load_json(prompt_path) - prompt["approved"] = True - prompt["approved_at"] = now_iso() - write_json(prompt_path, prompt) - return prompt - - -def prompt_ready(prompt: dict) -> tuple[bool, str]: - if not prompt.get("change_prompt"): - return False, "change_prompt is empty" - plan = prompt.get("build_test_plan") - if not isinstance(plan, list) or not plan: - return False, "build_test_plan is missing or empty" - return True, "ready" - - -def read_text(path: Path) -> str: - try: - return path.read_text(encoding="utf-8") - except Exception: - try: - return path.read_text(encoding="utf-8", errors="ignore") - except Exception: - return "" - - -def extract_candidate_commands(text: str) -> list[str]: - commands = [] - in_code_block = False - for line in text.splitlines(): - stripped = line.strip() - if stripped.startswith("```"): - in_code_block = not in_code_block - continue - if in_code_block: - if stripped and not stripped.startswith("#"): - commands.append(stripped) - continue - inline = re.findall(r"`([^`]+)`", line) - for cmd in inline: - cmd_str = cmd.strip() - if cmd_str: - commands.append(cmd_str) - if stripped.startswith("$"): - cmd = stripped.lstrip("$ ") - if cmd: - commands.append(cmd) - normalized = [] - seen = set() - for cmd in commands: - cleaned = cmd.strip() - if not cleaned or cleaned in seen: - continue - seen.add(cleaned) - normalized.append(cleaned) - return normalized - - -def extract_skill_refs(text: str) -> list[str]: - refs = set(SKILL_REF_PATTERN.findall(text)) - refs.update(SKILL_PATH_PATTERN.findall(text)) - return sorted(refs) - - -def resolve_skills_roots(repo_root: Path) -> list[Path]: - candidates = [] - codex_home = os.environ.get("CODEX_HOME") - if codex_home: - candidates.append(Path(codex_home) / "skills") - candidates.append(repo_root / ".codex" / "skills") - candidates.append(Path.home() / ".codex" / "skills") - - roots = [] - seen = set() - for candidate in candidates: - try: - resolved = candidate.expanduser().resolve() - except Exception: - resolved = candidate.expanduser() - key = str(resolved) - if key in seen: - continue - seen.add(key) - if resolved.exists(): - roots.append(resolved) - return roots - - -def classify_command(cmd: str) -> str | None: - lower = cmd.lower() - if any(keyword in lower for keyword in TEST_KEYWORDS): - return "test" - if any(keyword in lower for keyword in BUILD_KEYWORDS): - return "build" - return None - - -def infer_build_test_plan(repo_root: Path) -> list[dict]: - agents_path = repo_root / "AGENTS.md" - candidate_commands: list[str] = [] - skill_refs: list[str] = [] - if agents_path.exists(): - agents_text = read_text(agents_path) - if agents_text: - candidate_commands.extend(extract_candidate_commands(agents_text)) - skill_refs = extract_skill_refs(agents_text) - - if skill_refs: - skills_roots = resolve_skills_roots(repo_root) - for skill_name in skill_refs: - for root in skills_roots: - skill_path = root / skill_name / "SKILL.md" - if skill_path.exists(): - skill_text = read_text(skill_path) - if skill_text: - candidate_commands.extend(extract_candidate_commands(skill_text)) - break - - seen = set() - unique_cmds = [] - for cmd in candidate_commands: - if cmd in seen: - continue - seen.add(cmd) - unique_cmds.append(cmd) - - build_cmd = None - test_cmd = None - for cmd in unique_cmds: - label = classify_command(cmd) - if label == "build" and build_cmd is None: - build_cmd = cmd - elif label == "test" and test_cmd is None: - test_cmd = cmd - if build_cmd and test_cmd: - break - - plan: list[dict] = [] - if build_cmd: - plan.append({"label": "build", "cmd": build_cmd}) - if test_cmd: - plan.append({"label": "test", "cmd": test_cmd}) - return plan - - -def write_plan_json(run_dir: Path, prompt: dict, cwd: Path) -> Path: - plan = { - "cwd": str(cwd), - "commands": prompt.get("build_test_plan", []), - } - plan_path = run_dir / "plan.json" - write_json(plan_path, plan) - return plan_path - - -def valid_llm_result(result: dict) -> bool: - if not isinstance(result, dict): - return False - if result.get("status") not in VALID_STATUSES: - return False - if "rationale" not in result or "confidence" not in result: - return False - if not isinstance(result.get("evidence_quotes"), list): - return False - if not isinstance(result.get("recommendations"), list): - return False - return True - - -def prepare_codex_home(repo_root: Path) -> Path: - codex_home = repo_root / ".codex-home" - cache_root = codex_home / ".cache" - codex_home.mkdir(parents=True, exist_ok=True) - cache_root.mkdir(parents=True, exist_ok=True) - (cache_root / "codex").mkdir(parents=True, exist_ok=True) - return codex_home - - -def prompt_for_json(label: str, prompt_text: str, json_fix_text: str) -> dict | None: - attempts = 0 - while attempts < 3: - print(f"\n=== {label} ===") - print(prompt_text.rstrip()) - print("\nPaste JSON result. End with a line containing only END.") - lines = [] - while True: - try: - line = input() - except EOFError: - return None - if line.strip() == "END": - break - lines.append(line) - raw = "\n".join(lines).strip() - if not raw: - print("No input received.") - else: - try: - data = json.loads(raw) - except json.JSONDecodeError as exc: - print(f"Invalid JSON: {exc}") - if json_fix_text: - print("\n" + json_fix_text.rstrip() + "\n") - else: - if valid_llm_result(data): - return data - print("JSON missing required keys or invalid types.") - attempts += 1 - print("Please retry. End with a line containing only END.") - return None - - -def run_llm_evals_manual(run_dir: Path) -> int: - llm_path = run_dir / "llm_results.json" - if llm_path.exists(): - return 0 - if not sys.stdin.isatty(): - print("LLM eval skipped: stdin is not a TTY and llm_results.json is missing.") - return 2 - - prompts_dir = Path(__file__).resolve().parents[1] / "references" - json_fix_path = prompts_dir / "json_fix.md" - json_fix_text = json_fix_path.read_text(encoding="utf-8") if json_fix_path.exists() else "" - - print("LLM evaluation required. Use evidence and execution summary for context.") - print(f"Evidence: {run_dir / 'evidence.json'}") - print(f"Execution summary: {run_dir / 'execution_summary.json'}") - - results = {} - evals = [ - ("agentic_loop_eval", prompts_dir / "agentic_loop_eval.md"), - ("change_quality_eval", prompts_dir / "change_quality.md"), - ] - - for check_id, prompt_path in evals: - prompt_text = prompt_path.read_text(encoding="utf-8") - result = prompt_for_json(check_id, prompt_text, json_fix_text) - if result is None: - print(f"LLM eval for {check_id} not completed.") - return 2 - results[check_id] = result - - llm_path.write_text(json.dumps(results, indent=2), encoding="utf-8") - return 0 - - -def run_llm_evals_auto(run_dir: Path, base_dir: Path) -> int: - llm_path = run_dir / "llm_results.json" - if llm_path.exists(): - return 0 - cmd = [ - sys.executable, - str(Path(__file__).resolve().parent / "run_llm_eval.py"), - "--out-dir", - str(base_dir), - "--run-dir", - str(run_dir), - ] - return run_step(cmd) - - -def run_llm_evals(run_dir: Path, base_dir: Path, manual: bool) -> int: - if manual: - return run_llm_evals_manual(run_dir) - return run_llm_evals_auto(run_dir, base_dir) - - -def run_step(cmd: list[str]) -> int: - result = subprocess.run(cmd, text=True) - return result.returncode - - -def print_prompt_options(prompt_path: Path) -> None: - print("\nChoose one of the following options to create the prompt:") - print("1) Let Codex generate it: use references/generate_prompt.md and save JSON to:") - print(f" {prompt_path}") - print("2) Write it yourself: fill in change_prompt and build_test_plan in:") - print(f" {prompt_path}") - - -def archive_prompt(base_dir: Path, run_dir: Path, prompt: dict) -> Path: - archive_dir = base_dir / "prompts" - archive_dir.mkdir(parents=True, exist_ok=True) - archive_path = archive_dir / f"{run_dir.name}.json" - write_json(archive_path, prompt) - return archive_path - - -def format_prompt_summary(prompt: dict) -> str: - lines: list[str] = ["Prompt summary", ""] - origin = prompt.get("prompt_origin") or "unknown" - seed_task = prompt.get("seed_task") - change_prompt = str(prompt.get("change_prompt") or "").strip() - acceptance_criteria = prompt.get("acceptance_criteria") or [] - build_test_plan = prompt.get("build_test_plan") or [] - scoring_focus = prompt.get("scoring_focus") or [] - - lines.append(f"origin: {origin}") - if seed_task is None: - lines.append("seed_task: none") - else: - lines.append(f"seed_task: {seed_task}") - lines.append("") - lines.append("change_prompt:") - if change_prompt: - lines.append(change_prompt) - else: - lines.append("(missing)") - lines.append("") - lines.append("acceptance_criteria:") - if acceptance_criteria: - for item in acceptance_criteria: - lines.append(f"- {item}") - else: - lines.append("- (none)") - lines.append("") - lines.append("build_test_plan:") - if build_test_plan: - for step in build_test_plan: - if isinstance(step, str): - label = "step" - cmd = step - elif isinstance(step, dict): - label = step.get("label") or "step" - cmd = step.get("cmd") or "" - else: - label = "step" - cmd = "" - if cmd: - lines.append(f"- {label}: {cmd}") - else: - lines.append(f"- {label}") - else: - lines.append("- (none)") - lines.append("") - lines.append("scoring_focus:") - if scoring_focus: - for item in scoring_focus: - lines.append(f"- {item}") - else: - lines.append("- (none)") - return "\n".join(lines) + "\n" - - -def ensure_prompt_origin(prompt: dict, seed_task: str | None) -> None: - if "prompt_origin" in prompt: - return - if seed_task: - prompt["prompt_origin"] = "auto" - else: - prompt["prompt_origin"] = "manual" - - -def main() -> int: - parser = argparse.ArgumentParser(description="Run the codex-readiness-integration-test.") - parser.add_argument( - "--out-dir", default=".codex-readiness-integration-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument("--seed-task", default=None, help="Optional seed task") - parser.add_argument( - "--approve-prompt", action="store_true", help="Mark prompt.json as approved and continue" - ) - parser.add_argument( - "--skip-agentic-loop", action="store_true", help="Skip the agentic loop execution" - ) - parser.add_argument( - "--skip-llm-eval", action="store_true", help="Skip in-session LLM evaluation prompts" - ) - parser.add_argument( - "--manual-llm-eval", action="store_true", help="Prompt for manual LLM evaluation input" - ) - args = parser.parse_args() - - base_dir = Path(args.out_dir) - base_dir.mkdir(parents=True, exist_ok=True) - repo_root = Path.cwd() - codex_home = prepare_codex_home(repo_root) - print(f"Initialized Codex home at {codex_home}.") - cache_home = codex_home / ".cache" - print(f"Using repo-local Codex home at {codex_home}.") - print("If you have not authenticated with Codex for this repo, run:") - print(f" HOME={codex_home} XDG_CACHE_HOME={cache_home} codex login") - print(f" HOME={codex_home} XDG_CACHE_HOME={cache_home} codex login status") - prompt_path = ensure_prompt_template(base_dir / "prompt.pending.json", args.seed_task) - prompt = load_json(prompt_path) - - ensure_prompt_origin(prompt, args.seed_task) - - if args.approve_prompt: - prompt = approve_prompt(prompt_path) - - if not prompt.get("build_test_plan"): - inferred_plan = infer_build_test_plan(repo_root) - if inferred_plan: - prompt["build_test_plan"] = inferred_plan - write_json(prompt_path, prompt) - - ready, reason = prompt_ready(prompt) - if not ready: - print(f"Prompt not ready: {reason}") - if args.seed_task: - print( - "A seed task was provided, but prompt.json still needs a change_prompt and build_test_plan." - ) - else: - print("No seed task provided. Generate a prompt using references/generate_prompt.md.") - print_prompt_options(prompt_path) - print(f"\nEdit {prompt_path} and re-run with --approve-prompt.") - return 2 - - if not prompt.get("approved"): - print(f"Prompt not approved. Review {prompt_path} and re-run with --approve-prompt.") - return 2 - - run_dir = resolve_run_dir(base_dir, args.run_dir) - run_dir.mkdir(parents=True, exist_ok=True) - - latest_path = base_dir / "latest.json" - write_json(latest_path, {"run_dir": str(run_dir)}) - - prompt_run_path = run_dir / "prompt.json" - write_json(prompt_run_path, prompt) - archive_prompt(base_dir, run_dir, prompt) - print("\n" + format_prompt_summary(prompt).rstrip() + "\n") - - plan_path = write_plan_json(run_dir, prompt, Path.cwd()) - - if not args.skip_agentic_loop: - agentic_cmd = [ - sys.executable, - str(Path(__file__).resolve().parent / "run_agentic_loop.py"), - "--out-dir", - str(base_dir), - "--run-dir", - str(run_dir), - ] - agentic_status = run_step(agentic_cmd) - if agentic_status != 0: - print(f"Agentic loop exited with code {agentic_status}.") - return agentic_status - - agentic_summary_path = run_dir / "agentic_summary.json" - if agentic_summary_path.exists(): - agentic_summary = load_json(agentic_summary_path) - if agentic_summary.get("requires_escalation"): - print( - "Agentic loop indicates sandbox-blocked access. " - "Re-run the integration test with escalated permissions." - ) - return 3 - - run_plan_cmd = [ - sys.executable, - str(Path(__file__).resolve().parent / "run_plan.py"), - "--plan", - str(plan_path), - "--out-dir", - str(base_dir), - "--run-dir", - str(run_dir), - ] - run_step(run_plan_cmd) - - run_step( - [ - sys.executable, - str(Path(__file__).resolve().parent / "collect_evidence.py"), - "--out-dir", - str(base_dir), - "--run-dir", - str(run_dir), - ] - ) - - run_step( - [ - sys.executable, - str(Path(__file__).resolve().parent / "deterministic_rules.py"), - "--out-dir", - str(base_dir), - "--run-dir", - str(run_dir), - ] - ) - - if not args.skip_llm_eval: - llm_status = run_llm_evals(run_dir, base_dir, args.manual_llm_eval) - if llm_status != 0: - return llm_status - - run_step( - [ - sys.executable, - str(Path(__file__).resolve().parent / "scoring.py"), - "--out-dir", - str(base_dir), - "--run-dir", - str(run_dir), - ] - ) - - print(f"Run complete: {run_dir}") - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/skills/.experimental/codex-readiness-integration-test/scripts/run_llm_eval.py b/skills/.experimental/codex-readiness-integration-test/scripts/run_llm_eval.py deleted file mode 100644 index ae4da44..0000000 --- a/skills/.experimental/codex-readiness-integration-test/scripts/run_llm_eval.py +++ /dev/null @@ -1,363 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -import os -import shlex -import subprocess -from datetime import datetime, timezone -from pathlib import Path -from typing import Any - -VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"} - - -def now_iso() -> str: - return datetime.now(timezone.utc).isoformat() - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def write_json(path: Path, data: dict) -> None: - path.write_text(json.dumps(data, indent=2), encoding="utf-8") - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = load_json(latest_path) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "prompt.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def normalize_priority(value) -> int: - try: - priority = int(value) - except (TypeError, ValueError): - return 3 - return priority if priority in {0, 1, 2, 3} else 3 - - -def sort_checks_by_priority(checks: list[dict]) -> list[dict]: - return sorted( - checks, - key=lambda check: ( - normalize_priority(check.get("priority")), - check.get("id", ""), - ), - ) - - -def normalize_args(raw_args: Any) -> list[str]: - if raw_args is None: - return [] - if isinstance(raw_args, str): - return shlex.split(raw_args) - if isinstance(raw_args, list): - return [str(arg) for arg in raw_args] - return [] - - -def substitute_args(args: list[str], mapping: dict[str, str]) -> list[str]: - resolved = [] - for arg in args: - updated = arg - for key, value in mapping.items(): - updated = updated.replace(key, value) - resolved.append(updated) - return resolved - - -def build_command( - config: dict[str, Any], - eval_prompt_path: Path, - eval_input_path: Path, - eval_schema_path: Path, - eval_output_path: Path, - repo_root: Path, - run_dir: Path, - check_id: str, -) -> tuple[list[str], int]: - cmd = config.get("cmd") or "codex" - raw_args = config.get("args") or [ - "exec", - "--output-schema", - "{eval_schema_path}", - "--output-last-message", - "{eval_output_path}", - "--color", - "never", - "--sandbox", - "read-only", - "-C", - "{repo_root}", - "-", - ] - args = normalize_args(raw_args) - mapping = { - "{eval_prompt_path}": str(eval_prompt_path), - "{eval_input_path}": str(eval_input_path), - "{eval_schema_path}": str(eval_schema_path), - "{eval_output_path}": str(eval_output_path), - "{run_dir}": str(run_dir), - "{repo_root}": str(repo_root), - "{check_id}": check_id, - } - args = substitute_args(args, mapping) - timeout = int(config.get("timeout_seconds") or 600) - return [cmd] + args, timeout - - -def extract_json_blob(text: str) -> str | None: - stripped = text.strip() - if stripped.startswith("{") and stripped.endswith("}"): - return stripped - start = stripped.find("{") - end = stripped.rfind("}") - if start != -1 and end != -1 and end > start: - return stripped[start : end + 1] - return None - - -def valid_llm_result(result: dict) -> bool: - if not isinstance(result, dict): - return False - if result.get("status") not in VALID_STATUSES: - return False - if "rationale" not in result or "confidence" not in result: - return False - if not isinstance(result.get("evidence_quotes"), list): - return False - if not isinstance(result.get("recommendations"), list): - return False - return True - - -def fallback_result(reason: str) -> dict: - return { - "status": "WARN", - "rationale": reason, - "evidence_quotes": [], - "recommendations": ["Re-run the evaluator with json_fix prompt."], - "confidence": 0.0, - } - - -def run_eval_command( - cmd: list[str], cwd: Path, env: dict, timeout: int, input_text: str | None = None -) -> tuple[int | None, str]: - try: - result = subprocess.run( - cmd, - cwd=str(cwd), - env=env, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - input=input_text, - timeout=timeout, - ) - except FileNotFoundError as exc: - return None, f"Command not found: {exc}\n" - except subprocess.TimeoutExpired: - return None, f"Command timed out after {timeout} seconds.\n" - return result.returncode, result.stdout or "" - - -def build_stdin_prompt(prompt_path: Path, input_path: Path) -> str: - prompt_text = prompt_path.read_text(encoding="utf-8").rstrip() - input_text = input_path.read_text(encoding="utf-8").strip() - return f"{prompt_text}\n\nInput JSON:\n{input_text}\n" - - -def prepare_codex_env(repo_root: Path, env: dict) -> dict: - codex_home = repo_root / ".codex-home" - cache_root = codex_home / ".cache" / "codex" - cache_root.mkdir(parents=True, exist_ok=True) - env = dict(env) - env["HOME"] = str(codex_home) - env.setdefault("XDG_CACHE_HOME", str(codex_home / ".cache")) - env.setdefault("CODEX_NO_UPDATE", "1") - return env - - -def parse_llm_output(raw: str) -> dict | None: - candidate = raw.strip() - for text in [candidate, extract_json_blob(candidate)]: - if not text: - continue - try: - parsed = json.loads(text) - except json.JSONDecodeError: - continue - if valid_llm_result(parsed): - return parsed - return None - - -def render_json_fix_prompt(prompt_text: str, raw_output: str) -> str: - return prompt_text.replace("{{RAW_OUTPUT}}", raw_output) - - -def build_eval_input(run_dir: Path, check_id: str) -> dict: - evidence_path = run_dir / "evidence.json" - evidence = load_json(evidence_path) if evidence_path.exists() else {} - agentic_summary_path = run_dir / "agentic_summary.json" - agentic_summary = load_json(agentic_summary_path) if agentic_summary_path.exists() else None - execution_summary_path = run_dir / "execution_summary.json" - execution_summary = ( - load_json(execution_summary_path) if execution_summary_path.exists() else None - ) - - return { - "check_id": check_id, - "prompt": (evidence.get("prompt_json") or {}).get("content") or {}, - "evidence": evidence, - "git_diff": evidence.get("git_diff", ""), - "execution_summary": execution_summary, - "agentic_summary": agentic_summary, - } - - -def run_single_eval( - run_dir: Path, - repo_root: Path, - check_id: str, - prompt_path: Path, - config: dict[str, Any], - prompts_dir: Path, -) -> dict: - logs_dir = run_dir / "logs" - logs_dir.mkdir(parents=True, exist_ok=True) - - input_payload = build_eval_input(run_dir, check_id) - eval_input_path = run_dir / f"llm_input_{check_id}.json" - eval_input_path.write_text(json.dumps(input_payload, indent=2), encoding="utf-8") - - eval_schema_path = prompts_dir / "llm_eval_schema.json" - eval_output_path = run_dir / f"llm_output_{check_id}.json" - - cmd, timeout = build_command( - config, - prompt_path, - eval_input_path, - eval_schema_path, - eval_output_path, - repo_root, - run_dir, - check_id, - ) - env = prepare_codex_env(repo_root, os.environ.copy()) - input_text = build_stdin_prompt(prompt_path, eval_input_path) if "-" in cmd else None - exit_code, output = run_eval_command(cmd, repo_root, env, timeout, input_text=input_text) - log_path = logs_dir / f"llm_eval_{check_id}.log" - log_path.write_text(output, encoding="utf-8") - - output_text = ( - eval_output_path.read_text(encoding="utf-8") if eval_output_path.exists() else output - ) - parsed = parse_llm_output(output_text) - if parsed: - return parsed - - json_fix_path = prompts_dir / "json_fix.md" - json_fix_prompt = json_fix_path.read_text(encoding="utf-8") if json_fix_path.exists() else "" - if not json_fix_prompt: - return fallback_result( - "LLM evaluator returned invalid JSON and json_fix prompt is missing." - ) - - fix_prompt_path = run_dir / f"json_fix_{check_id}.md" - fix_prompt_path.write_text(render_json_fix_prompt(json_fix_prompt, output), encoding="utf-8") - fix_input_path = run_dir / f"json_fix_input_{check_id}.json" - fix_input_path.write_text(json.dumps({"raw_output": output}, indent=2), encoding="utf-8") - - fix_output_path = run_dir / f"llm_output_{check_id}_fix.json" - cmd, timeout = build_command( - config, - fix_prompt_path, - fix_input_path, - eval_schema_path, - fix_output_path, - repo_root, - run_dir, - check_id, - ) - fix_input_text = build_stdin_prompt(fix_prompt_path, fix_input_path) if "-" in cmd else None - exit_code, fix_output = run_eval_command( - cmd, repo_root, env, timeout, input_text=fix_input_text - ) - log_path = logs_dir / f"llm_eval_{check_id}_fix.log" - log_path.write_text(fix_output, encoding="utf-8") - - fix_output_text = ( - fix_output_path.read_text(encoding="utf-8") if fix_output_path.exists() else fix_output - ) - parsed = parse_llm_output(fix_output_text) - if parsed: - return parsed - - return fallback_result("LLM evaluator returned invalid JSON after json_fix.") - - -def main() -> int: - parser = argparse.ArgumentParser(description="Run automatic LLM evaluations via Codex CLI.") - parser.add_argument( - "--out-dir", default=".codex-readiness-integration-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument( - "--checks", - default=str(Path(__file__).resolve().parents[1] / "references" / "checks.json"), - ) - args = parser.parse_args() - - base_dir = Path(args.out_dir) - run_dir = resolve_run_dir(base_dir, args.run_dir) - run_dir.mkdir(parents=True, exist_ok=True) - repo_root = Path.cwd() - - prompt_path = run_dir / "prompt.json" - prompt_config = load_json(prompt_path) if prompt_path.exists() else {} - raw_llm_config = prompt_config.get("llm_eval") - llm_config: dict[str, Any] = raw_llm_config if isinstance(raw_llm_config, dict) else {} - - checks_data = load_json(Path(args.checks)) - prompts_dir = Path(__file__).resolve().parents[1] / "references" - - results = {} - for check in sort_checks_by_priority(checks_data.get("checks", [])): - if not check.get("enabled_by_default"): - continue - if check.get("type") != "LLM": - continue - check_id = check.get("id") - prompt_id = check.get("evaluator_prompt_id") - if not prompt_id: - continue - prompt_file = prompts_dir / f"{prompt_id}.md" - if not prompt_file.exists(): - results[check_id] = fallback_result(f"Evaluator prompt missing: {prompt_file}") - continue - results[check_id] = run_single_eval( - run_dir, repo_root, check_id, prompt_file, llm_config, prompts_dir - ) - - llm_path = run_dir / "llm_results.json" - write_json(llm_path, results) - print(str(llm_path)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/skills/.experimental/codex-readiness-integration-test/scripts/run_plan.py b/skills/.experimental/codex-readiness-integration-test/scripts/run_plan.py deleted file mode 100644 index 23438ea..0000000 --- a/skills/.experimental/codex-readiness-integration-test/scripts/run_plan.py +++ /dev/null @@ -1,277 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import io -import json -import os -import re -import selectors -import subprocess -import sys -import time -from datetime import datetime, timezone -from pathlib import Path -from typing import Any, TypedDict - -DENYLIST_PATTERNS = [ - r"\brm\s+-rf\b", - r"\brm\s+-fr\b", - r"\brm\s+-r\b", - r"\bgit\s+clean\s+-xfd\b", - r"\bmkfs\b", - r"\bdd\s+if=", - r"\bdiskutil\s+erase\b", - r"\b:;\s*\b", # basic fork bomb patterns - r"\bmkfs\.[a-z0-9]+\b", -] - - -class PlanCommand(TypedDict, total=False): - label: str - cmd: str - timeout_soft_seconds: int - timeout_hard_seconds: int - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = load_json(latest_path) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "evidence.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def now_iso() -> str: - return datetime.now(timezone.utc).isoformat() - - -def is_denylisted(cmd: str) -> bool: - lower = cmd.lower() - return any(re.search(pattern, lower) for pattern in DENYLIST_PATTERNS) - - -def normalize_plan(plan_data: dict[str, Any]) -> dict[str, Any]: - if "commands" not in plan_data: - raise ValueError("Plan JSON must include 'commands'.") - commands: list[PlanCommand] = [] - for entry in plan_data.get("commands", []): - if isinstance(entry, str): - command_str: PlanCommand = {"label": "step", "cmd": entry} - commands.append(command_str) - elif isinstance(entry, dict): - cmd = entry.get("cmd") - if not cmd: - raise ValueError("Each command entry must include 'cmd'.") - command: PlanCommand = { - "label": entry.get("label") or "step", - "cmd": cmd, - } - soft = entry.get("timeout_soft_seconds") - hard = entry.get("timeout_hard_seconds") - if soft is not None: - command["timeout_soft_seconds"] = int(soft) - if hard is not None: - command["timeout_hard_seconds"] = int(hard) - commands.append(command) - else: - raise ValueError("Commands must be strings or objects with 'cmd'.") - plan_data["commands"] = commands - return plan_data - - -def run_command( - cmd: str, cwd: Path, env: dict, soft_timeout: int, hard_timeout: int, log_path: Path -) -> dict: - started_at = now_iso() - start_time = time.time() - soft_exceeded = False - hard_exceeded = False - exit_code = None - - with log_path.open("w", encoding="utf-8") as log_file: - proc = subprocess.Popen( - cmd, - shell=True, - cwd=str(cwd), - env=env, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - selector = selectors.DefaultSelector() - if proc.stdout: - selector.register(proc.stdout, selectors.EVENT_READ) - - while True: - now = time.time() - if not soft_exceeded and now - start_time > soft_timeout: - soft_exceeded = True - if now - start_time > hard_timeout: - hard_exceeded = True - proc.terminate() - try: - proc.wait(timeout=5) - except subprocess.TimeoutExpired: - proc.kill() - break - events = selector.select(timeout=0.2) - for key, _ in events: - file_obj = key.fileobj - if isinstance(file_obj, io.TextIOBase): - line = file_obj.readline() - if line: - log_file.write(line) - if proc.poll() is not None: - break - - # Drain remaining output - if proc.stdout: - for line in proc.stdout: - log_file.write(line) - - exit_code = proc.returncode - - ended_at = now_iso() - duration = time.time() - start_time - - if hard_exceeded: - status = "FAIL" - elif exit_code == 0: - status = "WARN" if soft_exceeded else "PASS" - else: - status = "FAIL" - - return { - "cmd": cmd, - "status": status, - "exit_code": exit_code, - "duration_seconds": round(duration, 2), - "soft_timeout_seconds": soft_timeout, - "hard_timeout_seconds": hard_timeout, - "soft_timeout_exceeded": soft_exceeded, - "hard_timeout_exceeded": hard_exceeded, - "log_path": str(log_path), - "started_at": started_at, - "ended_at": ended_at, - } - - -def sanitize_label(label: str) -> str: - cleaned = re.sub(r"[^a-zA-Z0-9_.-]+", "-", label.strip().lower()) - return cleaned.strip("-") or "step" - - -def main() -> int: - parser = argparse.ArgumentParser(description="Execute a documented dev/build/test plan.") - parser.add_argument("--plan", required=True, help="Path to plan JSON") - parser.add_argument( - "--out-dir", - default=".codex-readiness-integration-test", - help="Base output directory", - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument( - "--soft-timeout-seconds", type=int, default=600, help="Soft timeout per command" - ) - parser.add_argument( - "--hard-timeout-multiplier", type=int, default=3, help="Hard timeout multiplier" - ) - args = parser.parse_args() - - plan_path = Path(args.plan) - if not plan_path.exists(): - raise SystemExit(f"Plan file not found: {plan_path}") - - plan_data = normalize_plan(load_json(plan_path)) - - cwd = Path(plan_data.get("cwd") or plan_data.get("project_dir") or Path.cwd()) - if not cwd.is_absolute(): - cwd = (Path.cwd() / cwd).resolve() - - env = os.environ.copy() - env.update(plan_data.get("env", {})) - - base_dir = Path(args.out_dir) - base_dir.mkdir(parents=True, exist_ok=True) - run_dir = resolve_run_dir(base_dir, args.run_dir) - run_dir.mkdir(parents=True, exist_ok=True) - logs_dir = run_dir / "logs" - logs_dir.mkdir(parents=True, exist_ok=True) - - steps = [] - for index, entry in enumerate(plan_data.get("commands", []), start=1): - label = entry.get("label") or f"step-{index}" - cmd = entry.get("cmd", "").strip() - soft_timeout = entry.get("timeout_soft_seconds") or args.soft_timeout_seconds - hard_timeout = entry.get("timeout_hard_seconds") or ( - soft_timeout * args.hard_timeout_multiplier - ) - - log_path = logs_dir / f"{index:02d}-{sanitize_label(label)}.log" - if is_denylisted(cmd): - steps.append( - { - "label": label, - "cmd": cmd, - "status": "FAIL", - "exit_code": None, - "duration_seconds": 0, - "soft_timeout_seconds": soft_timeout, - "hard_timeout_seconds": hard_timeout, - "soft_timeout_exceeded": False, - "hard_timeout_exceeded": False, - "denylisted": True, - "log_path": str(log_path), - "started_at": now_iso(), - "ended_at": now_iso(), - } - ) - continue - - result = run_command(cmd, cwd, env, soft_timeout, hard_timeout, log_path) - result["label"] = label - result["denylisted"] = False - steps.append(result) - - overall_status = "PASS" - for step in steps: - if step["status"] == "FAIL": - overall_status = "FAIL" - break - if step["status"] == "WARN": - overall_status = "WARN" - - summary = { - "plan_path": str(plan_path), - "project_dir": str(plan_data.get("project_dir", cwd)), - "cwd": str(cwd), - "soft_timeout_seconds": args.soft_timeout_seconds, - "hard_timeout_multiplier": args.hard_timeout_multiplier, - "steps": steps, - "overall_status": overall_status, - "started_at": steps[0]["started_at"] if steps else now_iso(), - "ended_at": steps[-1]["ended_at"] if steps else now_iso(), - } - - summary_path = run_dir / "execution_summary.json" - summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8") - print(str(summary_path)) - - return 0 if overall_status == "PASS" else 1 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/skills/.experimental/codex-readiness-integration-test/scripts/scoring.py b/skills/.experimental/codex-readiness-integration-test/scripts/scoring.py deleted file mode 100644 index 6fbc6b0..0000000 --- a/skills/.experimental/codex-readiness-integration-test/scripts/scoring.py +++ /dev/null @@ -1,456 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -from decimal import ROUND_HALF_UP, Decimal -from pathlib import Path -from typing import Any - -VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"} -PRIORITY_MULTIPLIERS = {0: 4, 1: 3, 2: 2, 3: 1} - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = load_json(latest_path) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "evidence.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def round_half_up(value: float) -> int: - return int(Decimal(value).quantize(Decimal("1"), rounding=ROUND_HALF_UP)) - - -def status_points(status: str) -> float: - if status == "PASS": - return 1.0 - if status == "WARN": - return 0.5 - return 0.0 - - -def normalize_priority(value) -> int: - try: - priority = int(value) - except (TypeError, ValueError): - priority = 3 - if priority in PRIORITY_MULTIPLIERS: - return priority - return 3 - - -def sort_checks_by_priority(checks: list[dict]) -> list[dict]: - return sorted( - checks, - key=lambda check: ( - normalize_priority(check.get("priority")), - check.get("id", ""), - ), - ) - - -def priority_label(priority: int) -> str: - return f"P{priority}" - - -def validate_result(result: dict) -> dict | None: - if not isinstance(result, dict): - return None - if result.get("status") not in VALID_STATUSES: - return None - if ( - "rationale" not in result - or "evidence_quotes" not in result - or "recommendations" not in result - or "confidence" not in result - ): - return None - if not isinstance(result.get("evidence_quotes"), list): - return None - if not isinstance(result.get("recommendations"), list): - return None - return result - - -def fallback_invalid_json() -> dict: - return { - "status": "WARN", - "rationale": "Invalid JSON from evaluator after retries.", - "evidence_quotes": [], - "recommendations": ["Re-run the evaluator with the json_fix prompt."], - "confidence": 0.0, - } - - -def build_weights(checks: list[dict]) -> dict: - enabled = [c for c in checks if c.get("enabled_by_default")] - raw_weights = [] - for check in enabled: - weight = check.get("weight") - base_weight = weight if isinstance(weight, (int, float)) else 1.0 - priority = normalize_priority(check.get("priority")) - raw_weights.append(base_weight * PRIORITY_MULTIPLIERS[priority]) - total = sum(raw_weights) if raw_weights else 1.0 - weights = {} - for check, raw in zip(enabled, raw_weights): - weights[check["id"]] = (raw / total) * 100.0 - return weights - - -def build_results( - checks: list[dict], - deterministic_results: dict, - llm_results: dict, - execution_summary: dict | None, -) -> dict: - results = {} - execution_status = None - if execution_summary: - execution_status = execution_summary.get("overall_status") - - for check in checks: - if not check.get("enabled_by_default"): - continue - check_id = check["id"] - check_type = check.get("type") - - if check_type == "DETERMINISTIC": - result = deterministic_results.get(check_id) - if result: - valid = validate_result(result) - results[check_id] = valid if valid else fallback_invalid_json() - else: - results[check_id] = { - "status": "FAIL", - "rationale": "Deterministic result missing for this check.", - "evidence_quotes": [], - "recommendations": ["Run deterministic_rules.py to populate results."], - "confidence": 0.0, - } - continue - - if check_type == "LLM": - llm_result = llm_results.get(check_id) - if llm_result: - valid = validate_result(llm_result) - results[check_id] = valid if valid else fallback_invalid_json() - else: - results[check_id] = { - "status": "WARN", - "rationale": "LLM evaluation missing for this check.", - "evidence_quotes": [], - "recommendations": ["Run the evaluator prompt for this check."], - "confidence": 0.0, - } - continue - - if check_type == "HYBRID": - status = ( - execution_status - or deterministic_results.get(check_id, {}).get("status") - or "NOT_RUN" - ) - llm_result = llm_results.get(check_id) - if llm_result: - valid = validate_result(llm_result) or fallback_invalid_json() - valid["status"] = status - results[check_id] = valid - else: - results[check_id] = { - "status": status, - "rationale": "Execution summary present but LLM rationale missing." - if status != "NOT_RUN" - else "Execution not run.", - "evidence_quotes": [], - "recommendations": ["Provide execution rationale using the evaluator prompt."], - "confidence": 0.0, - } - continue - - return results - - -def render_html(report: dict, prompt: dict | None = None, report_path: Path | None = None) -> str: - score = report["scorecard"]["score_total"] - status = report["scorecard"]["overall_status"] - results = report.get("results", {}) - enabled_checks = report.get("enabled_checks", []) - checks_by_id = {check["id"]: check for check in enabled_checks} - prompt = prompt or {} - change_prompt = prompt.get("change_prompt") or "" - acceptance_criteria = prompt.get("acceptance_criteria") or [] - - def status_class(value: str) -> str: - return value.lower() - - html = [ - "", - "", - "", - "", - "Codex Readiness Integration Test Report", - "", - "", - "", - "

Codex Readiness Integration Test Report

", - ] - if report_path: - html.append(f"

{report_path}

") - html.append( - f"

Overall score: {score} {status}

" - ) - - if change_prompt: - html.extend( - [ - "

Prompt

", - "", - "", - f"", - "
Change prompt
{change_prompt}
", - ] - ) - - if isinstance(acceptance_criteria, list) and acceptance_criteria: - html.extend( - [ - "

Acceptance criteria

", - "", - "", - ] - ) - for item in acceptance_criteria: - html.append(f"") - html.append("
Criteria
{item}
") - - html.append("

Checks

") - - html.append("") - html.append("") - for check_id, result in results.items(): - check = checks_by_id.get(check_id, {}) - title = check.get("title", check_id) - status_value = result.get("status", "NOT_RUN") - rationale = result.get("rationale", "") - html.append( - "" - f"" - f"" - f"" - "" - ) - html.append("
CheckStatusRationale
{title}{status_value}{rationale}
") - html.append("") - return "\n".join(html) - - -def summarize_diff(git_diff: str) -> dict[str, Any]: - files: list[str] = [] - additions = 0 - deletions = 0 - for line in git_diff.splitlines(): - if line.startswith("diff --git "): - parts = line.split() - if len(parts) >= 4: - left = parts[2].removeprefix("a/") - right = parts[3].removeprefix("b/") - if left and left not in files: - files.append(left) - if right and right not in files: - files.append(right) - continue - if line.startswith(("+++ ", "--- ")): - continue - if line.startswith("+"): - additions += 1 - elif line.startswith("-"): - deletions += 1 - return {"files": files, "additions": additions, "deletions": deletions} - - -def truncate_text(text: str, limit: int = 200) -> str: - if len(text) <= limit: - return text - return text[: limit - 3] + "..." - - -def render_summary_text( - report: dict, - evidence: dict, - deterministic_results: dict, - llm_results: dict, - execution_summary: dict | None, - agentic_summary: dict | None, - summary_path: Path, -) -> str: - lines = [ - "# Codex Readiness Integration Test Report", - f"## {summary_path}", - "", - ] - score = report["scorecard"]["score_total"] - status = report["scorecard"]["overall_status"] - lines.append(f"Overall: {status} (score {score})") - - prompt = (evidence.get("prompt_json") or {}).get("content") or {} - change_prompt = prompt.get("change_prompt") or "" - if change_prompt: - lines.append(f"Prompt: {truncate_text(change_prompt)}") - - if agentic_summary: - agentic_status = agentic_summary.get("status", "NOT_RUN") - exit_code = agentic_summary.get("exit_code") - duration = agentic_summary.get("duration_seconds") - lines.append( - f"Agentic loop: {agentic_status} (exit_code {exit_code}, duration {duration}s)" - ) - else: - lines.append("Agentic loop: NOT_RUN") - - diff_stats = summarize_diff(evidence.get("git_diff", "")) - if diff_stats.get("files"): - lines.append( - f"Diff: {len(diff_stats['files'])} file(s), +{diff_stats['additions']}/-{diff_stats['deletions']} lines" - ) - - path_check = deterministic_results.get("repo_root_only_changes", {}).get("status") - if path_check: - lines.append(f"Path policy: {path_check}") - - test_status = execution_summary.get("overall_status") if execution_summary else "NOT_RUN" - lines.append(f"Tests: {test_status}") - - agentic_eval = llm_results.get("agentic_loop_eval", {}).get("status", "NOT_RUN") - change_eval = llm_results.get("change_quality_eval", {}).get("status", "NOT_RUN") - lines.append(f"LLM eval: agentic_loop_eval={agentic_eval}, change_quality_eval={change_eval}") - - if agentic_summary: - questions = agentic_summary.get("clarifying_questions") or [] - if questions: - last_question = truncate_text(str(questions[-1])) - lines.append(f"Clarifying questions: {len(questions)}") - lines.append(f"Last question: {last_question}") - - return "\n".join(lines) + "\n" - - -def main() -> int: - parser = argparse.ArgumentParser(description="Score integration test results.") - parser.add_argument( - "--out-dir", default=".codex-readiness-integration-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument( - "--checks", - default=str(Path(__file__).resolve().parents[1] / "references" / "checks.json"), - ) - args = parser.parse_args() - - base_dir = Path(args.out_dir) - run_dir = resolve_run_dir(base_dir, args.run_dir) - - checks_data = load_json(Path(args.checks)) - checks = sort_checks_by_priority(checks_data.get("checks", [])) - - deterministic_results = ( - load_json(run_dir / "deterministic_results.json") - if (run_dir / "deterministic_results.json").exists() - else {} - ) - llm_results = ( - load_json(run_dir / "llm_results.json") if (run_dir / "llm_results.json").exists() else {} - ) - execution_summary = ( - load_json(run_dir / "execution_summary.json") - if (run_dir / "execution_summary.json").exists() - else None - ) - evidence = load_json(run_dir / "evidence.json") if (run_dir / "evidence.json").exists() else {} - agentic_summary = ( - load_json(run_dir / "agentic_summary.json") - if (run_dir / "agentic_summary.json").exists() - else None - ) - - weights = build_weights(checks) - results = build_results(checks, deterministic_results, llm_results, execution_summary) - - score_items = [] - for check_id, result in results.items(): - weight = weights.get(check_id, 0) - score_items.append(weight * status_points(result.get("status", "NOT_RUN"))) - score_total = round_half_up(sum(score_items)) - - overall_status = "PASS" - for result in results.values(): - if result["status"] == "FAIL": - overall_status = "FAIL" - break - if result["status"] in {"WARN", "NOT_RUN"}: - overall_status = "WARN" - - report = { - "scorecard": { - "score_total": score_total, - "overall_status": overall_status, - "weights": weights, - }, - "enabled_checks": [check for check in checks if check.get("enabled_by_default")], - "results": results, - } - - report_path = run_dir / "report.json" - report_path.write_text(json.dumps(report, indent=2), encoding="utf-8") - - prompt_content = (evidence.get("prompt_json") or {}).get("content") or {} - html_path = run_dir / "report.html" - html_path.write_text(render_html(report, prompt_content, html_path), encoding="utf-8") - - summary = { - "overall_status": overall_status, - "score_total": score_total, - } - summary_path = run_dir / "summary.json" - summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8") - - summary_text_path = run_dir / "summary.txt" - summary_text = render_summary_text( - report, - evidence, - deterministic_results, - llm_results, - execution_summary, - agentic_summary, - summary_text_path, - ) - summary_text_path.write_text(summary_text, encoding="utf-8") - print(summary_text) - - print(str(report_path)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/skills/.experimental/codex-readiness-integration-test/tests/test_planning_signal_rule.py b/skills/.experimental/codex-readiness-integration-test/tests/test_planning_signal_rule.py deleted file mode 100644 index 9996059..0000000 --- a/skills/.experimental/codex-readiness-integration-test/tests/test_planning_signal_rule.py +++ /dev/null @@ -1,100 +0,0 @@ -from __future__ import annotations - -import importlib.util -import json -from pathlib import Path - - -def load_rules_module(): - module_path = Path(__file__).resolve().parents[1] / "scripts" / "deterministic_rules.py" - spec = importlib.util.spec_from_file_location("deterministic_rules", module_path) - if spec is None or spec.loader is None: - raise RuntimeError(f"Unable to load deterministic_rules module from {module_path}") - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - return module - - -RULES = load_rules_module() - - -def write_run_dir(run_dir: Path, log_text: str) -> None: - (run_dir / "logs").mkdir(parents=True, exist_ok=True) - (run_dir / "prompt.json").write_text(json.dumps({"change_prompt": "x"}), encoding="utf-8") - (run_dir / "logs" / "agentic.log").write_text(log_text, encoding="utf-8") - - -def check(run_dir: Path) -> dict: - return RULES.check_exec_plan_before_code_changes( - run_dir, - {"prompt_path": "prompt.json", "agentic_log_path": "logs/agentic.log"}, - ) - - -def test_planning_signal_before_code_change_passes(tmp_path: Path) -> None: - run_dir = tmp_path / "run-pass" - log_text = "\n".join( - [ - "Plan: inspect code paths", - "file update", - "M src/app.py", - ] - ) - write_run_dir(run_dir, log_text) - - result = check(run_dir) - - assert result["status"] == "PASS" - assert "before code changes" in result["rationale"] - - -def test_planning_signal_after_code_change_fails(tmp_path: Path) -> None: - run_dir = tmp_path / "run-fail-ordering" - log_text = "\n".join( - [ - "file update", - "M src/app.py", - "Plan: now I will describe the approach", - ] - ) - write_run_dir(run_dir, log_text) - - result = check(run_dir) - - assert result["status"] == "FAIL" - assert "after code changes" in result["rationale"] - - -def test_plan_file_path_does_not_count_as_planning_signal(tmp_path: Path) -> None: - run_dir = tmp_path / "run-plan-path" - log_text = "\n".join( - [ - "file update", - "M docs/exec-plan.md", - "file update", - "M src/app.py", - ] - ) - write_run_dir(run_dir, log_text) - - result = check(run_dir) - - assert result["status"] == "FAIL" - assert "No planning signal detected" in result["rationale"] - - -def test_no_code_changes_warns(tmp_path: Path) -> None: - run_dir = tmp_path / "run-warn-no-code" - log_text = "\n".join( - [ - "Plan: investigate the issue", - "file update", - "M docs/notes.md", - ] - ) - write_run_dir(run_dir, log_text) - - result = check(run_dir) - - assert result["status"] == "WARN" - assert "ordering not evaluated" in result["rationale"] diff --git a/skills/.experimental/codex-readiness-integration-test/tests/test_verification_after_code_changes.py b/skills/.experimental/codex-readiness-integration-test/tests/test_verification_after_code_changes.py deleted file mode 100644 index af01f40..0000000 --- a/skills/.experimental/codex-readiness-integration-test/tests/test_verification_after_code_changes.py +++ /dev/null @@ -1,120 +0,0 @@ -from __future__ import annotations - -import importlib.util -import json -from pathlib import Path - - -def load_rules_module(): - module_path = Path(__file__).resolve().parents[1] / "scripts" / "deterministic_rules.py" - spec = importlib.util.spec_from_file_location("deterministic_rules", module_path) - if spec is None or spec.loader is None: - raise RuntimeError(f"Unable to load deterministic_rules module from {module_path}") - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - return module - - -RULES = load_rules_module() - - -def write_run_dir(run_dir: Path, log_text: str, build_test_plan: list[dict] | None = None) -> None: - (run_dir / "logs").mkdir(parents=True, exist_ok=True) - prompt = { - "change_prompt": "x", - "build_test_plan": build_test_plan or [], - } - (run_dir / "prompt.json").write_text(json.dumps(prompt), encoding="utf-8") - (run_dir / "logs" / "agentic.log").write_text(log_text, encoding="utf-8") - - -def check(run_dir: Path) -> dict: - return RULES.check_verification_after_code_changes( - run_dir, - {"prompt_path": "prompt.json", "agentic_log_path": "logs/agentic.log"}, - ) - - -def test_verification_command_after_code_change_passes(tmp_path: Path) -> None: - run_dir = tmp_path / "run-pass" - log_text = "\n".join( - [ - "file update", - "M src/app.py", - "$ pytest -q", - ] - ) - write_run_dir(run_dir, log_text) - - result = check(run_dir) - - assert result["status"] == "PASS" - assert "after code changes" in result["rationale"] - - -def test_verification_only_before_code_change_fails(tmp_path: Path) -> None: - run_dir = tmp_path / "run-fail-order" - log_text = "\n".join( - [ - "$ pytest -q", - "file update", - "M src/app.py", - ] - ) - write_run_dir(run_dir, log_text) - - result = check(run_dir) - - assert result["status"] == "FAIL" - assert "No build/test/lint verification command detected" in result["rationale"] - - -def test_prompt_plan_command_counts_as_verification(tmp_path: Path) -> None: - run_dir = tmp_path / "run-pass-prompt-command" - build_test_plan = [{"label": "verify", "cmd": "make check-all"}] - log_text = "\n".join( - [ - "file update", - "M src/core.py", - "$ make check-all", - ] - ) - write_run_dir(run_dir, log_text, build_test_plan=build_test_plan) - - result = check(run_dir) - - assert result["status"] == "PASS" - - -def test_codex_exec_lines_do_not_count_as_verification(tmp_path: Path) -> None: - run_dir = tmp_path / "run-fail-codex-line" - log_text = "\n".join( - [ - "file update", - "M src/app.py", - '$ codex exec -C /repo "please run tests after this change"', - ] - ) - write_run_dir(run_dir, log_text) - - result = check(run_dir) - - assert result["status"] == "FAIL" - assert "No build/test/lint verification command detected" in result["rationale"] - - -def test_no_code_changes_warns(tmp_path: Path) -> None: - run_dir = tmp_path / "run-warn-no-code" - log_text = "\n".join( - [ - "file update", - "M docs/notes.md", - "$ pytest -q", - ] - ) - write_run_dir(run_dir, log_text) - - result = check(run_dir) - - assert result["status"] == "WARN" - assert "ordering not evaluated" in result["rationale"] diff --git a/skills/.experimental/codex-readiness-unit-test/LICENSE.txt b/skills/.experimental/codex-readiness-unit-test/LICENSE.txt deleted file mode 100644 index d645695..0000000 --- a/skills/.experimental/codex-readiness-unit-test/LICENSE.txt +++ /dev/null @@ -1,202 +0,0 @@ - - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright [yyyy] [name of copyright owner] - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. diff --git a/skills/.experimental/codex-readiness-unit-test/SKILL.md b/skills/.experimental/codex-readiness-unit-test/SKILL.md deleted file mode 100644 index 4f2c583..0000000 --- a/skills/.experimental/codex-readiness-unit-test/SKILL.md +++ /dev/null @@ -1,129 +0,0 @@ ---- -name: codex-readiness-unit-test -description: Run the Codex Readiness unit test report. Use when you need deterministic checks plus in-session LLM evals for AGENTS.md/PLANS.md. -metadata: - short-description: Run Codex Readiness unit test report ---- - -# LLM Codex Readiness Unit Test - -Instruction-first, in-session "readiness" for evaluating AGENTS/PLANS documentation quality without any external APIs or SDKs. All checks run against the current working directory (cwd), with no monorepo discovery. Each run writes to `.codex-readiness-unit-test//` and updates `.codex-readiness-unit-test/latest.json`. Keep execution deterministic (filesystem scanning + local command execution only). All LLM evaluation happens in-session and must output strict JSON via the provided references. - -## Quick Start - -1) Collect evidence: - - `python skills/codex-readiness-unit-test/bin/collect_evidence.py` -2) Run deterministic checks: - - `python skills/codex-readiness-unit-test/bin/deterministic_rules.py` -3) Run LLM checks using references in `references/` and store `.codex-readiness-unit-test//llm_results.json`. -4) If execute mode is requested, build a plan, get confirmation, run: - - `python skills/codex-readiness-unit-test/bin/run_plan.py --plan .codex-readiness-unit-test//plan.json` -5) Generate the report: - - `python skills/codex-readiness-unit-test/bin/scoring.py --mode read-only|execute` - -Outputs (per run, under `.codex-readiness-unit-test//`): -- `report.json` -- `report.html` -- `summary.json` -- `logs/*` (execute mode) - -## Runbook - -This skill produces a deterministic evidence file plus an in-session LLM evaluation, then compiles a JSON report and HTML scorecard. It requires no OpenAI API key and makes no external HTTP calls. - -### Minimal Inputs -- `mode`: `read-only` or `execute` (required) -- `soft_timeout_seconds`: optional (default 600) - -### Modes (Read-only vs Execute) -- **Read-only**: Collect evidence, run deterministic rules, and run LLM checks #3–#5. No commands are executed, check #6 is marked `NOT_RUN`, and no execution logs/summary are produced. -- **Execute**: Everything in read-only **plus** a confirmed `plan.json` is executed via `run_plan.py`. This enables check #6 and produces execution logs + `execution_summary.json` for scoring. - -Always ask the user which mode to run (read-only vs. execute) before proceeding. - -### Check Types -- **Deterministic**: filesystem-only checks (#1 AGENTS.md exists, #2 PLANS.md exists, #3 AGENTS.md <= 300 lines, #4 config.toml exists at repo root, repo .codex/, or user .codex/) -- **LLM**: in-session Codex evaluation (#3 project context, #4 commands, #5 loops; commands may live in AGENTS or referenced skills) -- **Hybrid**: deterministic execution + LLM rationale (#6 execution) - -Skill references are discovered from AGENTS.md via `$SkillName` or `.codex/skills/` patterns; their `SKILL.md` files are added to evidence for the LLM checks. - -All checks run relative to the current working directory and are defined in `skills/codex-readiness-unit-test/references/checks/checks.json`, weighted equally by default. Each run writes outputs to `.codex-readiness-unit-test//` and updates `.codex-readiness-unit-test/latest.json`. -The helper scripts read `.codex-readiness-unit-test/latest.json` by default to locate the latest run directory. - -### Strict JSON + Retry Loop (Required) -For each LLM/HYBRID check: -1) Run the specialized prompt expecting **strict JSON**. -2) If JSON is invalid or missing keys, run `skills/codex-readiness-unit-test/references/json_fix.md` with the raw output. -3) Retry up to **2 additional attempts** (max 3 total). -4) If still invalid: mark the check as **WARN** with rationale: "Invalid JSON from evaluator after retries". - -The JSON schema is: -```json -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...","quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} -``` - -### Single Confirmation (Required) -Combine the command summary and execute plan into **one** concise confirmation step. Present: -- The extracted build/test/dev loop commands (human-readable, labeled). -- The planned execute details (cwd, ordered commands, soft timeout policy, env). -Ask for a single confirmation to proceed. **Do not** paste raw JSON, full evidence, or the full `plan.json`. If declined, mark execute-required checks as `NOT_RUN`. - -### Required Files -- `.codex-readiness-unit-test//evidence.json` (from `collect_evidence.py`) -- `.codex-readiness-unit-test//deterministic_results.json` (from `deterministic_rules.py`) -- `.codex-readiness-unit-test//llm_results.json` (from in-session references) -- `.codex-readiness-unit-test//execution_summary.json` (execute mode only) -- `.codex-readiness-unit-test//report.json` and `.codex-readiness-unit-test//report.html` (from `scoring.py`) -- `.codex-readiness-unit-test//summary.json` (structured pass/fail summary from `scoring.py`) -- `.codex-readiness-unit-test/latest.json` (stable pointer to the latest run directory) - -### Prompt Mapping -- #3 `project_context_specified` → `skills/codex-readiness-unit-test/references/project_context.md` -- #4 `build_test_commands_exist` → `skills/codex-readiness-unit-test/references/commands.md` -- #5 `dev_build_test_loops_documented` → `skills/codex-readiness-unit-test/references/loop_quality.md` -- #6 `dev_build_test_loop_execution` → `skills/codex-readiness-unit-test/references/execution_explanation.md` - -### plan.json schema (execute mode) -```json -{ - "project_dir": "relative/or/absolute/path (optional)", - "cwd": "optional/absolute/path (defaults to current directory)", - "commands": [ - {"label": "setup", "cmd": "npm install"}, - {"label": "build", "cmd": "npm run build"}, - {"label": "test", "cmd": "npm test"} - ], - "env": { - "EXAMPLE": "value" - } -} -``` -Place `plan.json` inside the run directory (e.g., `.codex-readiness-unit-test//plan.json`). - -### llm_results.json schema -```json -{ - "project_context_specified": {"status":"PASS","rationale":"...","evidence_quotes":[],"recommendations":[],"confidence":0.7}, - "build_test_commands_exist": {"status":"PASS","rationale":"...","evidence_quotes":[],"recommendations":[],"confidence":0.7}, - "dev_build_test_loops_documented": {"status":"WARN","rationale":"...","evidence_quotes":[],"recommendations":[],"confidence":0.6}, - "dev_build_test_loop_execution": {"status":"PASS","rationale":"...","evidence_quotes":[],"recommendations":[],"confidence":0.6} -} -``` - -### Scoring Rules -- PASS = 100% of weight -- WARN = 50% of weight -- FAIL/NOT_RUN = 0% -- Overall status: FAIL if any FAIL; else WARN if any WARN or NOT_RUN; else PASS. - -### Safety + Timeouts -- Denylisted commands are **not executed** and marked FAIL. -- Soft timeout defaults to 600s; hard cap defaults to 3x soft timeout. -- Execution logs are written to `.codex-readiness-unit-test//logs/`. diff --git a/skills/.experimental/codex-readiness-unit-test/references/checks/checks.json b/skills/.experimental/codex-readiness-unit-test/references/checks/checks.json deleted file mode 100644 index 897012a..0000000 --- a/skills/.experimental/codex-readiness-unit-test/references/checks/checks.json +++ /dev/null @@ -1,119 +0,0 @@ -{ - "schema_version": "1.0", - "checks": [ - { - "id": "agents_md_exists", - "title": "AGENTS.md exists (cwd)", - "description": "Verify AGENTS.md exists in the current working directory.", - "priority": 0, - "weight": null, - "type": "DETERMINISTIC", - "scope": "cwd", - "execute_required": false, - "evaluator_prompt_id": null, - "deterministic_rule_id": "agents_exists", - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "plans_md_exists", - "title": "Referenced planning markdown exists", - "description": "Verify AGENTS.md references a planning markdown file (plan-named or with planning headings) and the referenced file exists.", - "priority": 0, - "weight": null, - "type": "DETERMINISTIC", - "scope": "cwd", - "execute_required": false, - "evaluator_prompt_id": null, - "deterministic_rule_id": "plans_reference_exists", - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "agents_md_under_300_lines", - "title": "AGENTS.md is under 300 lines", - "description": "AGENTS.md must be 300 lines or fewer.", - "priority": 1, - "weight": null, - "type": "DETERMINISTIC", - "scope": "cwd", - "execute_required": false, - "evaluator_prompt_id": null, - "deterministic_rule_id": "agents_line_count_under_300", - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "config_toml_exists", - "title": "config.toml exists", - "description": "config.toml exists at repo root, repo .codex/config.toml, or ~/.codex/config.toml.", - "priority": 0, - "weight": null, - "type": "DETERMINISTIC", - "scope": "cwd", - "execute_required": false, - "evaluator_prompt_id": null, - "deterministic_rule_id": "config_toml_exists", - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "project_context_specified", - "title": "Project directory context is clear", - "description": "AGENTS docs include explicit project paths with short context for what each path is used for.", - "priority": 0, - "weight": null, - "type": "LLM", - "scope": "cwd", - "execute_required": false, - "evaluator_prompt_id": "project_context", - "deterministic_rule_id": null, - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "build_test_commands_exist", - "title": "Build/test commands are copy-pastable", - "description": "AGENTS docs or referenced skills provide concrete build/test commands with no placeholders.", - "priority": 0, - "weight": null, - "type": "LLM", - "scope": "cwd", - "execute_required": false, - "evaluator_prompt_id": "commands", - "deterministic_rule_id": null, - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "dev_build_test_loops_documented", - "title": "Dev/build/test loop is documented", - "description": "AGENTS docs describe ordering, when to run, and success criteria for one or more dev/build/test loops.", - "priority": 0, - "weight": null, - "type": "LLM", - "scope": "cwd", - "execute_required": false, - "evaluator_prompt_id": "loop_quality", - "deterministic_rule_id": null, - "deterministic_rule_params": {}, - "enabled_by_default": true - }, - { - "id": "dev_build_test_loop_execution", - "title": "Dev/build/test loop executes", - "description": "Execute documented loop (setup/dev/build/test) and validate outcomes.", - "priority": 0, - "weight": null, - "type": "HYBRID", - "scope": "cwd", - "execute_required": true, - "evaluator_prompt_id": "execution_explanation", - "deterministic_rule_id": "execution_summary_status", - "deterministic_rule_params": { - "summary_path": "execution_summary.json" - }, - "enabled_by_default": true - } - ] -} diff --git a/skills/.experimental/codex-readiness-unit-test/references/commands.md b/skills/.experimental/codex-readiness-unit-test/references/commands.md deleted file mode 100644 index a504d4b..0000000 --- a/skills/.experimental/codex-readiness-unit-test/references/commands.md +++ /dev/null @@ -1,29 +0,0 @@ -Evaluate whether the AGENTS documentation provides concrete, copy-pastable build/test commands for the current working directory. - -Definition of PASS: -- Commands are explicit and copy-pastable (no placeholders like "run unit tests"). -- Commands appear runnable from the project directory unless AGENTS states otherwise. -- Multiple commands are acceptable (unit/integration/build). - -Definition of FAIL: -- No build/test commands, or only vague prose without actual commands. - -Use evidence snippets in the evidence JSON. - -Return STRICT JSON only with this schema: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...", "quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- Evidence quotes must be exact excerpts from files; keep each quote short (<240 chars). -- No patches or diffs in recommendations. -- If commands exist but include placeholders or missing context (e.g., need extra args), use WARN. -- If no AGENTS content is available, use FAIL. - -Evidence: -{{EVIDENCE_JSON}} diff --git a/skills/.experimental/codex-readiness-unit-test/references/default.md b/skills/.experimental/codex-readiness-unit-test/references/default.md deleted file mode 100644 index 0364b86..0000000 --- a/skills/.experimental/codex-readiness-unit-test/references/default.md +++ /dev/null @@ -1,19 +0,0 @@ -You are evaluating repository onboarding quality based on provided evidence. - -Return STRICT JSON only with this schema: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...", "quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- Use only evidence provided (AGENTS/PLANS snippets and summaries). -- Evidence quotes must be exact excerpts from files; keep each quote short (<240 chars). -- No patches or diffs in recommendations. -- If evidence is missing, use WARN or FAIL and explain why. - -Evidence: -{{EVIDENCE_JSON}} diff --git a/skills/.experimental/codex-readiness-unit-test/references/execution_explanation.md b/skills/.experimental/codex-readiness-unit-test/references/execution_explanation.md deleted file mode 100644 index 6977f10..0000000 --- a/skills/.experimental/codex-readiness-unit-test/references/execution_explanation.md +++ /dev/null @@ -1,28 +0,0 @@ -You are summarizing execution results for a documented dev/build/test loop. - -The deterministic runner has already decided the status below. You MUST copy the provided status verbatim. - -Deterministic status: -{{DETERMINISTIC_STATUS}} - -Execution summary JSON: -{{EXECUTION_SUMMARY_JSON}} - -Return STRICT JSON only with this schema: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...", "quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- The status must equal the deterministic status shown above. -- Evidence quotes should come from the execution summary file path, not from AGENTS. -- Rationale must mention the executed command(s) from the execution summary (wrap them in backticks). -- Keep quotes short (<240 chars). -- No patches or diffs in recommendations. - -Execution summary path: -{{EXECUTION_SUMMARY_PATH}} diff --git a/skills/.experimental/codex-readiness-unit-test/references/json_fix.md b/skills/.experimental/codex-readiness-unit-test/references/json_fix.md deleted file mode 100644 index d963315..0000000 --- a/skills/.experimental/codex-readiness-unit-test/references/json_fix.md +++ /dev/null @@ -1,18 +0,0 @@ -You are fixing invalid JSON from a prior evaluator. - -Return ONLY valid JSON that matches this schema exactly: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...", "quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- Do not include any extra keys. -- Do not include markdown, commentary, or code fences. -- If the original content lacks evidence, keep evidence_quotes empty. - -Invalid output to fix: -{{RAW_OUTPUT}} diff --git a/skills/.experimental/codex-readiness-unit-test/references/loop_quality.md b/skills/.experimental/codex-readiness-unit-test/references/loop_quality.md deleted file mode 100644 index 770a824..0000000 --- a/skills/.experimental/codex-readiness-unit-test/references/loop_quality.md +++ /dev/null @@ -1,30 +0,0 @@ -Evaluate whether the AGENTS documentation describes the dev/build/test loop(s) for the current working directory. - -Definition of PASS: -- Documentation includes ordering (what to run first, next, last). -- It specifies when to run the loop (e.g., after each change, before PR). -- It defines success criteria (what output indicates success). -- Multiple loops are acceptable (fast vs full). - -Definition of FAIL: -- No loop guidance, or only vague statements with no ordering/criteria. - -Use evidence snippets in the evidence JSON. - -Return STRICT JSON only with this schema: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...", "quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- Evidence quotes must be exact excerpts from files; keep each quote short (<240 chars). -- No patches or diffs in recommendations. -- If ordering exists but missing when-to-run or success criteria, use WARN. -- If no AGENTS content is available, use FAIL. - -Evidence: -{{EVIDENCE_JSON}} diff --git a/skills/.experimental/codex-readiness-unit-test/references/project_context.md b/skills/.experimental/codex-readiness-unit-test/references/project_context.md deleted file mode 100644 index 4aeb0c2..0000000 --- a/skills/.experimental/codex-readiness-unit-test/references/project_context.md +++ /dev/null @@ -1,28 +0,0 @@ -Evaluate whether the AGENTS documentation in the current working directory provides clear context with explicit paths. - -Definition of PASS: -- The AGENTS docs include explicit paths to important directories/files (e.g., services/auth, src/, Makefile, pyproject.toml). -- Each path has a short context for what it is used for. - -Definition of FAIL: -- No explicit paths, or only vague prose without concrete paths. - -Use the evidence snippets in the evidence JSON. - -Return STRICT JSON only with this schema: -{ - "status": "PASS|WARN|FAIL|NOT_RUN", - "rationale": "string", - "evidence_quotes": [{"path":"...", "quote":"..."}], - "recommendations": ["..."], - "confidence": 0.0 -} - -Rules: -- Evidence quotes must be exact excerpts from files; keep each quote short (<240 chars). -- No patches or diffs in recommendations. -- If the docs contain some paths but lack context, use WARN. -- If no AGENTS content is available, use FAIL. - -Evidence: -{{EVIDENCE_JSON}} diff --git a/skills/.experimental/codex-readiness-unit-test/scripts/collect_evidence.py b/skills/.experimental/codex-readiness-unit-test/scripts/collect_evidence.py deleted file mode 100644 index 5690f51..0000000 --- a/skills/.experimental/codex-readiness-unit-test/scripts/collect_evidence.py +++ /dev/null @@ -1,324 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -import os -import re -import sys -from datetime import datetime, timezone -from pathlib import Path -from typing import Any, TypedDict, cast - -SKIP_DIRS = { - ".git", - ".codex-readiness-unit-test", - "node_modules", - "dist", - "build", - ".venv", - "venv", - "__pycache__", -} - -BUILD_SIGNAL_FILES = [ - "package.json", - "pnpm-workspace.yaml", - "yarn.lock", - "pnpm-lock.yaml", - "package-lock.json", - "pyproject.toml", - "setup.py", - "requirements.txt", - "Pipfile", - "poetry.lock", - "Makefile", - "CMakeLists.txt", - "go.mod", - "Cargo.toml", - "pom.xml", - "build.gradle", - "build.gradle.kts", - "Gemfile", - "composer.json", - "mix.exs", - "gradlew", - "tox.ini", - "pytest.ini", - "jest.config.js", - "vitest.config.ts", -] - -COMMAND_KEYWORDS = [ - "npm ", - "yarn ", - "pnpm ", - "make ", - "pytest", - "go test", - "go build", - "cargo ", - "mvn ", - "gradle ", - "./gradlew", - "bundle ", - "rake ", - "tox", - "poetry ", - "pip ", - "pipenv ", - "cmake ", -] - -SKILL_REF_PATTERN = re.compile(r"\$([A-Za-z0-9_.-]+)") -SKILL_PATH_PATTERN = re.compile( - r"(?:\.codex/skills|~/.codex/skills|/\.codex/skills|skills)/([A-Za-z0-9_.-]+)" -) - - -class SkillReference(TypedDict): - name: str - - -class SkillResolved(TypedDict): - name: str - path: str - - -class SkillsInfo(TypedDict): - roots: list[str] - referenced: list[str] - resolved: list[SkillResolved] - missing: list[SkillReference] - - -def read_text(path: Path) -> str: - try: - return path.read_text(encoding="utf-8") - except Exception: - try: - return path.read_text(encoding="utf-8", errors="ignore") - except Exception: - return "" - - -def extract_snippet(text: str, max_chars: int) -> str: - if len(text) <= max_chars: - return text - return text[: max_chars - 3] + "..." - - -def extract_candidate_commands(text: str) -> list[str]: - commands = [] - in_code_block = False - for line in text.splitlines(): - stripped = line.strip() - if stripped.startswith("```"): - in_code_block = not in_code_block - continue - if in_code_block: - if stripped and not stripped.startswith("#"): - commands.append(stripped) - continue - inline = re.findall(r"`([^`]+)`", line) - for cmd in inline: - cmd_str = cmd.strip() - if cmd_str: - commands.append(cmd_str) - if stripped.startswith("$"): - cmd = stripped.lstrip("$ ") - if cmd: - commands.append(cmd) - if any(keyword in stripped for keyword in COMMAND_KEYWORDS): - commands.append(stripped) - # Normalize and dedupe - normalized = [] - seen = set() - for cmd in commands: - cleaned = cmd.strip() - if not cleaned: - continue - if cleaned in seen: - continue - seen.add(cleaned) - normalized.append(cleaned) - return normalized - - -def extract_skill_refs(text: str) -> list[str]: - refs = set(SKILL_REF_PATTERN.findall(text)) - refs.update(SKILL_PATH_PATTERN.findall(text)) - return sorted(refs) - - -def resolve_skills_roots(repo_root: Path) -> list[Path]: - candidates = [] - codex_home = os.environ.get("CODEX_HOME") - if codex_home: - candidates.append(Path(codex_home) / "skills") - candidates.append(repo_root / ".codex" / "skills") - candidates.append(Path.home() / ".codex" / "skills") - - roots = [] - seen = set() - for candidate in candidates: - try: - resolved = candidate.expanduser().resolve() - except Exception: - resolved = candidate.expanduser() - key = str(resolved) - if key in seen: - continue - seen.add(key) - if resolved.exists(): - roots.append(resolved) - return roots - - -def find_repo_signals(repo_root: Path, max_results: int = 200) -> list[dict]: - results = [] - for dirpath, dirnames, filenames in os.walk(repo_root): - dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS] - for filename in filenames: - if filename in BUILD_SIGNAL_FILES: - path = Path(dirpath) / filename - results.append( - { - "path": str(path), - "type": "build_signal", - } - ) - if len(results) >= max_results: - return results - return results - - -def main() -> int: - parser = argparse.ArgumentParser( - description="Collect deterministic evidence for codex-readiness-unit-test." - ) - parser.add_argument( - "--out-dir", default=".codex-readiness-unit-test", help="Base output directory" - ) - parser.add_argument("--max-snippet-chars", type=int, default=2000, help="Max chars per snippet") - args = parser.parse_args() - - cwd = Path.cwd() - repo_root = cwd - agents_path = cwd / "AGENTS.md" - plans_path = cwd / "PLANS.md" - - snippets: list[dict[str, str]] = [] - candidate_commands: list[str] = [] - skill_refs: list[str] = [] - skills_info: SkillsInfo = cast( - SkillsInfo, - { - "roots": [], - "referenced": [], - "resolved": [], - "missing": [], - }, - ) - if agents_path.exists(): - text = read_text(agents_path) - if not text: - text = "" - snippets.append( - { - "path": str(agents_path), - "snippet": extract_snippet(text, args.max_snippet_chars), - } - ) - if text: - candidate_commands.extend(extract_candidate_commands(text)) - skill_refs = extract_skill_refs(text) - - if skill_refs: - skills_info["referenced"] = skill_refs - skills_roots = resolve_skills_roots(repo_root) - skills_info["roots"] = [str(root) for root in skills_roots] - for skill_name in skill_refs: - skill_path = None - for root in skills_roots: - candidate = root / skill_name / "SKILL.md" - if candidate.exists(): - skill_path = candidate - break - if skill_path is None: - skills_info["missing"].append({"name": skill_name}) - continue - skills_info["resolved"].append( - { - "name": skill_name, - "path": str(skill_path), - } - ) - skill_text = read_text(skill_path) - snippets.append( - { - "path": str(skill_path), - "snippet": extract_snippet(skill_text, args.max_snippet_chars), - } - ) - if skill_text: - candidate_commands.extend(extract_candidate_commands(skill_text)) - - # De-duplicate candidate commands - seen_cmds = set() - unique_cmds = [] - for cmd in candidate_commands: - if cmd in seen_cmds: - continue - seen_cmds.add(cmd) - unique_cmds.append(cmd) - - out_dir = Path(args.out_dir) - out_dir.mkdir(parents=True, exist_ok=True) - run_time = datetime.now(timezone.utc) - run_id = run_time.strftime("%Y-%m-%dT%H-%M-%SZ") - run_dir = (out_dir / run_id).resolve() - run_dir.mkdir(parents=True, exist_ok=True) - - evidence: dict[str, Any] = cast( - dict[str, Any], - { - "run_context": { - "cwd": str(cwd), - "repo_root": str(repo_root), - "run_dir": str(run_dir), - "run_id": run_id, - "timestamp": run_time.isoformat(), - }, - "agents_md": { - "exists": agents_path.exists(), - "path": str(agents_path) if agents_path.exists() else None, - }, - "plans_md": { - "exists": plans_path.exists(), - "path": str(plans_path) if plans_path.exists() else None, - }, - "skills": skills_info, - "snippets": snippets, - "inferred": { - "candidate_commands": unique_cmds[:50], - "repo_signals": find_repo_signals(repo_root), - }, - }, - ) - - evidence_path = run_dir / "evidence.json" - evidence_path.write_text(json.dumps(evidence, indent=2), encoding="utf-8") - - latest_path = out_dir / "latest.json" - latest_payload = { - "run_dir": str(run_dir), - "run_id": run_id, - "timestamp": evidence["run_context"]["timestamp"], - } - latest_path.write_text(json.dumps(latest_payload, indent=2), encoding="utf-8") - - print(str(evidence_path)) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/skills/.experimental/codex-readiness-unit-test/scripts/deterministic_rules.py b/skills/.experimental/codex-readiness-unit-test/scripts/deterministic_rules.py deleted file mode 100644 index 775f5d3..0000000 --- a/skills/.experimental/codex-readiness-unit-test/scripts/deterministic_rules.py +++ /dev/null @@ -1,396 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -import re -import sys -from pathlib import Path - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = load_json(latest_path) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "evidence.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def result( - status: str, - rationale: str, - evidence_path: str | None = None, - quote: str | None = None, - recommendations=None, - confidence: float = 1.0, -) -> dict: - if recommendations is None: - recommendations = [] - evidence_quotes = [] - if evidence_path and quote: - evidence_quotes.append({"path": evidence_path, "quote": quote}) - return { - "status": status, - "rationale": rationale, - "evidence_quotes": evidence_quotes, - "recommendations": recommendations, - "confidence": confidence, - } - - -def read_text(path: Path) -> str: - try: - return path.read_text(encoding="utf-8") - except Exception: - try: - return path.read_text(encoding="utf-8", errors="ignore") - except Exception: - return "" - - -def rule_agents_exists(evidence: dict) -> dict: - agents = evidence.get("agents_md", {}) - path = agents.get("path") - if agents.get("exists") and path and Path(path).exists(): - return result( - "PASS", - "AGENTS.md exists in the current directory.", - recommendations=[], - ) - return result( - "FAIL", - "AGENTS.md is missing in the current directory.", - recommendations=["Add an AGENTS.md in the current directory with onboarding context."], - confidence=1.0, - ) - - -PLAN_HEADINGS = [ - "## Purpose / Big Picture", - "## Progress", - "## Decision Log", - "## Outcomes & Retrospective", - "## Surprises & Discoveries", -] - - -def _clean_markdown_ref(raw_ref: str) -> str: - if not raw_ref: - return "" - cleaned = raw_ref.strip() - if cleaned.startswith("<") and cleaned.endswith(">"): - cleaned = cleaned[1:-1].strip() - cleaned = re.split(r"[?#]", cleaned, maxsplit=1)[0] - cleaned = cleaned.strip().strip("`'\"()[]{}<>.,:;") - return cleaned - - -def _is_markdown_path(ref: str) -> bool: - if not ref: - return False - try: - name = Path(ref).name.lower() - except Exception: - return False - return name.endswith((".md", ".markdown")) - - -def _extract_markdown_refs(text: str) -> list[tuple[str, str]]: - references: list[tuple[str, str]] = [] - for match in re.finditer(r"\[[^\]]*\]\(([^)]+)\)", text): - target = _clean_markdown_ref(match.group(1)) - if _is_markdown_path(target): - start = text.rfind("\n", 0, match.start()) - end = text.find("\n", match.start()) - line = text[start + 1 : end if end != -1 else None] - references.append((target, line.strip())) - for line in text.splitlines(): - for token in re.findall(r"(?i)[A-Za-z0-9_./\\-]*\.(?:md|markdown)", line): - cleaned = _clean_markdown_ref(token) - if _is_markdown_path(cleaned): - references.append((cleaned, line.strip())) - return references - - -def _is_plan_named(path: Path) -> bool: - return "plan" in path.name.lower() - - -def _has_planning_conventions(text: str) -> bool: - lowered = text.lower() - matches = sum(1 for heading in PLAN_HEADINGS if heading.lower() in lowered) - return matches >= 3 - - -def rule_plans_reference_exists(evidence: dict) -> dict: - agents = evidence.get("agents_md", {}) - agents_path = agents.get("path") - if not (agents.get("exists") and agents_path and Path(agents_path).exists()): - return result( - "FAIL", - "AGENTS.md is missing; cannot resolve referenced plans file.", - recommendations=["Add an AGENTS.md that references a plans markdown file."], - confidence=1.0, - ) - - text = read_text(Path(agents_path)) - references = _extract_markdown_refs(text) - if not references: - return result( - "FAIL", - "No planning markdown file reference found in AGENTS.md.", - recommendations=[ - "Reference a planning markdown file in AGENTS.md (e.g., PLANS.md or exec-plan.md)." - ], - confidence=1.0, - ) - - base_dir = Path(agents_path).parent - resolved = {} - for raw_ref, line in references: - cleaned = _clean_markdown_ref(raw_ref) - if not _is_markdown_path(cleaned): - continue - path = Path(cleaned) - resolved_path = path if path.is_absolute() else (base_dir / path).resolve() - if resolved_path not in resolved: - resolved[resolved_path] = line - - missing = [path for path in resolved if not path.exists()] - evidence_quotes = [] - for line in resolved.values(): - if line: - evidence_quotes.append({"path": agents_path, "quote": line[:240]}) - if len(evidence_quotes) >= 3: - break - - if missing: - missing_list = ", ".join(str(path) for path in missing) - return { - "status": "FAIL", - "rationale": f"Referenced planning markdown file(s) not found: {missing_list}.", - "evidence_quotes": evidence_quotes, - "recommendations": ["Create the referenced planning markdown file(s)."], - "confidence": 1.0, - } - - qualifying_paths = [] - non_qualifying_paths = [] - for path in resolved: - if _is_plan_named(path): - qualifying_paths.append(path) - continue - try: - content = read_text(path) - except Exception: - non_qualifying_paths.append(path) - continue - if _has_planning_conventions(content): - qualifying_paths.append(path) - else: - non_qualifying_paths.append(path) - - if qualifying_paths: - return { - "status": "PASS", - "rationale": "Referenced planning markdown file(s) exist and follow planning conventions.", - "evidence_quotes": evidence_quotes, - "recommendations": [], - "confidence": 1.0, - } - - missing_list = ", ".join(str(path) for path in non_qualifying_paths) - return { - "status": "FAIL", - "rationale": f"Referenced markdown file(s) do not appear to be planning docs: {missing_list}.", - "evidence_quotes": evidence_quotes, - "recommendations": [ - "Reference a planning markdown file (name contains 'plan') or add planning headings." - ], - "confidence": 1.0, - } - - -def rule_agents_line_count_under_300(evidence: dict) -> dict: - agents = evidence.get("agents_md", {}) - path = agents.get("path") - if not (agents.get("exists") and path and Path(path).exists()): - return result( - "FAIL", - "AGENTS.md is missing; cannot verify line count.", - recommendations=["Add an AGENTS.md in the current directory with onboarding context."], - confidence=1.0, - ) - - text = read_text(Path(path)) - line_count = len(text.splitlines()) - if line_count <= 300: - return result( - "PASS", - f"AGENTS.md has {line_count} lines (<= 300).", - recommendations=[], - confidence=1.0, - ) - return result( - "FAIL", - f"AGENTS.md has {line_count} lines (> 300).", - recommendations=["Trim AGENTS.md to 300 lines or fewer."], - confidence=1.0, - ) - - -def rule_config_toml_exists(evidence: dict) -> dict: - run_context = evidence.get("run_context", {}) - cwd = run_context.get("cwd") - if not cwd: - return result( - "FAIL", - "Run context missing; cannot resolve config.toml location.", - recommendations=["Ensure evidence.json includes run_context.cwd."], - confidence=1.0, - ) - - repo_root = Path(cwd) - repo_config = repo_root / "config.toml" - codex_config = repo_root / ".codex" / "config.toml" - user_codex_config = Path.home() / ".codex" / "config.toml" - - found_paths = [] - if repo_config.exists(): - found_paths.append(str(repo_config)) - if codex_config.exists(): - found_paths.append(str(codex_config)) - if user_codex_config.exists(): - found_paths.append(str(user_codex_config)) - - if found_paths: - return result( - "PASS", - f"config.toml found at: {', '.join(found_paths)}.", - recommendations=[], - confidence=1.0, - ) - - return result( - "FAIL", - "config.toml not found in repo root, repo .codex/, or user .codex/.", - recommendations=[ - "Add config.toml at the repo root, under .codex/config.toml, or in ~/.codex/config.toml." - ], - confidence=1.0, - ) - - -def rule_execution_summary_status(params: dict) -> dict: - summary_path = Path(params.get("summary_path", "execution_summary.json")) - if not summary_path.exists(): - return result( - "NOT_RUN", - "Execution summary not found; execution was not run.", - recommendations=["Run execute mode to validate the documented dev/build/test loop."], - confidence=1.0, - ) - try: - summary = load_json(summary_path) - except Exception: - return result( - "WARN", - "Execution summary exists but could not be parsed.", - recommendations=["Review the execution summary JSON for corruption."], - confidence=1.0, - ) - - status = summary.get("overall_status", "WARN") - rationale = f"Execution summary reports overall status: {status}." - return result( - status, - rationale, - recommendations=[], - confidence=1.0, - ) - - -RULES = { - "agents_exists": rule_agents_exists, - "plans_reference_exists": rule_plans_reference_exists, - "agents_line_count_under_300": rule_agents_line_count_under_300, - "config_toml_exists": rule_config_toml_exists, - "execution_summary_status": rule_execution_summary_status, -} - - -def main() -> int: - parser = argparse.ArgumentParser( - description="Run deterministic codex-readiness-unit-test rules." - ) - parser.add_argument( - "--out-dir", default=".codex-readiness-unit-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument("--evidence", default=None, help="Path to evidence.json (optional)") - parser.add_argument( - "--checks", - default=str( - Path(__file__).resolve().parents[1] - / "references" - / "checks" - / "checks.json" - ), - help="Path to checks.json", - ) - parser.add_argument("--out", default=None, help="Output path (optional)") - args = parser.parse_args() - - base_dir = Path(args.out_dir) - run_dir = resolve_run_dir(base_dir, args.run_dir) - evidence_path = Path(args.evidence) if args.evidence else (run_dir / "evidence.json") - checks_path = Path(args.checks) - if not evidence_path.exists(): - raise SystemExit(f"Evidence file not found: {evidence_path}") - if not checks_path.exists(): - raise SystemExit(f"Checks file not found: {checks_path}") - - evidence = load_json(evidence_path) - checks = load_json(checks_path) - - results = {} - for check in checks.get("checks", []): - if not check.get("enabled_by_default", False): - continue - rule_id = check.get("deterministic_rule_id") - if not rule_id: - continue - rule = RULES.get(rule_id) - if not rule: - continue - params = check.get("deterministic_rule_params", {}) - if rule_id == "execution_summary_status": - summary_path = params.get("summary_path", "execution_summary.json") - summary_candidate = Path(summary_path) - if not summary_candidate.is_absolute(): - summary_candidate = run_dir / summary_candidate - params = {**params, "summary_path": str(summary_candidate)} - results[check["id"]] = rule(params) - else: - results[check["id"]] = rule(evidence) - - out_path = Path(args.out) if args.out else (run_dir / "deterministic_results.json") - out_path.parent.mkdir(parents=True, exist_ok=True) - out_path.write_text(json.dumps({"results": results}, indent=2), encoding="utf-8") - print(str(out_path)) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/skills/.experimental/codex-readiness-unit-test/scripts/run_plan.py b/skills/.experimental/codex-readiness-unit-test/scripts/run_plan.py deleted file mode 100644 index 577f264..0000000 --- a/skills/.experimental/codex-readiness-unit-test/scripts/run_plan.py +++ /dev/null @@ -1,344 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import io -import json -import os -import re -import selectors -import subprocess -import sys -import time -from datetime import datetime, timezone -from pathlib import Path -from typing import Any, TypedDict - -DENYLIST_PATTERNS = [ - r"\brm\s+-rf\b", - r"\brm\s+-fr\b", - r"\brm\s+-r\b", - r"\bgit\s+clean\s+-xfd\b", - r"\bmkfs\b", - r"\bdd\s+if=", - r"\bdiskutil\s+erase\b", - r"\b:;\s*\b", # basic fork bomb patterns - r"\bmkfs\.[a-z0-9]+\b", -] -TEST_KEYWORDS = [ - " test", - "pytest", - "node --test", - "go test", - "cargo test", - "mvn test", - "gradle test", - "./gradlew test", -] -BUILD_KEYWORDS = [ - " build", - "compile", - "mvn package", - "gradle build", - "./gradlew build", - "go build", - "cargo build", -] - - -class PlanCommand(TypedDict, total=False): - label: str - cmd: str - timeout_soft_seconds: int - timeout_hard_seconds: int - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = load_json(latest_path) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "evidence.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def now_iso() -> str: - return datetime.now(timezone.utc).isoformat() - - -def is_denylisted(cmd: str) -> bool: - lower = cmd.lower() - return any(re.search(pattern, lower) for pattern in DENYLIST_PATTERNS) - - -def normalize_plan(plan_data: dict[str, Any]) -> dict[str, Any]: - if "commands" not in plan_data: - plan_data["commands"] = [] - commands: list[PlanCommand] = [] - for entry in plan_data.get("commands", []): - if isinstance(entry, str): - command_str: PlanCommand = {"label": "step", "cmd": entry} - commands.append(command_str) - elif isinstance(entry, dict): - cmd = entry.get("cmd") - if not cmd: - raise ValueError("Each command entry must include 'cmd'.") - command: PlanCommand = { - "label": entry.get("label") or "step", - "cmd": cmd, - } - soft = entry.get("timeout_soft_seconds") - hard = entry.get("timeout_hard_seconds") - if soft is not None: - command["timeout_soft_seconds"] = int(soft) - if hard is not None: - command["timeout_hard_seconds"] = int(hard) - commands.append(command) - else: - raise ValueError("Commands must be strings or objects with 'cmd'.") - plan_data["commands"] = commands - return plan_data - - -def classify_command(cmd: str) -> str | None: - lower = cmd.lower() - if any(keyword in lower for keyword in TEST_KEYWORDS): - return "test" - if any(keyword in lower for keyword in BUILD_KEYWORDS): - return "build" - return None - - -def infer_plan_commands(run_dir: Path) -> list[PlanCommand]: - evidence_path = run_dir / "evidence.json" - if not evidence_path.exists(): - return [] - evidence = load_json(evidence_path) - candidates = ( - evidence.get("inferred", {}).get("candidate_commands") if isinstance(evidence, dict) else [] - ) - if not isinstance(candidates, list): - return [] - - build_cmd = None - test_cmd = None - for cmd in candidates: - if not isinstance(cmd, str): - continue - label = classify_command(cmd) - if label == "build" and build_cmd is None: - build_cmd = cmd - elif label == "test" and test_cmd is None: - test_cmd = cmd - if build_cmd and test_cmd: - break - - commands: list[PlanCommand] = [] - if build_cmd: - commands.append({"label": "build", "cmd": build_cmd}) - if test_cmd: - commands.append({"label": "test", "cmd": test_cmd}) - return commands - - -def run_command( - cmd: str, cwd: Path, env: dict, soft_timeout: int, hard_timeout: int, log_path: Path -) -> dict: - started_at = now_iso() - start_time = time.time() - soft_exceeded = False - hard_exceeded = False - exit_code = None - - with log_path.open("w", encoding="utf-8") as log_file: - proc = subprocess.Popen( - cmd, - shell=True, - cwd=str(cwd), - env=env, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - bufsize=1, - ) - selector = selectors.DefaultSelector() - if proc.stdout: - selector.register(proc.stdout, selectors.EVENT_READ) - - while True: - now = time.time() - if not soft_exceeded and now - start_time > soft_timeout: - soft_exceeded = True - if now - start_time > hard_timeout: - hard_exceeded = True - proc.terminate() - try: - proc.wait(timeout=5) - except subprocess.TimeoutExpired: - proc.kill() - break - events = selector.select(timeout=0.2) - for key, _ in events: - file_obj = key.fileobj - if isinstance(file_obj, io.TextIOBase): - line = file_obj.readline() - if line: - log_file.write(line) - if proc.poll() is not None: - break - - # Drain remaining output - if proc.stdout: - for line in proc.stdout: - log_file.write(line) - - exit_code = proc.returncode - - ended_at = now_iso() - duration = time.time() - start_time - - if hard_exceeded: - status = "FAIL" - elif exit_code == 0: - status = "WARN" if soft_exceeded else "PASS" - else: - status = "FAIL" - - return { - "cmd": cmd, - "status": status, - "exit_code": exit_code, - "duration_seconds": round(duration, 2), - "soft_timeout_seconds": soft_timeout, - "hard_timeout_seconds": hard_timeout, - "soft_timeout_exceeded": soft_exceeded, - "hard_timeout_exceeded": hard_exceeded, - "log_path": str(log_path), - "started_at": started_at, - "ended_at": ended_at, - } - - -def sanitize_label(label: str) -> str: - cleaned = re.sub(r"[^a-zA-Z0-9_.-]+", "-", label.strip().lower()) - return cleaned.strip("-") or "step" - - -def main() -> int: - parser = argparse.ArgumentParser(description="Execute a documented dev/build/test plan.") - parser.add_argument("--plan", required=True, help="Path to plan JSON") - parser.add_argument( - "--out-dir", default=".codex-readiness-unit-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument( - "--soft-timeout-seconds", type=int, default=600, help="Soft timeout per command" - ) - parser.add_argument( - "--hard-timeout-multiplier", type=int, default=3, help="Hard timeout multiplier" - ) - args = parser.parse_args() - - plan_path = Path(args.plan) - if not plan_path.exists(): - raise SystemExit(f"Plan file not found: {plan_path}") - - plan_data = normalize_plan(load_json(plan_path)) - - cwd = Path(plan_data.get("cwd") or plan_data.get("project_dir") or Path.cwd()) - if not cwd.is_absolute(): - cwd = (Path.cwd() / cwd).resolve() - - env = os.environ.copy() - env.update(plan_data.get("env", {})) - - base_dir = Path(args.out_dir) - base_dir.mkdir(parents=True, exist_ok=True) - run_dir = resolve_run_dir(base_dir, args.run_dir) - run_dir.mkdir(parents=True, exist_ok=True) - logs_dir = run_dir / "logs" - logs_dir.mkdir(parents=True, exist_ok=True) - - if not plan_data.get("commands"): - inferred = infer_plan_commands(run_dir) - if inferred: - plan_data["commands"] = inferred - else: - raise ValueError( - "Plan JSON has no commands and no build/test commands could be inferred from skills." - ) - - steps = [] - for index, entry in enumerate(plan_data.get("commands", []), start=1): - label = entry.get("label") or f"step-{index}" - cmd = entry.get("cmd", "").strip() - soft_timeout = entry.get("timeout_soft_seconds") or args.soft_timeout_seconds - hard_timeout = entry.get("timeout_hard_seconds") or ( - soft_timeout * args.hard_timeout_multiplier - ) - - log_path = logs_dir / f"{index:02d}-{sanitize_label(label)}.log" - if is_denylisted(cmd): - steps.append( - { - "label": label, - "cmd": cmd, - "status": "FAIL", - "exit_code": None, - "duration_seconds": 0, - "soft_timeout_seconds": soft_timeout, - "hard_timeout_seconds": hard_timeout, - "soft_timeout_exceeded": False, - "hard_timeout_exceeded": False, - "denylisted": True, - "log_path": str(log_path), - "started_at": now_iso(), - "ended_at": now_iso(), - } - ) - continue - - result = run_command(cmd, cwd, env, soft_timeout, hard_timeout, log_path) - result["label"] = label - result["denylisted"] = False - steps.append(result) - - overall_status = "PASS" - for step in steps: - if step["status"] == "FAIL": - overall_status = "FAIL" - break - if step["status"] == "WARN": - overall_status = "WARN" - - summary = { - "plan_path": str(plan_path), - "project_dir": str(plan_data.get("project_dir", cwd)), - "cwd": str(cwd), - "soft_timeout_seconds": args.soft_timeout_seconds, - "hard_timeout_multiplier": args.hard_timeout_multiplier, - "steps": steps, - "overall_status": overall_status, - "started_at": steps[0]["started_at"] if steps else now_iso(), - "ended_at": steps[-1]["ended_at"] if steps else now_iso(), - } - - summary_path = run_dir / "execution_summary.json" - summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8") - print(str(summary_path)) - - return 0 if overall_status == "PASS" else 1 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/skills/.experimental/codex-readiness-unit-test/scripts/scoring.py b/skills/.experimental/codex-readiness-unit-test/scripts/scoring.py deleted file mode 100644 index 671a4f3..0000000 --- a/skills/.experimental/codex-readiness-unit-test/scripts/scoring.py +++ /dev/null @@ -1,452 +0,0 @@ -#!/usr/bin/env python3 -import argparse -import json -from decimal import ROUND_HALF_UP, Decimal -from pathlib import Path -from typing import Any - -VALID_STATUSES = {"PASS", "WARN", "FAIL", "NOT_RUN"} -PRIORITY_MULTIPLIERS = {0: 4, 1: 3, 2: 2, 3: 1} - - -def load_json(path: Path) -> dict: - return json.loads(path.read_text(encoding="utf-8")) - - -def resolve_run_dir(base_dir: Path, run_dir_arg: str | None) -> Path: - if run_dir_arg: - return Path(run_dir_arg).resolve() - latest_path = base_dir / "latest.json" - if latest_path.exists(): - try: - latest = load_json(latest_path) - run_dir = latest.get("run_dir") - if run_dir: - return Path(run_dir) - except Exception: - pass - if (base_dir / "evidence.json").exists(): - return base_dir.resolve() - return base_dir.resolve() - - -def round_half_up(value: float) -> int: - return int(Decimal(value).quantize(Decimal("1"), rounding=ROUND_HALF_UP)) - - -def status_points(status: str) -> float: - if status == "PASS": - return 1.0 - if status == "WARN": - return 0.5 - return 0.0 - - -def normalize_priority(value) -> int: - try: - priority = int(value) - except (TypeError, ValueError): - priority = 3 - if priority in PRIORITY_MULTIPLIERS: - return priority - return 3 - - -def priority_label(priority: int) -> str: - return f"P{priority}" - - -def validate_result(result: dict) -> dict | None: - if not isinstance(result, dict): - return None - if result.get("status") not in VALID_STATUSES: - return None - if ( - "rationale" not in result - or "evidence_quotes" not in result - or "recommendations" not in result - or "confidence" not in result - ): - return None - if not isinstance(result.get("evidence_quotes"), list): - return None - if not isinstance(result.get("recommendations"), list): - return None - return result - - -def fallback_invalid_json() -> dict: - return { - "status": "WARN", - "rationale": "Invalid JSON from evaluator after retries.", - "evidence_quotes": [], - "recommendations": ["Re-run the evaluator with the json_fix prompt."], - "confidence": 0.0, - } - - -def build_weights(checks: list[dict]) -> dict: - enabled = [c for c in checks if c.get("enabled_by_default")] - raw_weights = [] - for check in enabled: - weight = check.get("weight") - base_weight = weight if isinstance(weight, (int, float)) else 1.0 - priority = normalize_priority(check.get("priority")) - raw_weights.append(base_weight * PRIORITY_MULTIPLIERS[priority]) - total = sum(raw_weights) if raw_weights else 1.0 - weights = {} - for check, raw in zip(enabled, raw_weights): - weights[check["id"]] = (raw / total) * 100.0 - return weights - - -def build_results( - checks: list[dict], - deterministic_results: dict, - llm_results: dict, - execution_summary: dict | None, - mode: str, -) -> dict: - results = {} - execution_status = None - if execution_summary: - execution_status = execution_summary.get("overall_status") - - for check in checks: - if not check.get("enabled_by_default"): - continue - check_id = check["id"] - check_type = check.get("type") - - if check_type == "DETERMINISTIC": - result = deterministic_results.get(check_id) - if result: - valid = validate_result(result) - results[check_id] = valid if valid else fallback_invalid_json() - else: - results[check_id] = { - "status": "FAIL", - "rationale": "Deterministic result missing for this check.", - "evidence_quotes": [], - "recommendations": ["Run deterministic_rules.py to populate results."], - "confidence": 0.0, - } - continue - - if check_type == "LLM": - llm_result = llm_results.get(check_id) - if llm_result: - valid = validate_result(llm_result) - results[check_id] = valid if valid else fallback_invalid_json() - else: - results[check_id] = { - "status": "WARN", - "rationale": "LLM evaluation missing for this check.", - "evidence_quotes": [], - "recommendations": ["Run the evaluator prompt for this check."], - "confidence": 0.0, - } - continue - - if check_type == "HYBRID": - if mode == "read-only": - status = "NOT_RUN" - else: - status = ( - execution_status - or deterministic_results.get(check_id, {}).get("status") - or "NOT_RUN" - ) - llm_result = llm_results.get(check_id) - if llm_result: - valid = validate_result(llm_result) or fallback_invalid_json() - valid["status"] = status - results[check_id] = valid - else: - results[check_id] = { - "status": status, - "rationale": "Execution summary present but LLM rationale missing." - if status != "NOT_RUN" - else "Execution not run.", - "evidence_quotes": [], - "recommendations": [ - "Provide execution rationale using the execution_explanation prompt." - ], - "confidence": 0.0, - } - continue - - return results - - -def render_html(report: dict) -> str: - score = report["scorecard"]["score_total"] - status = report["scorecard"]["overall_status"] - results = report.get("results", {}) - enabled_checks = report.get("enabled_checks", []) - checks_by_id = {check["id"]: check for check in enabled_checks} - run_context = report.get("run_context", {}) - file_path = run_context.get("repo_root") or run_context.get("cwd") or "" - - def status_class(value: str) -> str: - return value.lower() - - html = [ - "", - "", - "", - "", - "Codex Readiness Unit Test Report", - "", - "", - "", - "

Codex Readiness Unit Test Report

", - f"

{file_path}

" if file_path else "", - f"

Overall score: {score} {status}

", - "

Checks

", - ] - colgroup = ( - "" - ) - grouped: dict[str, list[str]] = { - f"P{priority}": [] for priority in sorted(PRIORITY_MULTIPLIERS.keys()) - } - for check in enabled_checks: - check_id = check["id"] - label = check.get("priority_label") or priority_label( - normalize_priority(check.get("priority")) - ) - if label in grouped: - grouped[label].append(check_id) - - for label in ["P0", "P1", "P2", "P3"]: - check_ids = grouped.get(label, []) - if not check_ids: - continue - html.append(f"

{label}

") - html.append("") - html.append(colgroup) - html.append("") - for check_id in check_ids: - result = results.get(check_id, {}) - title = checks_by_id.get(check_id, {}).get("title") or check_id - rationale = result.get("rationale", "") - status_value = result.get("status", "WARN") - html.append( - f"" - ) - html.append("
CheckStatusRationale
{title}{status_value}{rationale}
") - html.append("") - return "\n".join(html) - - -def build_summary( - checks: list[dict], results: dict, overall_counts: dict, overall_status: str -) -> dict: - enabled_checks = [c for c in checks if c.get("enabled_by_default")] - by_status: dict[str, list[dict[str, Any]]] = {"PASS": [], "FAIL": [], "WARN": [], "NOT_RUN": []} - all_checks = [] - for check in enabled_checks: - check_id = check["id"] - result = results.get(check_id, {}) - status = result.get("status", "WARN") - priority = normalize_priority(check.get("priority")) - entry = { - "id": check_id, - "title": check.get("title") or check_id, - "status": status, - "priority": priority, - "priority_label": priority_label(priority), - } - all_checks.append(entry) - if status == "PASS": - by_status["PASS"].append(entry) - continue - detail = { - **entry, - "rationale": result.get("rationale", ""), - "recommendations": result.get("recommendations", []), - } - by_status.get(status, by_status["WARN"]).append(detail) - - return { - "overall_status": overall_status, - "counts": overall_counts, - "checks": all_checks, - "passed": by_status["PASS"], - "failed": by_status["FAIL"], - "warned": by_status["WARN"], - "not_run": by_status["NOT_RUN"], - } - - -def render_summary_text(summary: dict) -> str: - counts = summary.get("counts", {}) - lines = [ - "Summary", - f"Overall status: {summary.get('overall_status')}", - f"Counts: PASS={counts.get('PASS', 0)} WARN={counts.get('WARN', 0)} FAIL={counts.get('FAIL', 0)} NOT_RUN={counts.get('NOT_RUN', 0)}", - ] - - def render_section(label: str, items: list[dict], include_details: bool = False) -> None: - if not items: - return - lines.append(f"{label}:") - for item in items: - title = item.get("title") or item.get("id") - lines.append(f"- {item.get('priority_label')} {title} ({item.get('id')})") - if include_details: - rationale = item.get("rationale", "") - if rationale: - lines.append(f" Rationale: {rationale}") - recommendations = item.get("recommendations", []) - if recommendations: - lines.append(f" Recommendations: {', '.join(recommendations)}") - - render_section("Failed", summary.get("failed", []), include_details=True) - render_section("Warned", summary.get("warned", []), include_details=True) - render_section("Not run", summary.get("not_run", []), include_details=True) - render_section("Passed", summary.get("passed", []), include_details=False) - - return "\n".join(lines) - - -def main() -> int: - parser = argparse.ArgumentParser(description="Compute scorecard and render report outputs.") - parser.add_argument("--mode", choices=["read-only", "execute"], required=True, help="Run mode") - parser.add_argument( - "--out-dir", default=".codex-readiness-unit-test", help="Base output directory" - ) - parser.add_argument("--run-dir", default=None, help="Specific run directory to use") - parser.add_argument( - "--checks", - default=str( - Path(__file__).resolve().parents[1] - / "references" - / "checks" - / "checks.json" - ), - help="Path to checks.json", - ) - parser.add_argument("--evidence", default=None, help="Path to evidence.json (optional)") - parser.add_argument( - "--deterministic", default=None, help="Path to deterministic results (optional)" - ) - parser.add_argument("--llm", default=None, help="Path to LLM results (optional)") - args = parser.parse_args() - - base_dir = Path(args.out_dir) - base_dir.mkdir(parents=True, exist_ok=True) - run_dir = resolve_run_dir(base_dir, args.run_dir) - run_dir.mkdir(parents=True, exist_ok=True) - - checks = load_json(Path(args.checks)).get("checks", []) - evidence_path = Path(args.evidence) if args.evidence else (run_dir / "evidence.json") - deterministic_path = ( - Path(args.deterministic) if args.deterministic else (run_dir / "deterministic_results.json") - ) - llm_path = Path(args.llm) if args.llm else (run_dir / "llm_results.json") - evidence = load_json(evidence_path) - deterministic_results = ( - load_json(deterministic_path).get("results", {}) if deterministic_path.exists() else {} - ) - llm_results = load_json(llm_path) if llm_path.exists() else {} - - execution_summary_path = run_dir / "execution_summary.json" - execution_summary = ( - load_json(execution_summary_path) if execution_summary_path.exists() else None - ) - - weights = build_weights(checks) - results = build_results( - checks, deterministic_results, llm_results, execution_summary, args.mode - ) - - overall_counts = {"PASS": 0, "WARN": 0, "FAIL": 0, "NOT_RUN": 0} - per_check_contributions = {} - total = 0.0 - for check in checks: - if not check.get("enabled_by_default"): - continue - check_id = check["id"] - result = results.get(check_id, {}) - status = result.get("status", "WARN") - overall_counts[status] = overall_counts.get(status, 0) + 1 - weight = weights.get(check_id, 0.0) - contribution = weight * status_points(status) - total += contribution - per_check_contributions[check_id] = { - "status": status, - "weight": round(weight, 2), - "contribution": round(contribution, 2), - "priority": normalize_priority(check.get("priority")), - "priority_label": priority_label(normalize_priority(check.get("priority"))), - } - - overall_status = "PASS" - if overall_counts.get("FAIL"): - overall_status = "FAIL" - elif overall_counts.get("WARN") or overall_counts.get("NOT_RUN"): - overall_status = "WARN" - - enabled_checks = [] - for check in checks: - if not check.get("enabled_by_default"): - continue - priority = normalize_priority(check.get("priority")) - enabled_checks.append( - { - **check, - "priority": priority, - "priority_label": priority_label(priority), - "priority_multiplier": PRIORITY_MULTIPLIERS[priority], - "weight": round(weights.get(check["id"], 0.0), 2), - } - ) - - report = { - "schema_version": "1.0", - "tool_name": "codex-readiness-unit-test", - "tool_version": "0.1.0", - "run_context": evidence.get("run_context", {}), - "enabled_checks": enabled_checks, - "results": results, - "execution_summary": execution_summary if args.mode == "execute" else None, - "scorecard": { - "score_total_raw": round(total, 2), - "score_total": round_half_up(total), - "per_check_contributions": per_check_contributions, - "counts": overall_counts, - "overall_status": overall_status, - }, - } - - report_path = run_dir / "report.json" - report_path.write_text(json.dumps(report, indent=2), encoding="utf-8") - - html = render_html(report) - html_path = run_dir / "report.html" - html_path.write_text(html, encoding="utf-8") - - summary = build_summary(checks, results, overall_counts, overall_status) - summary_path = run_dir / "summary.json" - summary_path.write_text(json.dumps(summary, indent=2), encoding="utf-8") - - print(str(report_path)) - print(str(html_path)) - print(str(summary_path)) - print(render_summary_text(summary)) - return 0 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/skills/.experimental/create-plan/LICENSE.txt b/skills/.experimental/create-plan/LICENSE.txt deleted file mode 100644 index d645695..0000000 --- a/skills/.experimental/create-plan/LICENSE.txt +++ /dev/null @@ -1,202 +0,0 @@ - - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright [yyyy] [name of copyright owner] - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. diff --git a/skills/.experimental/create-plan/SKILL.md b/skills/.experimental/create-plan/SKILL.md deleted file mode 100644 index fdd0050..0000000 --- a/skills/.experimental/create-plan/SKILL.md +++ /dev/null @@ -1,74 +0,0 @@ ---- -name: create-plan -description: Create a concise plan. Use when a user explicitly asks for a plan related to a coding task. -metadata: - short-description: Create a plan ---- - -# Create Plan - -## Goal - -Turn a user prompt into a **single, actionable plan** delivered in the final assistant message. - -## Minimal workflow - -Throughout the entire workflow, operate in read-only mode. Do not write or update files. - -1. **Scan context quickly** - - Read `README.md` and any obvious docs (`docs/`, `CONTRIBUTING.md`, `ARCHITECTURE.md`). - - Skim relevant files (the ones most likely touched). - - Identify constraints (language, frameworks, CI/test commands, deployment shape). - -2. **Ask follow-ups only if blocking** - - Ask **at most 1–2 questions**. - - Only ask if you cannot responsibly plan without the answer; prefer multiple-choice. - - If unsure but not blocked, make a reasonable assumption and proceed. - -3. **Create a plan using the template below** - - Start with **1 short paragraph** describing the intent and approach. - - Clearly call out what is **in scope** and what is **not in scope** in short. - - Then provide a **small checklist** of action items (default 6–10 items). - - Each checklist item should be a concrete action and, when helpful, mention files/commands. - - **Make items atomic and ordered**: discovery → changes → tests → rollout. - - **Verb-first**: “Add…”, “Refactor…”, “Verify…”, “Ship…”. - - Include at least one item for **tests/validation** and one for **edge cases/risk** when applicable. - - If there are unknowns, include a tiny **Open questions** section (max 3). - -4. **Do not preface the plan with meta explanations; output only the plan as per template** - -## Plan template (follow exactly) - -```markdown -# Plan - -<1–3 sentences: what we’re doing, why, and the high-level approach.> - -## Scope -- In: -- Out: - -## Action items -[ ] -[ ] -[ ] -[ ] -[ ] -[ ] - -## Open questions -- -- -- -``` - -## Checklist item guidance -Good checklist items: -- Point to likely files/modules: src/..., app/..., services/... -- Name concrete validation: “Run npm test”, “Add unit tests for X” -- Include safe rollout when relevant: feature flag, migration plan, rollback note - -Avoid: -- Vague steps (“handle backend”, “do auth”) -- Too many micro-steps -- Writing code snippets (keep the plan implementation-agnostic) diff --git a/skills/.experimental/gitlab-address-comments/LICENSE.txt b/skills/.experimental/gitlab-address-comments/LICENSE.txt deleted file mode 100644 index 7a4a3ea..0000000 --- a/skills/.experimental/gitlab-address-comments/LICENSE.txt +++ /dev/null @@ -1,202 +0,0 @@ - - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright [yyyy] [name of copyright owner] - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. \ No newline at end of file diff --git a/skills/.experimental/gitlab-address-comments/SKILL.md b/skills/.experimental/gitlab-address-comments/SKILL.md deleted file mode 100644 index be9c445..0000000 --- a/skills/.experimental/gitlab-address-comments/SKILL.md +++ /dev/null @@ -1,59 +0,0 @@ ---- -name: gitlab-address-comments -description: Help address review/issue comments on the open GitLab MR for the current branch using glab CLI. Use when the user wants help addressing review/issue comments on an open GitLab MR -metadata: - short-description: Address comments in a GitLab MR review ---- - -# MR Comment Handler - -Find the open MR for the current branch and address its review threads using `glab`. Run all `glab` commands with elevated network access. - -## Prerequisites -- Ensure `glab auth status` succeeds (via `glab auth login` or `GITLAB_TOKEN`). -- Ensure `glab` is at least v1.80.4. -- When sandboxing blocks network calls, rerun with `sandbox_permissions=require_escalated`. -- Sanity check auth up front: -```bash -glab auth status -``` - -## 1) Resolve the MR for the current branch -Do a quick check so we know which MR we are about to operate on: -```bash -branch="$(git rev-parse --abbrev-ref HEAD)" -glab mr view "$branch" --output json -``` -If this fails, the fetch script below will still try to locate the MR by `source_branch`. - -## 2) Fetch unresolved discussions to `/tmp` -Use the local script to fetch MR discussions via `glab api`. This filters out bot/system-only threads and returns unresolved discussions when `--open-comments` is set. -```bash -skill_dir="" -branch="$(git rev-parse --abbrev-ref HEAD)" -safe_branch="${branch//\//_}" -out="/tmp/${safe_branch}_mr_open_discussions.json" -python "$skill_dir/scripts/fetch_comments.py" --open-comments --output "$out" -``` -If you want the full payload (including resolved discussions), drop `--open-comments`. - -## 3) Summarize, triage, and ask once -- Load the JSON and number each unresolved discussion. -- Start with a compact summary list instead of dumping full threads. -- Sort by: unresolved first (already filtered), then most recently updated. -- When possible, group or label by file path to reduce context switching. -- In the summary list, show: number, discussion id, author, file:line (if present), and a one-line summary. -- Ask for a batch selection in one shot. Accept: `1,3,5-7`, `all`, `none`, or `top N`. -- If the user does not choose, suggest a small default set (for example `top 3`) with a short rationale. -- Only after selection, show the full thread and code context for the selected numbers. -- When showing code context, include 3 lines before and after and clearly mark the referenced line(s). - -## 4) Implement fixes for the selected discussions -- Apply focused fixes that address the selected threads. -- Run the most relevant tests or checks you can in-repo. -- Report back with: what changed, which discussion numbers were addressed, and any follow-ups. - -Notes: -- If `glab` hits auth or rate issues, prompt the user to run `glab auth login` or re-export `GITLAB_TOKEN`, then retry. -- If no open MR is found for the branch, say so clearly and ask for the MR URL or IID. -- Do not prompt “address or skip?” one comment at a time unless the user explicitly asks for that mode. diff --git a/skills/.experimental/gitlab-address-comments/scripts/fetch_comments.py b/skills/.experimental/gitlab-address-comments/scripts/fetch_comments.py deleted file mode 100755 index 4594ce3..0000000 --- a/skills/.experimental/gitlab-address-comments/scripts/fetch_comments.py +++ /dev/null @@ -1,274 +0,0 @@ -#!/usr/bin/env python3 -""" -Fetch GitLab merge request discussions (including inline threads) for the MR -associated with the current git branch, by shelling out to: - - glab api - -Requires: - - `glab auth status` succeeds (uses glab config/keyring or GITLAB_TOKEN) - - current branch has an associated open MR - -Usage: - python scripts/fetch_comments.py > /tmp/mr_comments.json - python scripts/fetch_comments.py --open-comments > /tmp/open_threads.json - python scripts/fetch_comments.py --output /tmp/mr_comments.json -""" - -from __future__ import annotations - -import argparse -import json -import re -from pathlib import Path -import subprocess -import sys -from typing import Any, Iterable -from urllib.parse import quote - - -def _run(cmd: list[str]) -> str: - p = subprocess.run(cmd, capture_output=True, text=True) - if p.returncode != 0: - raise RuntimeError(f"Command failed: {' '.join(cmd)}\n{p.stderr}") - return p.stdout - - -def _run_json(cmd: list[str]) -> Any: - out = _run(cmd) - try: - return json.loads(out) - except json.JSONDecodeError as e: - raise RuntimeError(f"Failed to parse JSON from command output: {e}\nRaw:\n{out}") from e - - -def _ensure_glab_authenticated() -> None: - try: - _run(["glab", "auth", "status"]) - except RuntimeError as exc: - raise RuntimeError( - "glab auth status failed; run `glab auth login` or set GITLAB_TOKEN" - ) from exc - - -def _git_current_branch() -> str: - return _run(["git", "rev-parse", "--abbrev-ref", "HEAD"]).strip() - - -def _git_origin_url() -> str: - return _run(["git", "remote", "get-url", "origin"]).strip() - - -def _strip_dot_git(path: str) -> str: - return path[:-4] if path.endswith(".git") else path - - -def _parse_project_path(remote_url: str) -> str: - """ - Convert a git remote URL into a GitLab project path (group/subgroup/project). - Supports common SSH and HTTPS formats. - """ - # HTTPS: https://gitlab.example.com/group/project.git - https_match = re.match(r"^https?://[^/]+/(.+)$", remote_url) - if https_match: - return _strip_dot_git(https_match.group(1)) - - # SSH scp-like: git@gitlab.example.com:group/project.git - ssh_match = re.match(r"^(?:ssh://)?git@[^:/]+[:/](.+)$", remote_url) - if ssh_match: - return _strip_dot_git(ssh_match.group(1)) - - raise RuntimeError(f"Unable to parse GitLab project path from origin URL: {remote_url}") - - -def _glab_api_get(endpoint: str, params: dict[str, Any] | None = None) -> Any: - cmd = ["glab", "api", endpoint, "-X", "GET"] - for key, value in (params or {}).items(): - if value is None: - continue - cmd += ["-F", f"{key}={value}"] - return _run_json(cmd) - - -def _paginate(endpoint: str, base_params: dict[str, Any], per_page: int = 100, max_pages: int = 20) -> list[Any]: - results: list[Any] = [] - page = 1 - - while page <= max_pages: - params = dict(base_params) - params.update({"per_page": per_page, "page": page}) - chunk = _glab_api_get(endpoint, params=params) - - if not isinstance(chunk, list) or not chunk: - break - - results.extend(chunk) - - if len(chunk) < per_page: - break - page += 1 - - return results - - -def _encode_project_path(project_path: str) -> str: - # GitLab API accepts URL-encoded project paths in place of numeric IDs. - return quote(project_path, safe="") - - -def _find_open_mr_for_branch(project: str, branch: str) -> dict[str, Any]: - endpoint = f"/projects/{project}/merge_requests" - mrs = _paginate( - endpoint, - base_params={ - "state": "opened", - "source_branch": branch, - "order_by": "updated_at", - "sort": "desc", - }, - ) - if not mrs: - raise RuntimeError(f"No open merge request found for source branch: {branch}") - # Prefer the most recently updated MR for this branch. - return mrs[0] - - -def _get_mr(project: str, mr_iid: int) -> dict[str, Any]: - endpoint = f"/projects/{project}/merge_requests/{mr_iid}" - mr = _glab_api_get(endpoint) - if not isinstance(mr, dict): - raise RuntimeError("Unexpected response when fetching merge request details") - return mr - - -def _discussion_notes(discussion: dict[str, Any]) -> list[dict[str, Any]]: - notes = discussion.get("notes") - return notes if isinstance(notes, list) else [] - - -def _is_bot_or_system_note(note: dict[str, Any]) -> bool: - if note.get("system"): - return True - author = note.get("author") or {} - if author.get("bot"): - return True - username = str(author.get("username") or "").lower() - return username.endswith("[bot]") or username.endswith("-bot") - - -def _filter_bot_notes(notes: Iterable[dict[str, Any]]) -> list[dict[str, Any]]: - return [n for n in notes if not _is_bot_or_system_note(n)] - - -def _discussion_is_open(discussion: dict[str, Any]) -> bool: - # Prefer the top-level resolved flag if present. - resolved = discussion.get("resolved") - if resolved is True: - return False - - notes = _discussion_notes(discussion) - resolvable_notes = [n for n in notes if n.get("resolvable")] - if resolvable_notes: - # If any resolvable note is still unresolved, treat the discussion as open. - return any(not bool(n.get("resolved")) for n in resolvable_notes) - - # Fallback: treat unresolved/unknown as open. - return not bool(resolved) - - -def _discussion_has_non_bot_content(discussion: dict[str, Any]) -> bool: - notes = _discussion_notes(discussion) - return len(_filter_bot_notes(notes)) > 0 - - -def fetch_all(project_path: str, branch: str) -> dict[str, Any]: - encoded_project = _encode_project_path(project_path) - - mr_hint = _find_open_mr_for_branch(encoded_project, branch) - mr_iid = int(mr_hint["iid"]) - mr = _get_mr(encoded_project, mr_iid) - - discussions_endpoint = f"/projects/{encoded_project}/merge_requests/{mr_iid}/discussions" - discussions = _paginate(discussions_endpoint, base_params={}) - - # Drop pure bot/system discussions. - discussions = [d for d in discussions if _discussion_has_non_bot_content(d)] - - open_discussions = [d for d in discussions if _discussion_is_open(d)] - - mr_meta = { - "iid": mr.get("iid"), - "project_id": mr.get("project_id"), - "web_url": mr.get("web_url"), - "title": mr.get("title"), - "state": mr.get("state"), - "source_branch": mr.get("source_branch"), - "target_branch": mr.get("target_branch"), - "updated_at": mr.get("updated_at"), - } - - return { - "merge_request": mr_meta, - "project": { - "path": project_path, - "encoded_path": encoded_project, - "branch": branch, - }, - "discussions": discussions, - "open_discussions": open_discussions, - } - - -def _build_arg_parser() -> argparse.ArgumentParser: - parser = argparse.ArgumentParser(description="Fetch GitLab MR discussions for the current branch") - parser.add_argument( - "--open-comments", - action="store_true", - help="emit only unresolved/open discussions (still includes merge_request metadata)", - ) - parser.add_argument( - "--output", - help="optional path to write JSON output; defaults to stdout", - ) - return parser - - -def main() -> None: - parser = _build_arg_parser() - args = parser.parse_args() - - _ensure_glab_authenticated() - - branch = _git_current_branch() - origin = _git_origin_url() - project_path = _parse_project_path(origin) - - result = fetch_all(project_path=project_path, branch=branch) - - if args.open_comments: - payload: Any = { - "merge_request": result["merge_request"], - "project": result["project"], - "open_discussions": result["open_discussions"], - } - else: - payload = result - - output = json.dumps(payload, indent=2) - - if args.output: - out_path = Path(args.output) - out_path.parent.mkdir(parents=True, exist_ok=True) - with open(out_path, "w", encoding="utf-8") as f: - f.write(output) - f.write("\n") - else: - print(output) - - -if __name__ == "__main__": - try: - main() - except RuntimeError as exc: - print(str(exc), file=sys.stderr) - sys.exit(2) diff --git a/skills/.experimental/wrapped/LICENSE.txt b/skills/.experimental/wrapped/LICENSE.txt deleted file mode 100644 index d645695..0000000 --- a/skills/.experimental/wrapped/LICENSE.txt +++ /dev/null @@ -1,202 +0,0 @@ - - Apache License - Version 2.0, January 2004 - http://www.apache.org/licenses/ - - TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION - - 1. Definitions. - - "License" shall mean the terms and conditions for use, reproduction, - and distribution as defined by Sections 1 through 9 of this document. - - "Licensor" shall mean the copyright owner or entity authorized by - the copyright owner that is granting the License. - - "Legal Entity" shall mean the union of the acting entity and all - other entities that control, are controlled by, or are under common - control with that entity. For the purposes of this definition, - "control" means (i) the power, direct or indirect, to cause the - direction or management of such entity, whether by contract or - otherwise, or (ii) ownership of fifty percent (50%) or more of the - outstanding shares, or (iii) beneficial ownership of such entity. - - "You" (or "Your") shall mean an individual or Legal Entity - exercising permissions granted by this License. - - "Source" form shall mean the preferred form for making modifications, - including but not limited to software source code, documentation - source, and configuration files. - - "Object" form shall mean any form resulting from mechanical - transformation or translation of a Source form, including but - not limited to compiled object code, generated documentation, - and conversions to other media types. - - "Work" shall mean the work of authorship, whether in Source or - Object form, made available under the License, as indicated by a - copyright notice that is included in or attached to the work - (an example is provided in the Appendix below). - - "Derivative Works" shall mean any work, whether in Source or Object - form, that is based on (or derived from) the Work and for which the - editorial revisions, annotations, elaborations, or other modifications - represent, as a whole, an original work of authorship. For the purposes - of this License, Derivative Works shall not include works that remain - separable from, or merely link (or bind by name) to the interfaces of, - the Work and Derivative Works thereof. - - "Contribution" shall mean any work of authorship, including - the original version of the Work and any modifications or additions - to that Work or Derivative Works thereof, that is intentionally - submitted to Licensor for inclusion in the Work by the copyright owner - or by an individual or Legal Entity authorized to submit on behalf of - the copyright owner. For the purposes of this definition, "submitted" - means any form of electronic, verbal, or written communication sent - to the Licensor or its representatives, including but not limited to - communication on electronic mailing lists, source code control systems, - and issue tracking systems that are managed by, or on behalf of, the - Licensor for the purpose of discussing and improving the Work, but - excluding communication that is conspicuously marked or otherwise - designated in writing by the copyright owner as "Not a Contribution." - - "Contributor" shall mean Licensor and any individual or Legal Entity - on behalf of whom a Contribution has been received by Licensor and - subsequently incorporated within the Work. - - 2. Grant of Copyright License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - copyright license to reproduce, prepare Derivative Works of, - publicly display, publicly perform, sublicense, and distribute the - Work and such Derivative Works in Source or Object form. - - 3. Grant of Patent License. Subject to the terms and conditions of - this License, each Contributor hereby grants to You a perpetual, - worldwide, non-exclusive, no-charge, royalty-free, irrevocable - (except as stated in this section) patent license to make, have made, - use, offer to sell, sell, import, and otherwise transfer the Work, - where such license applies only to those patent claims licensable - by such Contributor that are necessarily infringed by their - Contribution(s) alone or by combination of their Contribution(s) - with the Work to which such Contribution(s) was submitted. If You - institute patent litigation against any entity (including a - cross-claim or counterclaim in a lawsuit) alleging that the Work - or a Contribution incorporated within the Work constitutes direct - or contributory patent infringement, then any patent licenses - granted to You under this License for that Work shall terminate - as of the date such litigation is filed. - - 4. Redistribution. You may reproduce and distribute copies of the - Work or Derivative Works thereof in any medium, with or without - modifications, and in Source or Object form, provided that You - meet the following conditions: - - (a) You must give any other recipients of the Work or - Derivative Works a copy of this License; and - - (b) You must cause any modified files to carry prominent notices - stating that You changed the files; and - - (c) You must retain, in the Source form of any Derivative Works - that You distribute, all copyright, patent, trademark, and - attribution notices from the Source form of the Work, - excluding those notices that do not pertain to any part of - the Derivative Works; and - - (d) If the Work includes a "NOTICE" text file as part of its - distribution, then any Derivative Works that You distribute must - include a readable copy of the attribution notices contained - within such NOTICE file, excluding those notices that do not - pertain to any part of the Derivative Works, in at least one - of the following places: within a NOTICE text file distributed - as part of the Derivative Works; within the Source form or - documentation, if provided along with the Derivative Works; or, - within a display generated by the Derivative Works, if and - wherever such third-party notices normally appear. The contents - of the NOTICE file are for informational purposes only and - do not modify the License. You may add Your own attribution - notices within Derivative Works that You distribute, alongside - or as an addendum to the NOTICE text from the Work, provided - that such additional attribution notices cannot be construed - as modifying the License. - - You may add Your own copyright statement to Your modifications and - may provide additional or different license terms and conditions - for use, reproduction, or distribution of Your modifications, or - for any such Derivative Works as a whole, provided Your use, - reproduction, and distribution of the Work otherwise complies with - the conditions stated in this License. - - 5. Submission of Contributions. Unless You explicitly state otherwise, - any Contribution intentionally submitted for inclusion in the Work - by You to the Licensor shall be under the terms and conditions of - this License, without any additional terms or conditions. - Notwithstanding the above, nothing herein shall supersede or modify - the terms of any separate license agreement you may have executed - with Licensor regarding such Contributions. - - 6. Trademarks. This License does not grant permission to use the trade - names, trademarks, service marks, or product names of the Licensor, - except as required for reasonable and customary use in describing the - origin of the Work and reproducing the content of the NOTICE file. - - 7. Disclaimer of Warranty. Unless required by applicable law or - agreed to in writing, Licensor provides the Work (and each - Contributor provides its Contributions) on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or - implied, including, without limitation, any warranties or conditions - of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A - PARTICULAR PURPOSE. You are solely responsible for determining the - appropriateness of using or redistributing the Work and assume any - risks associated with Your exercise of permissions under this License. - - 8. Limitation of Liability. In no event and under no legal theory, - whether in tort (including negligence), contract, or otherwise, - unless required by applicable law (such as deliberate and grossly - negligent acts) or agreed to in writing, shall any Contributor be - liable to You for damages, including any direct, indirect, special, - incidental, or consequential damages of any character arising as a - result of this License or out of the use or inability to use the - Work (including but not limited to damages for loss of goodwill, - work stoppage, computer failure or malfunction, or any and all - other commercial damages or losses), even if such Contributor - has been advised of the possibility of such damages. - - 9. Accepting Warranty or Additional Liability. While redistributing - the Work or Derivative Works thereof, You may choose to offer, - and charge a fee for, acceptance of support, warranty, indemnity, - or other liability obligations and/or rights consistent with this - License. However, in accepting such obligations, You may act only - on Your own behalf and on Your sole responsibility, not on behalf - of any other Contributor, and only if You agree to indemnify, - defend, and hold each Contributor harmless for any liability - incurred by, or claims asserted against, such Contributor by reason - of your accepting any such warranty or additional liability. - - END OF TERMS AND CONDITIONS - - APPENDIX: How to apply the Apache License to your work. - - To apply the Apache License to your work, attach the following - boilerplate notice, with the fields enclosed by brackets "[]" - replaced with your own identifying information. (Don't include - the brackets!) The text should be enclosed in the appropriate - comment syntax for the file format. We also recommend that a - file or class name and description of purpose be included on the - same "printed page" as the copyright notice for easier - identification within third-party archives. - - Copyright [yyyy] [name of copyright owner] - - Licensed under the Apache License, Version 2.0 (the "License"); - you may not use this file except in compliance with the License. - You may obtain a copy of the License at - - http://www.apache.org/licenses/LICENSE-2.0 - - Unless required by applicable law or agreed to in writing, software - distributed under the License is distributed on an "AS IS" BASIS, - WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - See the License for the specific language governing permissions and - limitations under the License. diff --git a/skills/.experimental/wrapped/SKILL.md b/skills/.experimental/wrapped/SKILL.md deleted file mode 100644 index 6e28ae2..0000000 --- a/skills/.experimental/wrapped/SKILL.md +++ /dev/null @@ -1,40 +0,0 @@ ---- -name: "codex-wrapped" -description: "Generate a Codex Wrapped usage recap from local Codex logs, including last 30 days, last 7 days, and an all-time focus-hours callout. Use when the user asks for a usage summary, activity recap, or Codex Wrapped report." ---- - - -# Codex Wrapped - -Use this skill whenever the user wants a Codex Wrapped report or usage insights. Render text-only output (no image generation). - -The report must be year-agnostic and should highlight last 30 days and last 7 days, while still calling out all-time focus hours. - -## Quick Commands (run in order) - -1) **Compute stats** -```bash -python3 .codex/skills/codex-wrapped/scripts/get_codex_stats.py \ - --output /tmp/wrapped_stats.json -``` -(Defaults to the system timezone; override `--timezone` only if the user requests it.) - -2) **Render text report** -```bash -.codex/skills/codex-wrapped/scripts/report.sh \ - --stats-file /tmp/wrapped_stats.json -``` -This prints the report directly to stdout. - -## Files -- `scripts/get_codex_stats.py` -- computes rolling-window stats to `/tmp/wrapped_stats.json`. -- `scripts/report.sh` -- text report renderer. - -## Responding to the user -- Paste the report text exactly as printed, wrapped in triple backticks (```), to preserve spacing/box drawing. -- If something fails, state what you ran and the error. - -## Notes -- Keep `/tmp/wrapped_stats.json` unless sensitive; rerun stats if outdated. -- The report adapts to terminal width. Set `WRAPPED_WIDTH=120` (or similar) to force a wider layout. -- Layout options: default is `columns` (two-column). Use `--layout table` (or `WRAPPED_LAYOUT=table`) to switch back to the compact grid. diff --git a/skills/.experimental/wrapped/agents/openai.yaml b/skills/.experimental/wrapped/agents/openai.yaml deleted file mode 100644 index bb5d4ae..0000000 --- a/skills/.experimental/wrapped/agents/openai.yaml +++ /dev/null @@ -1,4 +0,0 @@ -interface: - display_name: "Wrapped" - short_description: "Create a Codex activity report from local usage data" - default_prompt: "Generate a Codex Wrapped usage report from my local Codex logs." diff --git a/skills/.experimental/wrapped/scripts/get_codex_stats.py b/skills/.experimental/wrapped/scripts/get_codex_stats.py deleted file mode 100644 index 404af47..0000000 --- a/skills/.experimental/wrapped/scripts/get_codex_stats.py +++ /dev/null @@ -1,451 +0,0 @@ -#!/usr/bin/env python3 -""" -Aggregate Codex usage metrics for the Wrapped report. - -Outputs JSON with rolling windows: -- all_time -- last_30_days -- last_7_days -""" - -from __future__ import annotations - -import argparse -import json -import sys -from collections import defaultdict -from dataclasses import dataclass, field -from datetime import datetime, timedelta, timezone -from pathlib import Path -from typing import Iterable -from zoneinfo import ZoneInfo, ZoneInfoNotFoundError - -CODEX_HOME = Path.home() / ".codex" -SESSION_DIRS = ["sessions", "archived_sessions"] -DEFAULT_TIMEZONE = None -DEFAULT_OUTPUT_PATH = Path(__file__).with_name("wrapped_stats.json") -WINDOW_DELTAS = { - "all_time": None, - "last_30_days": 30, - "last_7_days": 7, -} - - -@dataclass -class WindowAccumulator: - name: str - start: datetime | None - session_count: int = 0 - total_assistant_messages: int = 0 - total_user_messages: int = 0 - turn_usage_seconds: float = 0.0 - session_span_seconds: float = 0.0 - day_tokens: dict[datetime.date, int] = field(default_factory=lambda: defaultdict(int)) - active_days: set[datetime.date] = field(default_factory=set) - hour_usage: dict[int, int] = field(default_factory=lambda: defaultdict(int)) - repo_usage: dict[str, int] = field(default_factory=lambda: defaultdict(int)) - longest_turn_duration: float = 0.0 - longest_turn_timestamp: datetime | None = None - longest_turn_session: str | None = None - - -def parse_timestamp(ts: str) -> datetime: - if not ts: - raise ValueError("Empty timestamp") - if ts.endswith("Z"): - ts = ts[:-1] + "+00:00" - return datetime.fromisoformat(ts) - - -def iter_session_files() -> Iterable[Path]: - for rel in SESSION_DIRS: - root = CODEX_HOME / rel - if not root.exists(): - continue - yield from root.rglob("*.jsonl") - - -def format_tokens(value: int) -> str: - if value <= 0: - return "0" - if value >= 1_000_000: - scaled = value / 1_000_000 - return f"{scaled:.1f}M".replace(".0M", "M") - if value >= 1_000: - scaled = value / 1_000 - return f"{scaled:.1f}k".replace(".0k", "k") - return f"{value:,}" - - -def render_usage_hours(seconds: float) -> str: - if seconds <= 0: - return "0 minutes" - hours = seconds / 3600 - if hours < 1: - minutes = int(round(seconds / 60)) - minutes = max(minutes, 1) - return f"{minutes} minute{'s' if minutes != 1 else ''}" - rounded_hours = int(round(hours)) - rounded_hours = max(rounded_hours, 1) - return f"{rounded_hours} hour{'s' if rounded_hours != 1 else ''}" - - -def build_contrib_lines(active_days: set[datetime.date]) -> list[str]: - if not active_days: - return [] - - day_set = set(active_days) - first_date = min(day_set) - last_date = max(day_set) - - def to_sunday(dt: datetime.date) -> datetime.date: - offset = (dt.weekday() + 1) % 7 - return dt - timedelta(days=offset) - - def to_saturday(dt: datetime.date) -> datetime.date: - offset = 6 - ((dt.weekday() + 1) % 7) - return dt + timedelta(days=offset) - - start = to_sunday(first_date) - end = to_saturday(last_date) - total_days = (end - start).days + 1 - total_weeks = total_days // 7 - - weekday_labels = ["Sun", "Mon", "Tue", "Wed", "Thu", "Fri", "Sat"] - line_chars: list[list[str]] = [[] for _ in range(7)] - - for week_index in range(total_weeks): - week_start = start + timedelta(days=week_index * 7) - for day_offset in range(7): - current_day = week_start + timedelta(days=day_offset) - char = "•" if current_day in day_set else "◦" - line_chars[day_offset].append(char) - - contrib_lines = [ - f"{weekday_labels[idx]} {''.join(chars)}" for idx, chars in enumerate(line_chars) if chars - ] - return contrib_lines - - -def local_timezone_name() -> str: - local_tz = datetime.now().astimezone().tzinfo - if hasattr(local_tz, "key") and local_tz.key: - return local_tz.key - if local_tz: - return str(local_tz) - return "UTC" - - -def get_timezone(tz_name: str) -> ZoneInfo: - try: - return ZoneInfo(tz_name) - except ZoneInfoNotFoundError: - fallback = local_timezone_name() - try: - return ZoneInfo(fallback) - except ZoneInfoNotFoundError: - return ZoneInfo("UTC") - - -def classify_hour(hour: int | None) -> str: - if hour is None: - return "unknown" - if 22 <= hour or hour <= 3: - return "owl" - if 4 <= hour <= 9: - return "bird" - if 10 <= hour <= 16: - return "day" - return "eve" - - -def longest_streak(active_days: set[datetime.date]) -> int: - if not active_days: - return 0 - sorted_days = sorted(active_days) - longest = 1 - current = 1 - prev_day = sorted_days[0] - for day in sorted_days[1:]: - if day == prev_day + timedelta(days=1): - current += 1 - elif day == prev_day: - pass - else: - longest = max(longest, current) - current = 1 - prev_day = day - longest = max(longest, current) - return longest - - -def window_stats(window: WindowAccumulator, tz_abbrev: str) -> dict[str, object]: - total_tokens = sum(window.day_tokens.values()) - active_days_count = len(window.active_days) - peak_hour = None - if window.hour_usage: - peak_hour = max(window.hour_usage.items(), key=lambda item: item[1])[0] - - peak_hour_display = f"{peak_hour:02d}:00" if peak_hour is not None else "unknown" - peak_hour_label = classify_hour(peak_hour) - if peak_hour is not None: - peak_hour_slot = f"{peak_hour_display} {peak_hour_label}" - else: - peak_hour_slot = "unknown" - - biggest_day_date = None - biggest_day_tokens = 0 - if window.day_tokens: - biggest_day_date, biggest_day_tokens = max( - window.day_tokens.items(), key=lambda item: item[1] - ) - if biggest_day_date: - biggest_day_display = f"{biggest_day_date.day} {biggest_day_date.strftime('%b')}" - else: - biggest_day_display = "unknown" - - top_repo = None - if window.repo_usage: - top_repo = max(window.repo_usage.items(), key=lambda item: item[1])[0] - if top_repo: - repo_path = Path(top_repo) - top_repo_label = repo_path.name or str(repo_path) - else: - top_repo_label = "unknown" - - usage_seconds = max(window.session_span_seconds, window.turn_usage_seconds) - usage_hours_display = render_usage_hours(usage_seconds) - - streak_days = longest_streak(window.active_days) - usage_streak_display = ( - f"{streak_days} day" if streak_days == 1 else f"{streak_days} days" if streak_days else "unknown" - ) - - longest_turn_display = "unknown" - if window.longest_turn_timestamp is not None: - minutes = int(window.longest_turn_duration // 60) - seconds = int(round(window.longest_turn_duration % 60)) - longest_turn_display = f"{minutes}m {seconds}s" - - return { - "sessions": window.session_count, - "sessions_display": f"{window.session_count:,}", - "assistant_messages": window.total_assistant_messages, - "assistant_messages_display": f"{window.total_assistant_messages:,}", - "user_messages": window.total_user_messages, - "user_messages_display": f"{window.total_user_messages:,}", - "active_days": active_days_count, - "active_days_display": f"{active_days_count:,}", - "total_tokens": total_tokens, - "total_tokens_display": format_tokens(total_tokens), - "usage_hours_display": usage_hours_display, - "peak_hour": peak_hour, - "peak_hour_display": peak_hour_display, - "peak_hour_label": peak_hour_label, - "peak_hour_slot": peak_hour_slot, - "timezone_abbrev": tz_abbrev, - "usage_streak_days": streak_days, - "usage_streak_display": usage_streak_display, - "biggest_day_display": biggest_day_display, - "biggest_day_tokens": biggest_day_tokens, - "top_repo_display": top_repo_label, - "longest_turn_display": longest_turn_display, - } - - -def gather_metrics(tz_name: str) -> dict[str, object]: - target_tz = get_timezone(tz_name) - now_local = datetime.now(target_tz) - - windows: dict[str, WindowAccumulator] = {} - for name, delta in WINDOW_DELTAS.items(): - start = None if delta is None else now_local - timedelta(days=delta) - windows[name] = WindowAccumulator(name=name, start=start) - - joined_at: datetime | None = None - - for session_path in iter_session_files(): - current_turn_start: datetime | None = None - session_start: datetime | None = None - session_end: datetime | None = None - session_workspace: str | None = None - - try: - with session_path.open("r", encoding="utf-8") as handle: - for raw_line in handle: - raw_line = raw_line.strip() - if not raw_line: - continue - try: - record = json.loads(raw_line) - except json.JSONDecodeError: - continue - - ts_str = record.get("timestamp") - if not ts_str: - ts_str = record.get("payload", {}).get("timestamp") - if not ts_str: - continue - try: - ts = parse_timestamp(ts_str) - except ValueError: - continue - - local_ts = ts.astimezone(target_tz) - if session_start is None: - session_start = local_ts - session_end = local_ts - - rec_type = record.get("type") - payload = record.get("payload", {}) - - for window in windows.values(): - if window.start is None or local_ts >= window.start: - window.active_days.add(local_ts.date()) - - if rec_type == "response_item" and payload.get("type") == "message": - role = payload.get("role") - for window in windows.values(): - if window.start is None or local_ts >= window.start: - if role == "assistant": - window.total_assistant_messages += 1 - elif role == "user": - window.total_user_messages += 1 - window.hour_usage[local_ts.hour] += 1 - - if rec_type == "session_meta": - session_workspace = ( - payload.get("workspacePath") - or payload.get("workspace_path") - or payload.get("cwd") - or session_workspace - ) - - if rec_type == "turn_context": - session_workspace = ( - payload.get("workspacePath") - or payload.get("workspace_path") - or payload.get("cwd") - or session_workspace - ) - current_turn_start = ts - continue - - if rec_type == "event_msg" and payload.get("type") == "token_count": - info = payload.get("info") - if not info or current_turn_start is None: - continue - duration = (ts - current_turn_start).total_seconds() - if duration < 0: - duration = 0 - usage = info.get("last_token_usage") or info.get("total_token_usage") - tokens = 0 - if usage and usage.get("total_tokens"): - tokens = usage["total_tokens"] - - for window in windows.values(): - if window.start is None or local_ts >= window.start: - window.turn_usage_seconds += duration - if duration > window.longest_turn_duration: - window.longest_turn_duration = duration - window.longest_turn_timestamp = local_ts - window.longest_turn_session = session_path.name - if tokens: - window.day_tokens[local_ts.date()] += tokens - current_turn_start = None - - except OSError: - continue - - if session_start and (joined_at is None or session_start < joined_at): - joined_at = session_start - - if session_start and session_end: - for window in windows.values(): - if window.start is None: - window_start = session_start - else: - window_start = max(window.start, session_start) - window_end = min(session_end, now_local) - overlap = (window_end - window_start).total_seconds() - if overlap > 0: - window.session_span_seconds += overlap - window.session_count += 1 - if session_workspace: - window.repo_usage[session_workspace] += 1 - - tz_abbrev = now_local.strftime("%Z") - if joined_at: - joined_str = joined_at.date().isoformat() - joined_label = joined_at.strftime("%b %d") - days_diff = (now_local.date() - joined_at.date()).days - days_ago = f"{days_diff} day{'s' if days_diff != 1 else ''} ago" - joined_display = f"{joined_label} ({days_ago})" - else: - joined_str = "unknown" - days_ago = "unknown" - joined_display = "unknown" - - metrics: dict[str, object] = { - "timezone": tz_name, - "timezone_abbrev": tz_abbrev, - "generated_at": now_local.isoformat(), - "joined_date": joined_str, - "joined_days_ago": days_ago, - "joined_display": joined_display, - "joined_relative_display": days_ago, - "windows": {name: window_stats(win, tz_abbrev) for name, win in windows.items()}, - "contrib_lines": build_contrib_lines(windows["all_time"].active_days), - } - - return metrics - - -def print_text(metrics: dict[str, object]) -> None: - all_time = metrics.get("windows", {}).get("all_time", {}) - print(f"Joined Codex: {metrics.get('joined_display', 'unknown')}") - print(f"Sessions: {all_time.get('sessions_display', '0')}") - print(f"Assistant messages: {all_time.get('assistant_messages_display', '0')}") - print(f"Prompts: {all_time.get('user_messages_display', '0')}") - print(f"Tokens: {all_time.get('total_tokens_display', '0')}") - print(f"Usage time: {all_time.get('usage_hours_display', 'unknown')}") - print(f"Peak hour: {all_time.get('peak_hour_slot', 'unknown')}") - - -def main() -> None: - parser = argparse.ArgumentParser(description="Compute Codex Wrapped usage metrics.") - parser.add_argument( - "--json", - action="store_true", - help="Emit metrics as JSON for scripting.", - ) - parser.add_argument( - "--timezone", - default=local_timezone_name(), - help="IANA timezone name for local stats (defaults to system timezone).", - ) - parser.add_argument( - "--output", - default=str(DEFAULT_OUTPUT_PATH), - help=f"Path to save metrics JSON (default: {DEFAULT_OUTPUT_PATH}).", - ) - args = parser.parse_args() - - metrics = gather_metrics(args.timezone) - - output_path = Path(args.output).expanduser() - with output_path.open("w", encoding="utf-8") as handle: - json.dump(metrics, handle, indent=2) - handle.write("\n") - - if args.json: - json.dump(metrics, fp=sys.stdout, indent=2) - print() - return - - print(f"Wrote stats to {output_path}") - print_text(metrics) - - -if __name__ == "__main__": - main() diff --git a/skills/.experimental/wrapped/scripts/report.sh b/skills/.experimental/wrapped/scripts/report.sh deleted file mode 100755 index 7864838..0000000 --- a/skills/.experimental/wrapped/scripts/report.sh +++ /dev/null @@ -1,488 +0,0 @@ -#!/usr/bin/env bash - -# Config knobs -LABEL_MIN=5 -LABEL_MAX=15 -MIN_PANEL_WIDTH=80 -DEFAULT_PANEL_WIDTH=96 -FRAME_BUFFER=7 - -detect_term_width() { - local cols="" - if [[ -t 1 ]]; then - if [[ -n "${COLUMNS:-}" ]]; then - cols="$COLUMNS" - fi - if command -v tput >/dev/null 2>&1; then - local tcols - tcols=$(tput cols 2>/dev/null || true) - if [[ "$tcols" =~ ^[0-9]+$ ]]; then - cols="$tcols" - fi - fi - fi - if [[ "$cols" =~ ^[0-9]+$ && "$cols" -gt 0 ]]; then - echo "$cols" - fi -} - -TERM_WIDTH=$(detect_term_width) -if [[ -n "${WRAPPED_WIDTH:-}" && "${WRAPPED_WIDTH}" =~ ^[0-9]+$ ]]; then - TERM_WIDTH="$WRAPPED_WIDTH" -fi -if [[ -z "$TERM_WIDTH" ]]; then - TERM_WIDTH=$DEFAULT_PANEL_WIDTH -fi -DATA_LIMIT=$((TERM_WIDTH - 2)) -if (( DATA_LIMIT < MIN_PANEL_WIDTH )); then - DATA_LIMIT=$MIN_PANEL_WIDTH -fi - -DECOR_CHUNK=".:*~*:._.:*~*:._.:*~*:._.:*~*:._.:*~*:." -TITLE="OpenAI Codex Wrapped" -SUBTITLE="" - -# Defaults (overwritten by stats JSON) -joined_line="Member since unknown" -timezone_line="Local time: unknown" -table_rows=() -hero_line="" -highlight_lines=() -activity_lines=() -column_rows=() -column_head_left="Last 30 days" -column_head_right="Last 7 days" - -die() { echo "$*" >&2; exit 1; } - -show_help() { - cat <<'EOF' -Usage: report.sh --stats-file [--layout table|columns] - - --stats-file Path to the metrics JSON produced by get_codex_stats.py. - --layout Layout style (table or columns). Default: columns. - -h, --help Show this message and exit. -EOF -} - -strip_ansi_len() { - local raw="$1" - local clean - clean=$(printf '%s' "$raw" | perl -pe 's/\e\[[0-9;]*[a-zA-Z]//g') - printf '%s' "${#clean}" -} - -calc_panel_width() { - local max_seen=0 - for entry in "$@"; do - local span - span=$(strip_ansi_len "$entry") - (( span > max_seen )) && max_seen=$span - done - max_seen=$((max_seen + FRAME_BUFFER)) - if (( max_seen < MIN_PANEL_WIDTH )); then - max_seen=$MIN_PANEL_WIDTH - fi - if (( max_seen > DATA_LIMIT )); then - max_seen=$DATA_LIMIT - fi - echo "$max_seen" -} - -center_line() { - local text="$1" - local width="$2" - local text_len - text_len=$(strip_ansi_len "$text") - local left_pad=$(( (width - text_len) / 2 )) - local right_pad=$(( width - text_len - left_pad )) - printf "│%*s%s%*s│\n" "$left_pad" "" "$text" "$right_pad" "" -} - -divider_line() { - local width="$1" - local left="$2" - local mid="$3" - local right="$4" - local bar="$left" - for ((i=0; i l_width )); then - label=$(echo "$label" | cut -c1-$((l_width-3)))... - else - label=$(printf "%-${l_width}s" "$label") - fi - - if (( ${#value} > r_width )); then - value=$(echo "$value" | cut -c1-$((r_width-3)))... - else - value=$(printf "%-${r_width}s" "$value") - fi - - printf "│ %s │ %s │\n" "$label" "$value" -} - -left_line() { - local text="$1" - local width="$2" - local content=" $text" - content=$(fit_cell "$content" "$width") - printf "│%s│\n" "$content" -} - -fit_cell() { - local text="$1" - local width="$2" - local len - len=$(strip_ansi_len "$text") - if (( len > width )); then - if (( width > 3 )); then - text=$(echo "$text" | cut -c1-$((width-3)))... - else - text=$(echo "$text" | cut -c1-"$width") - fi - fi - printf "%-${width}s" "$text" -} - -center_cell() { - local text="$1" - local width="$2" - local text_len - text_len=$(strip_ansi_len "$text") - if (( text_len >= width )); then - printf "%s" "$(fit_cell "$text" "$width")" - return - fi - local left_pad=$(( (width - text_len) / 2 )) - local right_pad=$(( width - text_len - left_pad )) - printf "%*s%s%*s" "$left_pad" "" "$text" "$right_pad" "" -} - -trim_ws() { - local s="$1" - s="${s#"${s%%[![:space:]]*}"}" - s="${s%"${s##*[![:space:]]}"}" - printf "%s" "$s" -} - -load_stats() { - local src="$1" - [[ -r "$src" ]] || die "Unable to read stats file: $src" - local tsv - tsv=$(python3 - "$src" <<'PY' -import json, sys -from pathlib import Path - -data = json.loads(Path(sys.argv[1]).read_text()) -windows = data.get("windows", {}) -all_time = windows.get("all_time", {}) -last_30 = windows.get("last_30_days", {}) -last_7 = windows.get("last_7_days", {}) - -def triple(key, default="unknown"): - return ( - all_time.get(key, default), - last_30.get(key, default), - last_7.get(key, default), - ) - -def compress_time(value: str) -> str: - if not isinstance(value, str): - return str(value) - value = value.replace(" hours", "h").replace(" hour", "h") - value = value.replace(" minutes", "m").replace(" minute", "m") - return value - -def compress_days(value: str) -> str: - if not isinstance(value, str): - return str(value) - return value.replace(" days", "d").replace(" day", "d") - -def compress_peak(value: str) -> str: - if not isinstance(value, str): - return str(value) - return ( - value.replace(" bird", " b") - .replace(" owl", " o") - .replace(" day", " d") - .replace(" eve", " e") - ) - -def double_str(key, default="unknown", compress=None, collapse=False): - _, m, w = triple(key, default) - if compress: - m = compress(m) - w = compress(w) - if collapse and m == w: - return str(m) - return f"30d {m} | 7d {w}" - -rows = [ - ("Window", "Last 30 days | Last 7 days"), - ("Sessions", double_str("sessions_display", "0")), - ("Assistant messages", double_str("assistant_messages_display", "0")), - ("Tokens", double_str("total_tokens_display", "0")), - ("Usage time", double_str("usage_hours_display", "0")), - ("Biggest day", double_str("biggest_day_display", "unknown")), - ("Top repo", double_str("top_repo_display", "unknown", collapse=True)), - ("Longest turn", double_str("longest_turn_display", "unknown")), -] - -column_metrics = [ - ("Sessions", "sessions_display"), - ("Prompts", "user_messages_display"), - ("Tokens", "total_tokens_display"), - ("Usage time", "usage_hours_display"), - ("Biggest day", "biggest_day_display"), -] -stack_sections = [ - ("Last 30 days", "last_30_days"), - ("Last 7 days", "last_7_days"), -] - -joined_display = data.get("joined_display", "unknown") -timezone = data.get("timezone_abbrev") or data.get("timezone", "unknown") -def compress_focus(value: str) -> str: - return str(value) - -print(f"joined_line\tMember since {joined_display}") -print(f"timezone_line\tLocal time: {timezone}") -for label, value in rows: - print(f"row\t{label}\t{value}") - -def format_stack(stat_key: str, value: str) -> str: - return str(value) - -def bar(active: int, total: int, width: int = 10) -> str: - if total <= 0: - return "." * width - ratio = max(0.0, min(1.0, active / total)) - filled = int(round(ratio * width)) - return "#" * filled + "." * (width - filled) - -def percent(active: int, total: int) -> str: - if total <= 0: - return "0%" - return f"{int(round(active / total * 100))}%" - -all_time = windows.get("all_time", {}) -last_30 = windows.get("last_30_days", {}) -last_7 = windows.get("last_7_days", {}) - -high_30_prompts = last_30.get("user_messages_display", "0") -high_30_active = last_30.get("active_days_display", "0") -high_peak = last_30.get("peak_hour_slot", "unknown") -high_streak = last_30.get("usage_streak_display", "unknown") -high_focus = all_time.get("usage_hours_display", "unknown") - -print(f"hero_line\t{high_30_prompts} prompts in 30d — peak {high_peak}") -print(f"highlight_line\t{joined_display} | {timezone}") -print(f"highlight_line\tActive days: {high_30_active} in 30d | Streak: {high_streak}") -print(f"highlight_line\tAll-time focus: {high_focus}") - -act_30 = last_30.get("active_days", 0) -act_7 = last_7.get("active_days", 0) -bar_30 = bar(int(act_30), 30) -bar_7 = bar(int(act_7), 7) -print(f"activity_line\t30d activity: {bar_30} ({int(act_30)}/30)") -print(f"activity_line\t7d activity: {bar_7} ({int(act_7)}/7)") - -print("column_head\tLast 30 days\tLast 7 days") -for label, stat_key in column_metrics: - left = format_stack(stat_key, last_30.get(stat_key, "unknown")) - right = format_stack(stat_key, last_7.get(stat_key, "unknown")) - print(f"column_row\t{label}\t{left}\t{right}") -for line in data.get("contrib_lines", []): - print(f"contrib_line\t{line}") -PY - ) || die "Failed to parse stats JSON: $src" - - while IFS=$'\t' read -r key label value extra; do - case "$key" in - joined_line) joined_line="$label" ;; - timezone_line) timezone_line="$label" ;; - focus_line) ;; - row) table_rows+=("${label}"$'\t'"${value}") ;; - stack_section) stack_rows+=("SECTION"$'\t'"${label}") ;; - stack_item) stack_rows+=("ITEM"$'\t'"${label}"$'\t'"${value}") ;; - stack_blank) stack_rows+=("BLANK") ;; - hero_line) hero_line="$label" ;; - highlight_line) highlight_lines+=("$label") ;; - activity_line) activity_lines+=("$label") ;; - column_head) - column_head_left="$label" - column_head_right="$value" - ;; - column_row) - column_rows+=("${label}"$'\t'"${value}"$'\t'"${extra}") ;; - esac - done <<<"$tsv" -} - -render() { - local stats_path="$1" - local layout_override="${2:-}" - load_stats "$stats_path" - if ((${#contrib_lines[@]} == 0)); then - contrib_lines=("Sun " "Mon " "Tue " "Wed " "Thu " "Fri " "Sat ") - fi - - holiday_lines=() - - local layout="${WRAPPED_LAYOUT:-columns}" - if [[ -n "$layout_override" ]]; then - layout="$layout_override" - elif [[ -n "${REPORT_LAYOUT:-}" ]]; then - layout="${REPORT_LAYOUT}" - fi - - local width_input=("$DECOR_CHUNK" "$TITLE" "$SUBTITLE" "$hero_line") - width_input+=("${highlight_lines[@]}") - width_input+=("${activity_lines[@]}") - if [[ "$layout" == "table" ]]; then - for row in "${table_rows[@]}"; do - IFS=$'\t' read -r label value <<<"$row" - width_input+=("$label" "$value") - done - else - width_input+=("$column_head_left" "$column_head_right") - for row in "${column_rows[@]}"; do - IFS=$'\t' read -r label value_left value_right <<<"$row" - width_input+=("$label" "$value_left" "$value_right") - done - fi - PANEL_WIDTH=$(calc_panel_width "${width_input[@]}") - - local max_label=0 - for row in "${table_rows[@]}"; do - IFS=$'\t' read -r label value <<<"$row" - local span - span=$(strip_ansi_len "$label") - (( span > max_label )) && max_label=$span - done - local left_span=$max_label - (( left_span < LABEL_MIN )) && left_span=$LABEL_MIN - (( left_span > LABEL_MAX )) && left_span=$LABEL_MAX - local right_span=$((PANEL_WIDTH - 5 - left_span)) - if (( right_span < 20 )); then - right_span=20 - left_span=$((PANEL_WIDTH - 5 - right_span)) - (( left_span < LABEL_MIN )) && left_span=$LABEL_MIN - fi - local stack_rows_mode=0 - if (( right_span < 40 )); then - stack_rows_mode=1 - fi - - divider_line "$((PANEL_WIDTH))" "┌" "─" "┐" - local deco_line="$DECOR_CHUNK" - while (( $(strip_ansi_len "$deco_line") < PANEL_WIDTH )); do - deco_line+="$DECOR_CHUNK" - done - deco_line=${deco_line:0:PANEL_WIDTH} - printf "│%s│\n" "$deco_line" - center_line "$TITLE" "$PANEL_WIDTH" - if [[ -n "$SUBTITLE" ]]; then - center_line "$SUBTITLE" "$PANEL_WIDTH" - fi - if [[ -n "$hero_line" ]]; then - center_line "$hero_line" "$PANEL_WIDTH" - fi - printf "│%s│\n" "$deco_line" - - divider_line "$((PANEL_WIDTH))" "├" "─" "┤" - center_line "Highlights" "$PANEL_WIDTH" - for ln in "${highlight_lines[@]}"; do - left_line "$ln" "$PANEL_WIDTH" - done - for ln in "${activity_lines[@]}"; do - left_line "$ln" "$PANEL_WIDTH" - done - center_line "" "$PANEL_WIDTH" - - if [[ "$layout" == "table" ]]; then - local horiz_left="├"; local horiz_mid="┬"; local horiz_right="┤" - local lbar; printf -v lbar '%*s' $((left_span + 2)) ""; lbar=${lbar// /─} - local rbar; printf -v rbar '%*s' $((right_span + 2)) ""; rbar=${rbar// /─} - printf "%s%s%s%s%s\n" "$horiz_left" "$lbar" "$horiz_mid" "$rbar" "$horiz_right" - - for row in "${table_rows[@]}"; do - IFS=$'\t' read -r label value <<<"$row" - if (( stack_rows_mode == 1 )) && [[ "$label" == "Window" ]]; then - continue - fi - if (( stack_rows_mode == 1 )) && [[ "$value" == *"|"* ]]; then - IFS='|' read -r part_a part_b part_c <<<"$value" - part_a=$(trim_ws "$part_a") - part_b=$(trim_ws "$part_b") - part_c=$(trim_ws "$part_c") - split_row "$label" "$part_a" "$left_span" "$right_span" - if [[ -n "$part_b" ]]; then - split_row "" "$part_b" "$left_span" "$right_span" - fi - if [[ -n "$part_c" ]]; then - split_row "" "$part_c" "$left_span" "$right_span" - fi - else - split_row "$label" "$value" "$left_span" "$right_span" - fi - done - - local tail_left="└"; local tail_mid="┴"; local tail_right="┘" - printf "%s%s%s%s%s\n" "$tail_left" "$lbar" "$tail_mid" "$rbar" "$tail_right" - else - divider_line "$((PANEL_WIDTH))" "├" "─" "┤" - local inner_width=$((PANEL_WIDTH)) - local gap=" │ " - local gap_len=3 - local left_width=$(( (inner_width - gap_len) / 2 )) - local right_width=$(( inner_width - gap_len - left_width )) - local head_left - local head_right - head_left=$(center_cell "$column_head_left" "$left_width") - head_right=$(center_cell "$column_head_right" "$right_width") - printf "│%s%s%s│\n" "$head_left" "$gap" "$head_right" - - for row in "${column_rows[@]}"; do - IFS=$'\t' read -r label value_left value_right <<<"$row" - local ltext=" ${label}: ${value_left}" - local rtext=" ${label}: ${value_right}" - ltext=$(fit_cell "$ltext" "$left_width") - rtext=$(fit_cell "$rtext" "$right_width") - printf "│%s%s%s│\n" "$ltext" "$gap" "$rtext" - done - - divider_line "$((PANEL_WIDTH))" "└" "─" "┘" - fi -} - -main() { - local stats_path="" - local layout_arg="" - while [[ $# -gt 0 ]]; do - case "$1" in - --stats-file) stats_path="$2"; shift 2 ;; - --layout) layout_arg="$2"; shift 2 ;; - -h|--help) show_help; exit 0 ;; - *) show_help; exit 1 ;; - esac - done - [[ -n "$stats_path" ]] || die "--stats-file is required." - render "$stats_path" "$layout_arg" -} - -main "$@"