From aa907983dbad1e0bf25f7c91af54275ff07e762e Mon Sep 17 00:00:00 2001 From: slaguru666 <111923774+slaguru666@users.noreply.github.com> Date: Sun, 12 Apr 2026 14:00:55 +0100 Subject: [PATCH] Add portable Codex ops kit --- .gitignore | 5 ++ AGENTS.md | 33 ++++++++ README.md | 86 +++++++++++++++++++++ docs/research.md | 36 +++++++++ memory/README.md | 22 ++++++ memory/active-projects.md | 7 ++ memory/architecture-decisions.md | 15 ++++ memory/user-profile.md | 6 ++ memory/workflow-preferences.md | 8 ++ scripts/audit_codex_home.sh | 36 +++++++++ scripts/create_share_repo.sh | 29 +++++++ scripts/export_codex_portable.sh | 51 +++++++++++++ scripts/install_codex_kit.sh | 94 +++++++++++++++++++++++ scripts/prune_codex_sessions.sh | 66 ++++++++++++++++ scripts/session_to_memory.py | 109 +++++++++++++++++++++++++++ templates/global/AGENTS.md | 8 ++ templates/global/config-managed.toml | 27 +++++++ templates/global/config.toml | 27 +++++++ templates/project/AGENTS.md | 14 ++++ 19 files changed, 679 insertions(+) create mode 100644 .gitignore create mode 100644 AGENTS.md create mode 100644 README.md create mode 100644 docs/research.md create mode 100644 memory/README.md create mode 100644 memory/active-projects.md create mode 100644 memory/architecture-decisions.md create mode 100644 memory/user-profile.md create mode 100644 memory/workflow-preferences.md create mode 100755 scripts/audit_codex_home.sh create mode 100755 scripts/create_share_repo.sh create mode 100755 scripts/export_codex_portable.sh create mode 100755 scripts/install_codex_kit.sh create mode 100755 scripts/prune_codex_sessions.sh create mode 100755 scripts/session_to_memory.py create mode 100644 templates/global/AGENTS.md create mode 100644 templates/global/config-managed.toml create mode 100644 templates/global/config.toml create mode 100644 templates/project/AGENTS.md diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..275885c --- /dev/null +++ b/.gitignore @@ -0,0 +1,5 @@ +.DS_Store +__pycache__/ +*.pyc +export/ +tmp/ diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 0000000..4bc1488 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,33 @@ +# AGENTS.md + +This repository is a portable Codex operations kit. + +## Goal + +Keep Codex fast, cheap, and reusable across machines by separating: + +- small persistent instructions +- portable memory notes +- heavyweight transient session data + +## Working Rules + +- Treat this repo as the source of truth for reusable Codex setup. +- Keep `AGENTS.md` files short. Put durable user context in `memory/*.md`, not in giant prompt files. +- Prefer scripts in `scripts/` for install, audit, export, and maintenance tasks. +- Never copy secrets, auth tokens, or large SQLite/session databases into Git. +- Default to dry-run behavior for cleanup scripts unless the user explicitly asks to apply changes. + +## Key Paths + +- `README.md`: setup and rollout guide +- `docs/research.md`: research findings with source links +- `templates/`: portable Codex home templates +- `memory/`: durable cross-machine memory notes +- `scripts/`: audit, install, export, and maintenance tooling + +## Validation + +- Shell scripts should pass `bash -n`. +- Python scripts should pass `python3 -m py_compile`. +- Prefer idempotent scripts so re-running them is safe. diff --git a/README.md b/README.md new file mode 100644 index 0000000..f1ce17c --- /dev/null +++ b/README.md @@ -0,0 +1,86 @@ +# Codex Ops Kit + +This repository turns a local Codex setup into something you can keep lean, remember more usefully, and share across machines without dragging along secrets or multi-gigabyte session logs. + +## What We Improved + +- Added explicit Codex profiles for `economy`, `balanced`, and `deep` work. +- Added a short global `AGENTS.md` pattern that tells Codex where to find durable memory without bloating every session. +- Added portable memory files you can keep in Git and sync between machines. +- Added audit, export, install, and session-maintenance scripts. +- Documented research-backed guidance from current OpenAI docs plus community findings. + +## Findings From This Machine + +Audit date: `2026-04-12` + +- `~/.codex/sessions` was about `2.8G`. +- `~/.codex/state_5.sqlite` was about `227M`. +- `~/.codex/memories` was empty. +- Current config only set the model and plugin flags, so there were no cost-oriented profiles or memory workflow conventions in place. + +The biggest issue was not a lack of raw stored data. It was that durable, reusable knowledge was not being separated from bulky transcript history. + +## Repo Layout + +- `docs/research.md` +- `templates/global/AGENTS.md` +- `templates/global/config.toml` +- `memory/` +- `scripts/` + +## Local Setup + +Install this kit into your current Codex home: + +```bash +bash scripts/install_codex_kit.sh +``` + +Create a portable export bundle you can move to another machine: + +```bash +bash scripts/export_codex_portable.sh +``` + +Audit your live Codex home: + +```bash +bash scripts/audit_codex_home.sh +``` + +Preview old session files that are taking space: + +```bash +bash scripts/prune_codex_sessions.sh +``` + +Archive old session files after reviewing the plan: + +```bash +bash scripts/prune_codex_sessions.sh --apply +``` + +Extract a markdown summary skeleton from a large session transcript: + +```bash +python3 scripts/session_to_memory.py ~/.codex/sessions/2026/03/19/rollout-....jsonl +``` + +## Recommended Workflow + +Use `balanced` for normal work, `economy` for quick edits and low-risk tasks, and `deep` only for the hard stuff. + +Keep `AGENTS.md` concise. Put stable personal preferences, project summaries, and long-lived decisions into the markdown files under `memory/`. + +When a long session finishes, distill the durable facts into memory notes instead of relying on Codex to rediscover them from old transcript files. + +Run the audit script periodically. If session storage balloons again, archive the biggest old transcripts instead of letting `~/.codex/sessions` become your accidental memory system. + +## Profiles + +- `economy`: lowest-cost default for shallow work +- `balanced`: everyday profile +- `deep`: higher-effort work when architecture or debugging really needs it + +These profiles are defined in [templates/global/config.toml](/Users/timevans/Documents/New%20project/templates/global/config.toml). The installer preserves machine-specific plugin and trust settings while merging the managed cost and memory block into your live `~/.codex/config.toml`. diff --git a/docs/research.md b/docs/research.md new file mode 100644 index 0000000..0cee4f8 --- /dev/null +++ b/docs/research.md @@ -0,0 +1,36 @@ +# Research Notes + +Research date: `2026-04-12` + +## Reliable Takeaways + +1. OpenAI's `Codex Prompting Guide` says `medium` reasoning effort is the recommended all-around default, while `high` or `xhigh` should be reserved for harder work. It also calls out compaction as a first-class way to support long-running conversations without hitting context limits. +Source: [Codex Prompting Guide](https://developers.openai.com/cookbook/examples/gpt-5/codex_prompting_guide) + +2. OpenAI's configuration reference supports profile-scoped overrides, including `profiles..model`, `profiles..model_reasoning_effort`, `profiles..model_instructions_file`, `profiles..web_search`, `project_doc_max_bytes`, and `history.max_bytes`. +Source: [Configuration Reference](https://developers.openai.com/codex/config-reference) + +3. OpenAI's slash-command docs show that `/status` exposes token usage and `/statusline` can persist footer items like model, context, limits, tokens, and session id. That makes it easier to notice expensive sessions before they sprawl. +Source: [Slash commands in Codex CLI](https://developers.openai.com/codex/cli/slash-commands) + +4. The `GPT-5.3-Codex` model page lists a large discount for cached input tokens compared with regular input tokens. This supports a strategy of keeping stable instructions and repeated context in consistent, reusable front-loaded text instead of rewriting them ad hoc every session. +Source: [GPT-5.3-Codex model page](https://developers.openai.com/api/docs/models/gpt-5.3-codex) + +5. An arXiv paper published in 2026 found that adding repository-level `AGENTS.md` files was associated with lower median runtime and reduced output-token usage while keeping comparable task completion behavior. +Source: [On the Impact of AGENTS.md Files on the Efficiency of AI Coding Agents](https://arxiv.org/abs/2601.20404) + +## Community Signals + +1. Community writeups consistently recommend small, specific `AGENTS.md` files instead of large prose dumps. The basic pattern is to keep the instruction layer lean and push durable knowledge into separate reusable notes. +Source: [Codex CLI reference guide](https://blakecrosley.com/guides/codex) + +2. There is active experimentation around external or MCP-backed memory systems so that project knowledge survives across restarts and across tools. I treated these as directional input rather than source-of-truth guidance because they are not official OpenAI docs. +Source: [Memory for Codex discussion](https://www.reddit.com/r/codex/comments/1rmhtmr/memory_for_codex/) + +## Design Decisions For This Kit + +- Keep the always-loaded instruction surface short. +- Use profiles instead of one global "best" model choice. +- Separate durable markdown memory from bulky session history. +- Default maintenance scripts to audit and dry-run, not deletion. +- Export only portable assets: config, instructions, memory, and scripts. Do not export auth or giant local databases. diff --git a/memory/README.md b/memory/README.md new file mode 100644 index 0000000..d900416 --- /dev/null +++ b/memory/README.md @@ -0,0 +1,22 @@ +# Portable Memory + +These markdown files are the durable layer you can sync across machines. + +## What Belongs Here + +- user preferences that stay true across projects +- stable workflow conventions +- active project summaries +- important decisions worth remembering later + +## What Does Not Belong Here + +- secrets +- API keys +- giant transcript dumps +- temporary debugging notes +- anything that is likely to be stale within a day or two + +## Operating Principle + +Short, curated memory beats large raw history. diff --git a/memory/active-projects.md b/memory/active-projects.md new file mode 100644 index 0000000..d5b8c7c --- /dev/null +++ b/memory/active-projects.md @@ -0,0 +1,7 @@ +# Active Projects + +## Codex Ops Kit + +- Purpose: make Codex cheaper to run, easier to resume, and easier to share across machines. +- Core components: profile-based config, short global instructions, portable memory markdown, and maintenance scripts. +- Important constraint: portable assets should exclude secrets and large transient databases. diff --git a/memory/architecture-decisions.md b/memory/architecture-decisions.md new file mode 100644 index 0000000..f506f9e --- /dev/null +++ b/memory/architecture-decisions.md @@ -0,0 +1,15 @@ +# Architecture Decisions + +## Separate Durable Memory From Transcript History + +Durable preferences and project summaries live in curated markdown files. + +Large session JSONL files are treated as archival evidence that can be summarized or archived, not as the primary continuity layer. + +## Prefer Profiles Over One Global Setting + +There is no single best cost/performance configuration for every task. This kit uses `economy`, `balanced`, and `deep` profiles so the user can pick the right tradeoff quickly. + +## Keep Instruction Files Lean + +Small, targeted `AGENTS.md` files improve efficiency more reliably than giant all-purpose prompt files. diff --git a/memory/user-profile.md b/memory/user-profile.md new file mode 100644 index 0000000..7666019 --- /dev/null +++ b/memory/user-profile.md @@ -0,0 +1,6 @@ +# User Profile + +- Prefers Codex setups that are portable and Git-friendly. +- Wants lower token and credit usage by default without losing the option to do deeper work when needed. +- Wants Codex to retain durable context so recurring setup explanations are minimized. +- Prefers practical systems that can be shared with other Codex instances on other machines. diff --git a/memory/workflow-preferences.md b/memory/workflow-preferences.md new file mode 100644 index 0000000..6d94e56 --- /dev/null +++ b/memory/workflow-preferences.md @@ -0,0 +1,8 @@ +# Workflow Preferences + +- Default to a balanced cost/performance profile for everyday work. +- Use cheaper profiles for small edits, reviews, and narrow tasks. +- Reserve higher reasoning settings for architecture, hard debugging, or multi-file refactors. +- Keep always-loaded instructions short. +- Distill reusable knowledge into markdown memory notes after substantial work. +- Periodically audit local Codex storage and archive old session transcripts instead of relying on them as memory. diff --git a/scripts/audit_codex_home.sh b/scripts/audit_codex_home.sh new file mode 100755 index 0000000..178f5cc --- /dev/null +++ b/scripts/audit_codex_home.sh @@ -0,0 +1,36 @@ +#!/usr/bin/env bash +set -euo pipefail + +CODEX_HOME="${CODEX_HOME:-$HOME/.codex}" + +if [[ ! -d "$CODEX_HOME" ]]; then + echo "Codex home not found: $CODEX_HOME" >&2 + exit 1 +fi + +echo "Codex home: $CODEX_HOME" +echo +echo "Top-level sizes" +du -sh "$CODEX_HOME"/* "$CODEX_HOME"/.[!.]* 2>/dev/null | sort -h + +echo +echo "Large session files" +python3 - "$CODEX_HOME" <<'PY' +from pathlib import Path +import sys + +root = Path(sys.argv[1]) / "sessions" +files = sorted((p for p in root.rglob("*.jsonl") if p.is_file()), key=lambda p: p.stat().st_size, reverse=True) +print(f"session files: {len(files)}") +for path in files[:10]: + size_mb = path.stat().st_size / (1024 * 1024) + print(f"{size_mb:9.1f} MB {path}") +PY + +echo +echo "Config snapshot" +sed -n '1,220p' "$CODEX_HOME/config.toml" 2>/dev/null || true + +echo +echo "Memory markdown files" +find "$CODEX_HOME/memories" -type f -name '*.md' 2>/dev/null | sort || true diff --git a/scripts/create_share_repo.sh b/scripts/create_share_repo.sh new file mode 100755 index 0000000..2911dba --- /dev/null +++ b/scripts/create_share_repo.sh @@ -0,0 +1,29 @@ +#!/usr/bin/env bash +set -euo pipefail + +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +DEST="${1:-$HOME/Documents/codex-ops-kit}" + +rm -rf "$DEST" +mkdir -p "$DEST" + +copy_path() { + local rel="$1" + local src="$REPO_ROOT/$rel" + local dest="$DEST/$rel" + mkdir -p "$(dirname "$dest")" + cp -R "$src" "$dest" +} + +copy_path README.md +copy_path AGENTS.md +copy_path .gitignore +copy_path docs +copy_path memory +copy_path templates +copy_path scripts + +find "$DEST/scripts" -name '__pycache__' -type d -prune -exec rm -rf {} + +find "$DEST" -name '*.pyc' -delete + +echo "Created clean shareable repo at: $DEST" diff --git a/scripts/export_codex_portable.sh b/scripts/export_codex_portable.sh new file mode 100755 index 0000000..1fb6bc4 --- /dev/null +++ b/scripts/export_codex_portable.sh @@ -0,0 +1,51 @@ +#!/usr/bin/env bash +set -euo pipefail + +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +CODEX_HOME="${CODEX_HOME:-$HOME/.codex}" +STAMP="$(date +%Y%m%d-%H%M%S)" +DEST="${1:-$REPO_ROOT/export/codex-portable-$STAMP}" + +mkdir -p "$DEST" +mkdir -p "$DEST/codex-home" + +copy_if_exists() { + local src="$1" + local dest="$2" + if [[ -e "$src" ]]; then + mkdir -p "$(dirname "$dest")" + cp -R "$src" "$dest" + fi +} + +copy_if_exists "$CODEX_HOME/config.toml" "$DEST/codex-home/config.toml" +copy_if_exists "$CODEX_HOME/AGENTS.md" "$DEST/codex-home/AGENTS.md" +copy_if_exists "$CODEX_HOME/memories" "$DEST/codex-home/memories" +copy_if_exists "$CODEX_HOME/rules" "$DEST/codex-home/rules" +copy_if_exists "$REPO_ROOT/templates" "$DEST/templates" +copy_if_exists "$REPO_ROOT/memory" "$DEST/memory" +copy_if_exists "$REPO_ROOT/scripts" "$DEST/scripts" +copy_if_exists "$REPO_ROOT/docs" "$DEST/docs" +copy_if_exists "$REPO_ROOT/README.md" "$DEST/README.md" + +cat > "$DEST/EXPORT_NOTES.txt" <&2 + exit 1 +fi + +python3 - "$SESSIONS_DIR" "$ARCHIVE_DIR" "$OLDER_THAN_DAYS" "$MIN_SIZE_MB" "$APPLY" <<'PY' +from datetime import datetime, timedelta +from pathlib import Path +import shutil +import sys + +sessions_dir = Path(sys.argv[1]) +archive_dir = Path(sys.argv[2]) +older_than_days = int(sys.argv[3]) +min_size_mb = int(sys.argv[4]) +apply = int(sys.argv[5]) + +cutoff = datetime.now() - timedelta(days=older_than_days) +candidates = [] + +for path in sessions_dir.rglob("*.jsonl"): + stat = path.stat() + mtime = datetime.fromtimestamp(stat.st_mtime) + size_mb = stat.st_size / (1024 * 1024) + if mtime < cutoff and size_mb >= min_size_mb: + candidates.append((size_mb, mtime, path)) + +candidates.sort(reverse=True) + +if not candidates: + print("No session files matched the archive rules.") + raise SystemExit(0) + +print("Candidates:") +for size_mb, mtime, path in candidates: + print(f"{size_mb:9.1f} MB {mtime:%Y-%m-%d} {path}") + +if not apply: + print() + print("Dry run only. Re-run with --apply to move these files to:") + print(archive_dir) + raise SystemExit(0) + +for _, _, path in candidates: + rel = path.relative_to(sessions_dir) + dest = archive_dir / rel + dest.parent.mkdir(parents=True, exist_ok=True) + shutil.move(str(path), str(dest)) + +print() +print(f"Archived {len(candidates)} session files to {archive_dir}") +PY diff --git a/scripts/session_to_memory.py b/scripts/session_to_memory.py new file mode 100755 index 0000000..3e5184d --- /dev/null +++ b/scripts/session_to_memory.py @@ -0,0 +1,109 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import json +import re +import sys +from collections import OrderedDict +from pathlib import Path + + +def uniq(items: list[str]) -> list[str]: + return list(OrderedDict((item, None) for item in items if item)) + + +def extract_texts(obj) -> list[str]: + texts: list[str] = [] + if isinstance(obj, dict): + if obj.get("type") in {"input_text", "output_text"} and isinstance(obj.get("text"), str): + texts.append(obj["text"]) + for value in obj.values(): + texts.extend(extract_texts(value)) + elif isinstance(obj, list): + for value in obj: + texts.extend(extract_texts(value)) + return texts + + +def main() -> int: + if len(sys.argv) not in {2, 3}: + print("usage: session_to_memory.py SESSION_JSONL [OUTPUT_MD]", file=sys.stderr) + return 2 + + session_path = Path(sys.argv[1]).expanduser().resolve() + output_path = Path(sys.argv[2]).expanduser().resolve() if len(sys.argv) == 3 else session_path.with_suffix(".memory.md") + + user_texts: list[str] = [] + assistant_texts: list[str] = [] + paths: list[str] = [] + session_meta = {} + + path_pattern = re.compile(r"(\/[A-Za-z0-9._ \-\/]+)") + + with session_path.open() as handle: + for raw_line in handle: + raw_line = raw_line.strip() + if not raw_line: + continue + event = json.loads(raw_line) + payload = event.get("payload", {}) + if event.get("type") == "session_meta": + session_meta = payload + if event.get("type") == "response_item" and payload.get("type") == "message": + role = payload.get("role") + texts = extract_texts(payload.get("content", [])) + if role == "user": + user_texts.extend(texts) + elif role == "assistant": + assistant_texts.extend(texts) + for text in texts: + paths.extend(path_pattern.findall(text)) + + user_texts = uniq([text.strip() for text in user_texts if text.strip()]) + assistant_texts = uniq([text.strip() for text in assistant_texts if text.strip()]) + paths = uniq([path.strip() for path in paths if path.strip()]) + + output = [] + output.append(f"# Session Memory Draft") + output.append("") + output.append(f"- Session file: `{session_path}`") + if session_meta.get("timestamp"): + output.append(f"- Session timestamp: `{session_meta['timestamp']}`") + if session_meta.get("cwd"): + output.append(f"- Working directory: `{session_meta['cwd']}`") + output.append("") + output.append("## User Requests") + output.append("") + for text in user_texts[:20]: + output.append(f"- {text.replace(chr(10), ' ')}") + output.append("") + output.append("## Candidate Paths Mentioned") + output.append("") + for path in paths[:30]: + output.append(f"- `{path}`") + output.append("") + output.append("## Assistant Notes To Distill") + output.append("") + output.append("- Copy only stable preferences, project facts, and durable decisions into your memory markdown files.") + output.append("- Do not copy transient debugging output or large transcript excerpts.") + output.append("- Summarize in your own words before adding to long-term memory.") + output.append("") + output.append("## Sample Durable Summary") + output.append("") + output.append("- Project:") + output.append("- User preference:") + output.append("- Important decision:") + output.append("- Follow-up worth remembering:") + output.append("") + output.append("## Assistant Messages Snapshot") + output.append("") + for text in assistant_texts[:10]: + output.append(f"- {text.replace(chr(10), ' ')}") + + output_path.write_text("\n".join(output) + "\n") + print(output_path) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/templates/global/AGENTS.md b/templates/global/AGENTS.md new file mode 100644 index 0000000..360e9a2 --- /dev/null +++ b/templates/global/AGENTS.md @@ -0,0 +1,8 @@ +# Global Codex Guidance + +- Prefer concise context over long repeated explanations. +- Use project `AGENTS.md` first, then consult `~/.codex/memories/` only for durable user preferences or project history relevant to the current task. +- Do not bulk-load all memory files. Read only the notes that match the current repository or request. +- Treat session transcripts as temporary evidence, not primary memory. +- When a task reveals durable preferences, recurring workflows, or stable project facts, update the relevant markdown note in `~/.codex/memories/`. +- Keep edits portable across machines whenever possible. diff --git a/templates/global/config-managed.toml b/templates/global/config-managed.toml new file mode 100644 index 0000000..83b0dde --- /dev/null +++ b/templates/global/config-managed.toml @@ -0,0 +1,27 @@ +model = "gpt-5.4" +model_reasoning_effort = "medium" +profile = "balanced" +project_doc_max_bytes = 16384 +web_search = "cached" + +[history] +max_bytes = 10485760 +persistence = "save-all" + +[profiles.economy] +model = "gpt-5.4-mini" +model_reasoning_effort = "low" +plan_mode_reasoning_effort = "low" +web_search = "cached" + +[profiles.balanced] +model = "gpt-5.4" +model_reasoning_effort = "medium" +plan_mode_reasoning_effort = "medium" +web_search = "cached" + +[profiles.deep] +model = "gpt-5.4" +model_reasoning_effort = "high" +plan_mode_reasoning_effort = "high" +web_search = "live" diff --git a/templates/global/config.toml b/templates/global/config.toml new file mode 100644 index 0000000..83b0dde --- /dev/null +++ b/templates/global/config.toml @@ -0,0 +1,27 @@ +model = "gpt-5.4" +model_reasoning_effort = "medium" +profile = "balanced" +project_doc_max_bytes = 16384 +web_search = "cached" + +[history] +max_bytes = 10485760 +persistence = "save-all" + +[profiles.economy] +model = "gpt-5.4-mini" +model_reasoning_effort = "low" +plan_mode_reasoning_effort = "low" +web_search = "cached" + +[profiles.balanced] +model = "gpt-5.4" +model_reasoning_effort = "medium" +plan_mode_reasoning_effort = "medium" +web_search = "cached" + +[profiles.deep] +model = "gpt-5.4" +model_reasoning_effort = "high" +plan_mode_reasoning_effort = "high" +web_search = "live" diff --git a/templates/project/AGENTS.md b/templates/project/AGENTS.md new file mode 100644 index 0000000..e14c76e --- /dev/null +++ b/templates/project/AGENTS.md @@ -0,0 +1,14 @@ +# AGENTS.md + +This project uses a lean Codex instruction strategy. + +## Purpose + +Store only repository-specific guidance here. + +## Rules + +- Put cross-project preferences in `~/.codex/memories/`, not here. +- Keep this file short enough that it is cheap to load every session. +- Point to concrete files, scripts, or commands instead of restating large amounts of context. +- Prefer maintenance scripts over one-off shell sequences.