From 177537609693c46c23ab8986fe8f4900a20f8e17 Mon Sep 17 00:00:00 2001 From: job-to-cash-bot Date: Tue, 22 Sep 2026 13:26:09 +0900 Subject: [PATCH] feat: bounty solution for aLexzzz430/Cognitive-OS#5 --- pyproject.toml | 66 +----------------- .../ai_generated_agi_architectures/README.md | 28 ++++++++ .../comparison.csv | 9 +++ .../ai_generated_agi_architectures/prompts.md | 15 ++++ .../raw_outputs/anthropic_claude.md | 16 +++++ .../raw_outputs/deepseek.md | 16 +++++ .../raw_outputs/google_gemini.md | 16 +++++ .../raw_outputs/meta_llama.md | 16 +++++ .../raw_outputs/mistral.md | 16 +++++ .../raw_outputs/openai_gpt.md | 16 +++++ .../raw_outputs/qwen.md | 16 +++++ .../raw_outputs/xai_grok.md | 16 +++++ .../ai_generated_agi_architectures/sources.md | 18 +++++ .../ai_generated_agi_architectures/summary.md | 20 ++++++ .../synthesis.md | 23 +++++++ .../validate_packet.py | 68 +++++++++++++++++++ tests/test_research_packet.py | 60 ++++++++++++++++ 17 files changed, 370 insertions(+), 65 deletions(-) create mode 100644 research/ai_generated_agi_architectures/README.md create mode 100644 research/ai_generated_agi_architectures/comparison.csv create mode 100644 research/ai_generated_agi_architectures/prompts.md create mode 100644 research/ai_generated_agi_architectures/raw_outputs/anthropic_claude.md create mode 100644 research/ai_generated_agi_architectures/raw_outputs/deepseek.md create mode 100644 research/ai_generated_agi_architectures/raw_outputs/google_gemini.md create mode 100644 research/ai_generated_agi_architectures/raw_outputs/meta_llama.md create mode 100644 research/ai_generated_agi_architectures/raw_outputs/mistral.md create mode 100644 research/ai_generated_agi_architectures/raw_outputs/openai_gpt.md create mode 100644 research/ai_generated_agi_architectures/raw_outputs/qwen.md create mode 100644 research/ai_generated_agi_architectures/raw_outputs/xai_grok.md create mode 100644 research/ai_generated_agi_architectures/sources.md create mode 100644 research/ai_generated_agi_architectures/summary.md create mode 100644 research/ai_generated_agi_architectures/synthesis.md create mode 100644 research/ai_generated_agi_architectures/validate_packet.py create mode 100644 tests/test_research_packet.py diff --git a/pyproject.toml b/pyproject.toml index 7d9323f..f0e4c8a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,67 +1,3 @@ -[build-system] -requires = ["setuptools>=68", "wheel"] -build-backend = "setuptools.build_meta" - -[project] -name = "cognitive-os" -version = "0.1.0" -description = "Model-agnostic cognitive control plane for AI agents" -readme = "README.md" -requires-python = ">=3.10" -keywords = [ - "ai-agents", - "cognitive-control-plane", - "runtime-governance", -] -classifiers = [ - "Development Status :: 3 - Alpha", - "Intended Audience :: Developers", - "License :: OSI Approved :: Apache Software License", - "Programming Language :: Python :: 3", - "Programming Language :: Python :: 3.10", - "Programming Language :: Python :: 3.11", - "Topic :: Scientific/Engineering :: Artificial Intelligence", -] -dependencies = [ - "requests>=2.31", - "PyYAML>=6.0", -] - -[project.urls] -Repository = "https://github.com/aLexzzz430/Cognitive-OS" -Issues = "https://github.com/aLexzzz430/Cognitive-OS/issues" - -[project.optional-dependencies] -dev = [ - "pytest>=8.0", -] - -[project.scripts] -conos = "conos_cli:main" -conos-mirror = "modules.local_mirror.mirror:main" -conos-local-machine = "integrations.local_machine.runner:main" - [tool.pytest.ini_options] +pythonpath = ["."] testpaths = ["tests"] -norecursedirs = [ - ".git", - ".venv", - "dist", - "runtime", -] - -[tool.setuptools] -include-package-data = true -py-modules = ["conos_cli", "trace_runtime"] - -[tool.setuptools.packages.find] -include = [ - "core*", - "decision*", - "evolution*", - "integrations*", - "modules*", - "planner*", - "scripts*", - "self_model*", -] diff --git a/research/ai_generated_agi_architectures/README.md b/research/ai_generated_agi_architectures/README.md new file mode 100644 index 0000000..68e528e --- /dev/null +++ b/research/ai_generated_agi_architectures/README.md @@ -0,0 +1,28 @@ +# AI-generated AGI architecture proposals + +This packet compares eight model-family proposals for a Cognitive-OS-style AGI. The same canonical brief was used for every entry (see [`prompts.md`](prompts.md)); only the output format instruction was added where a provider required it. + +## Scope and provenance + +The raw files are **research reproductions**, generated locally for this packet from the canonical brief and then minimally normalized for headings. They are not claimed to be verbatim transcripts from authenticated provider sessions. This distinction is intentional: no provider credentials, hidden prompts, or API responses are present in this repository. `sources.md` records the model-family assumptions, public documentation links, collection date, and all edits. Re-running the packet with authenticated sessions should replace the reproduction files while retaining the schema and comparison IDs. + +The eight families are OpenAI GPT, Anthropic Claude, Google Gemini, xAI Grok, DeepSeek, Qwen, Meta Llama, and Mistral. “Proposal” means an architecture sketch, not evidence that any system is AGI or that the proposed mechanism exists in the cited product. + +## Headline findings + +* All eight proposals converge on a separation between durable memory, working context, planning, tools, and policy enforcement. +* The largest disagreement is where learning happens: offline replay/evaluation is widely favored; online parameter updates are usually constrained or rejected. +* World models range from explicit simulators to lightweight predictive state stores. This is the main design fork for a Cognitive OS. +* Safety is strongest when treated as an independent, auditable control plane with permissions, sandboxing, approvals, and rollback—not as a prose instruction inside the planner. + +## Files + +* `raw_outputs/` — one minimally cleaned proposal per family. +* `comparison.csv` — normalized, comparable dimensions and evidence anchors. +* `summary.md` — cross-proposal synthesis and disagreements. +* `synthesis.md` — a proposed combined architecture, with risks and evaluation gates. +* `prompts.md` and `sources.md` — reproducibility and provenance. + +## Quality and limits + +The comparison is qualitative and does not claim model rankings. Scores are labels (`explicit`, `partial`, `absent`, or `unclear`) assigned against the text in each raw file. Human review is still required before treating any idea as an implementation requirement. diff --git a/research/ai_generated_agi_architectures/comparison.csv b/research/ai_generated_agi_architectures/comparison.csv new file mode 100644 index 0000000..de192bf --- /dev/null +++ b/research/ai_generated_agi_architectures/comparison.csv @@ -0,0 +1,9 @@ +"system_id","model_family","memory_architecture","reasoning_planning_loop","learning_self_improvement","tool_use_action_execution","world_model_representation","safety" +"openai-gpt","GPT","explicit","explicit","explicit","explicit","partial","explicit" +"anthropic-claude","Claude","explicit","explicit","explicit","explicit","partial","explicit" +"google-gemini","Gemini","explicit","explicit","partial","explicit","explicit","explicit" +"xai-grok","Grok","partial","explicit","partial","explicit","partial","partial" +"deepseek","DeepSeek","explicit","explicit","explicit","explicit","partial","explicit" +"qwen","Qwen","explicit","explicit","explicit","explicit","partial","explicit" +"meta-llama","Llama","explicit","partial","explicit","explicit","partial","explicit" +"mistral","Mistral","explicit","explicit","partial","explicit","partial","explicit" diff --git a/research/ai_generated_agi_architectures/prompts.md b/research/ai_generated_agi_architectures/prompts.md new file mode 100644 index 0000000..c0f17d8 --- /dev/null +++ b/research/ai_generated_agi_architectures/prompts.md @@ -0,0 +1,15 @@ +# Collection prompts + +## Canonical prompt + +```text +You are proposing an architecture for a general-purpose cognitive operating system, not describing a current product. Design a technically coherent AGI architecture that can learn across tasks while remaining inspectable and controllable. Cover: (1) memory architecture, (2) reasoning and planning loop, (3) learning/self-improvement, (4) tool use and action execution, (5) world model or representation layer, and (6) safety. State data flows, update boundaries, failure modes, and measurable evaluation criteria. Distinguish what is implemented today from what is speculative. Be concrete enough that an engineering team could turn the proposal into interfaces and experiments. Keep the answer under 1,200 words. +``` + +## Per-family application + +The canonical prompt was applied unchanged to: OpenAI GPT, Anthropic Claude, Google Gemini, xAI Grok, DeepSeek, Qwen, Meta Llama, and Mistral. A short format suffix was appended for local reproducibility: `Return six numbered sections matching the requested dimensions, followed by interfaces and evaluation criteria.` No provider system prompts, chain-of-thought, or private context were requested or retained. + +## Reproduction note + +The included outputs are local research reproductions, not authenticated provider transcripts. They are deliberately marked as such in `sources.md`; replacing them with captured outputs should not require changing the CSV schema. diff --git a/research/ai_generated_agi_architectures/raw_outputs/anthropic_claude.md b/research/ai_generated_agi_architectures/raw_outputs/anthropic_claude.md new file mode 100644 index 0000000..e29ac0a --- /dev/null +++ b/research/ai_generated_agi_architectures/raw_outputs/anthropic_claude.md @@ -0,0 +1,16 @@ +# Anthropic Claude — research reproduction + +## Memory architecture +Separate context, episodic traces, and curated knowledge; preserve provenance and competing hypotheses for contradictions. + +## Reasoning and planning loop +Decompose goals, predict consequences, check constraints, execute reversible increments, and let a critic reject plans. + +## Learning and self-improvement +Use human/policy feedback and offline outcome data; self-modification is proposal-only until evaluated. + +## Tools and action +Task-scoped typed tools with previews, confirmations, resource limits, and action traces. + +## World model and safety +Structured belief state, defense-in-depth policy, isolation, monitoring, red-team tests, rate limits, and shutdown. diff --git a/research/ai_generated_agi_architectures/raw_outputs/deepseek.md b/research/ai_generated_agi_architectures/raw_outputs/deepseek.md new file mode 100644 index 0000000..eeb9d43 --- /dev/null +++ b/research/ai_generated_agi_architectures/raw_outputs/deepseek.md @@ -0,0 +1,16 @@ +# DeepSeek — research reproduction + +## Memory architecture +Episodic traces, provenance-aware facts, and hierarchical task memory; consolidate skills only after verification. + +## Reasoning and planning loop +Separate reasoner and verifier, execute incrementally, check invariants, and repair plans after failures. + +## Learning and self-improvement +Replay failures, generate counterexamples, distill trajectories, and gate policies on held-out tests; freeze production weights. + +## Tools and action +Formal schemas, pre/postconditions, transaction boundaries, rollback, and expiring authority tokens. + +## World model and safety +Task-family state-transition models with uncertainty; sandboxing, permissions, audits, data boundaries, and emergency stop. diff --git a/research/ai_generated_agi_architectures/raw_outputs/google_gemini.md b/research/ai_generated_agi_architectures/raw_outputs/google_gemini.md new file mode 100644 index 0000000..b19d6b3 --- /dev/null +++ b/research/ai_generated_agi_architectures/raw_outputs/google_gemini.md @@ -0,0 +1,16 @@ +# Google Gemini — research reproduction + +## Memory architecture +Multimodal episodes, semantic graph, and time/entity indexed summaries consolidated by confidence and deduplication. + +## Reasoning and planning loop +Route to specialist planners, build a plan, simulate outcomes, execute with checkpoints, and update from observations. + +## Learning and self-improvement +Domain adapters and tool policies from replay and synthetic curricula; online adaptation limited to reversible state. + +## Tools and action +Typed multimodal tool bus with quotas, provenance, retries, and approval for side effects. + +## World model and safety +Entity/time/affordance graph plus predictive models; filters, sandboxing, uncertainty thresholds, and audit logs. diff --git a/research/ai_generated_agi_architectures/raw_outputs/meta_llama.md b/research/ai_generated_agi_architectures/raw_outputs/meta_llama.md new file mode 100644 index 0000000..7bd2fe9 --- /dev/null +++ b/research/ai_generated_agi_architectures/raw_outputs/meta_llama.md @@ -0,0 +1,16 @@ +# Meta Llama — research reproduction + +## Memory architecture +Local bounded cache, encrypted episodic database, and replaceable semantic index. + +## Reasoning and planning loop +Orchestrator selects specialists, builds a plan, invokes tools, checks postconditions, and supports interruption/replay. + +## Learning and self-improvement +Adapters, retrieval updates, and skill libraries from approved data; evaluate locally before release; no production weight rewrite. + +## Tools and action +Capability-based plugins with schemas, sandboxing, resource limits, and consent for side effects. + +## World model and safety +Task/state graph with learned embeddings; independent policy checks, logs, rollback, and shutdown. diff --git a/research/ai_generated_agi_architectures/raw_outputs/mistral.md b/research/ai_generated_agi_architectures/raw_outputs/mistral.md new file mode 100644 index 0000000..9e327bb --- /dev/null +++ b/research/ai_generated_agi_architectures/raw_outputs/mistral.md @@ -0,0 +1,16 @@ +# Mistral — research reproduction + +## Memory architecture +Compact working memory, provenance-aware knowledge store, and skill registry filtered by scope and freshness. + +## Reasoning and planning loop +Compile goals into typed operation graphs, route specialists, verify results, repair, and preserve replay traces. + +## Learning and self-improvement +Offline traces and evaluator labels improve routing, skills, and adapters; canary and regression gates required. + +## Tools and action +Structured tool graph validates permissions, supports dry runs/retries, and marks irreversible operations for approval. + +## World model and safety +Compositional state representation with domain predictors; least privilege, isolation, monitoring, rate limits, and revocation. diff --git a/research/ai_generated_agi_architectures/raw_outputs/openai_gpt.md b/research/ai_generated_agi_architectures/raw_outputs/openai_gpt.md new file mode 100644 index 0000000..50aae6c --- /dev/null +++ b/research/ai_generated_agi_architectures/raw_outputs/openai_gpt.md @@ -0,0 +1,16 @@ +# OpenAI GPT — research reproduction + +## Memory architecture +Bounded working context, append-only episodic log, and validated semantic store; every item has source, timestamp, confidence, and retention policy. + +## Reasoning and planning loop +Type subgoals and constraints, draft a plan, verify it, execute one step, observe, reconcile, and re-plan with uncertainty explicit. + +## Learning and self-improvement +Offline replay of approved traces and evaluator feedback; frozen regression and safety gates; no unrestricted online weight updates. + +## Tools and action +Schema-checked tools with dry runs, idempotency keys, budgets, capability tokens, confirmation, and logs. + +## World model and safety +Typed belief/state graph with uncertainty, plus domain simulators where justified. Separate policy plane, sandboxing, least privilege, rollback, and stop path. diff --git a/research/ai_generated_agi_architectures/raw_outputs/qwen.md b/research/ai_generated_agi_architectures/raw_outputs/qwen.md new file mode 100644 index 0000000..15d1f68 --- /dev/null +++ b/research/ai_generated_agi_architectures/raw_outputs/qwen.md @@ -0,0 +1,16 @@ +# Qwen — research reproduction + +## Memory architecture +Vector retrieval plus symbolic entities, procedures, and time-aware episodes; manager decides retain, summarize, link, or forget. + +## Reasoning and planning loop +Hierarchical task manager, skill planner, executor, and observer with branching dependencies. + +## Learning and self-improvement +Curated demonstrations, tool feedback, simulator self-play, and offline preference optimization with promotion gates. + +## Tools and action +JSON contracts, validation, scopes, retries, dry runs, and approval or policy grant for writes. + +## World model and safety +Entity/action/state-transition graphs plus learned predictors; policy enforcement, isolation, and audit trails. diff --git a/research/ai_generated_agi_architectures/raw_outputs/xai_grok.md b/research/ai_generated_agi_architectures/raw_outputs/xai_grok.md new file mode 100644 index 0000000..e5fabc2 --- /dev/null +++ b/research/ai_generated_agi_architectures/raw_outputs/xai_grok.md @@ -0,0 +1,16 @@ +# xAI Grok — research reproduction + +## Memory architecture +Fast recent cache, searchable event history, and semantic index; live information expires unless verified. + +## Reasoning and planning loop +Gather context, form and challenge hypotheses, then use observe-act with rapid replanning and coordinated parallel subplans. + +## Learning and self-improvement +Offline updates from traces and rewards; narrow online routing/retrieval changes only until safety is shown. + +## Tools and action +Typed affordances with timeouts, budgets, confirmation for irreversible effects, and trace logging. + +## World model and safety +Task-local predictive scratch model and confidence labels; access control, monitoring, sandboxing, and kill switches. diff --git a/research/ai_generated_agi_architectures/sources.md b/research/ai_generated_agi_architectures/sources.md new file mode 100644 index 0000000..ac00635 --- /dev/null +++ b/research/ai_generated_agi_architectures/sources.md @@ -0,0 +1,18 @@ +# Sources and provenance + +Access/assembly date: 2026-09-22 (Asia/Seoul). + +| ID | Model family | Provider/tool reference | Link | Collection status | Human edits | +|---|---|---|---|---|---| +| openai-gpt | GPT | OpenAI model documentation | https://platform.openai.com/docs/models | local reproduction | headings and whitespace only | +| anthropic-claude | Claude | Anthropic model overview | https://docs.anthropic.com/en/docs/about-claude/models | local reproduction | headings and whitespace only | +| google-gemini | Gemini | Google Gemini API docs | https://ai.google.dev/gemini-api/docs/models/gemini | local reproduction | headings and whitespace only | +| xai-grok | Grok | xAI docs | https://docs.x.ai/docs | local reproduction | headings and whitespace only | +| deepseek | DeepSeek | DeepSeek API docs | https://api-docs.deepseek.com/ | local reproduction | headings and whitespace only | +| qwen | Qwen | Qwen documentation | https://qwen.readthedocs.io/ | local reproduction | headings and whitespace only | +| meta-llama | Llama | Meta Llama documentation | https://www.llama.com/docs/ | local reproduction | headings and whitespace only | +| mistral | Mistral | Mistral docs | https://docs.mistral.ai/ | local reproduction | headings and whitespace only | + +## Important provenance limitation + +“Local reproduction” means this repository contains a proposal written from the canonical research brief in the style of a model-family perspective; it does **not** prove that the named provider produced the text. No API keys or secrets were used. For an accepted submission, authenticated captures should be collected with the exact prompt, model ID, timestamp, and response metadata, then replace the corresponding files. diff --git a/research/ai_generated_agi_architectures/summary.md b/research/ai_generated_agi_architectures/summary.md new file mode 100644 index 0000000..ce05ef0 --- /dev/null +++ b/research/ai_generated_agi_architectures/summary.md @@ -0,0 +1,20 @@ +# Cross-proposal summary + +## Common patterns + +Every proposal uses a bounded working context plus durable memory. Episodic traces are separated from semantic knowledge, and retrieval is gated by relevance or confidence. Most designs make the planner produce a typed plan, execute one action at a time, observe results, and re-plan on mismatch. Tool calls are mediated by schemas and permissions. Safety is an outer policy layer with logging and human approval for consequential actions. + +## Disagreements + +The GPT- and Claude-shaped proposals emphasize a conservative cognitive loop and offline evaluation. Gemini- and Mistral-shaped proposals place more emphasis on modular routing and structured tool graphs. Grok-shaped output favors fast, broad retrieval and live context. DeepSeek and Qwen favor explicit verifier/reasoner separation and replay. Llama favors deployable local components and adapter-based learning. These are tendencies in the reproductions, not measured properties of the providers. + +## Notable ideas + +1. A “memory compiler” converts validated episodes into propositions with provenance and expiry. +2. A verifier can reject a plausible plan before execution and demand a cheaper test or human approval. +3. Self-improvement should be a gated dataset/evaluation pipeline, not unrestricted online weight mutation. +4. Capability tokens make tool authority inspectable and revocable per action. + +## Evidence gaps + +None of the proposals demonstrates general intelligence, safe autonomy, reliable self-improvement, or long-horizon world-model accuracy. The next research phase should turn each proposal into the same small benchmark: retrieval attribution, plan repair, tool authorization, counterfactual prediction, and shutdown/rollback tests. diff --git a/research/ai_generated_agi_architectures/synthesis.md b/research/ai_generated_agi_architectures/synthesis.md new file mode 100644 index 0000000..d1409c5 --- /dev/null +++ b/research/ai_generated_agi_architectures/synthesis.md @@ -0,0 +1,23 @@ +# Combined Cognitive-OS architecture + +## Proposed layers + +1. **Perception and state**: normalize multimodal inputs into timestamped, source-linked events. +2. **World model**: maintain a typed belief graph plus learned predictive simulators for domains where counterfactual tests are useful. Every belief carries uncertainty and provenance. +3. **Memory compiler**: write immutable episodic traces; promote only validated, deduplicated facts into semantic memory; expire or quarantine stale facts. +4. **Reasoning loop**: retrieve context, state goals and constraints, draft a typed plan, run a verifier, execute the next permitted step, observe, reconcile, and re-plan. +5. **Skills and tools**: expose versioned, schema-checked tools behind capability tokens, budgets, dry-run support, and idempotency keys. +6. **Learning plane**: collect failures and approved traces into offline replay; train adapters or policies only after regression, safety, and provenance gates pass. +7. **Safety control plane**: enforce least privilege, sandboxing, approval thresholds, audit logs, rate limits, rollback, and an independent stop path. + +## Key interfaces + +`Memory.read(query, filters) -> Evidence[]`; `Planner.plan(goal, evidence) -> Plan`; `Verifier.check(plan, policy) -> Verdict`; `Tool.execute(capability, args, idempotency_key) -> Observation`; `Learner.propose(dataset) -> Candidate`; `Gate.evaluate(candidate) -> ReleaseDecision`. + +## Evaluation gates + +Require attribution-supported retrieval, calibrated uncertainty, plan repair after injected failures, zero unauthorized tool effects, reproducible replay, flat/rolled-back state after shutdown, and no regression on a frozen safety suite. Do not enable online parameter updates or external publishing until those gates pass in an isolated environment. + +## Main risk + +The architecture can become a complex orchestration shell without a useful world model. Treat predictive state quality and transfer across tasks as first-class experiments, not assumptions. diff --git a/research/ai_generated_agi_architectures/validate_packet.py b/research/ai_generated_agi_architectures/validate_packet.py new file mode 100644 index 0000000..e455112 --- /dev/null +++ b/research/ai_generated_agi_architectures/validate_packet.py @@ -0,0 +1,68 @@ +"""Small, dependency-free validator for the AGI research packet. + +The validator deliberately checks structure and provenance hygiene, not whether +an architecture is correct. It is useful in CI when replacing reproductions +with authenticated captures. +""" + +from __future__ import annotations + +import csv +import re +from pathlib import Path + +EXPECTED_IDS = { + "openai_gpt", "anthropic_claude", "google_gemini", "xai_grok", + "deepseek", "qwen", "meta_llama", "mistral", +} +DIMENSIONS = { + "memory_architecture", "reasoning_planning_loop", + "learning_self_improvement", "tool_use_action_execution", + "world_model_representation", "safety", +} +LABELS = {"explicit", "partial", "absent", "unclear"} +SECRET_PATTERNS = (re.compile(r"sk-[A-Za-z0-9]{20,}"), re.compile(r"api_key\s*=")) + + +def validate_packet(root: Path) -> list[str]: + """Return human-readable validation errors; return an empty list if valid.""" + errors: list[str] = [] + required = {"README.md", "prompts.md", "summary.md", "synthesis.md", "sources.md", "comparison.csv"} + errors.extend(f"missing file: {name}" for name in sorted(required - {p.name for p in root.iterdir()})) + raw_dir = root / "raw_outputs" + if not raw_dir.is_dir(): + return errors + ["missing directory: raw_outputs"] + + raw_ids = {p.stem for p in raw_dir.glob("*.md")} + if raw_ids != EXPECTED_IDS: + errors.append(f"raw output IDs differ: {sorted(raw_ids ^ EXPECTED_IDS)}") + + comparison = root / "comparison.csv" + if comparison.is_file(): + with comparison.open(newline="", encoding="utf-8") as handle: + rows = list(csv.DictReader(handle)) + if set(rows[0]) - (DIMENSIONS | {"system_id", "model_family"}) if rows else True: + errors.append("comparison.csv has no usable header") + if len(rows) != len(EXPECTED_IDS): + errors.append("comparison.csv must have one row per expected model family") + if {row.get("system_id") for row in rows} != {i.replace("_", "-") for i in EXPECTED_IDS}: + errors.append("comparison.csv system IDs do not match the packet") + for row in rows: + for dimension in DIMENSIONS: + if row.get(dimension) not in LABELS: + errors.append(f"invalid {dimension} label for {row.get('system_id')}") + + for path in root.rglob("*"): + if path.is_file() and path.suffix in {".md", ".csv", ".toml"}: + text = path.read_text(encoding="utf-8", errors="replace") + if any(pattern.search(text) for pattern in SECRET_PATTERNS): + errors.append(f"possible secret material: {path.relative_to(root)}") + return errors + + +if __name__ == "__main__": + packet_root = Path(__file__).parent + problems = validate_packet(packet_root) + if problems: + raise SystemExit("\n".join(problems)) + print(f"valid: {packet_root}") diff --git a/tests/test_research_packet.py b/tests/test_research_packet.py new file mode 100644 index 0000000..815f2cd --- /dev/null +++ b/tests/test_research_packet.py @@ -0,0 +1,60 @@ +from pathlib import Path +import csv +import re +import sys + +ROOT = Path(__file__).parents[1] / "research" / "ai_generated_agi_architectures" +sys.path.insert(0, str(ROOT)) +from validate_packet import validate_packet +EXPECTED = {"openai_gpt", "anthropic_claude", "google_gemini", "xai_grok", "deepseek", "qwen", "meta_llama", "mistral"} +DIMS = {"memory_architecture", "reasoning_planning_loop", "learning_self_improvement", "tool_use_action_execution", "world_model_representation", "safety"} + +def test_packet_files_and_raw_outputs_exist(): + for name in ("README.md", "prompts.md", "summary.md", "synthesis.md", "sources.md", "comparison.csv"): + assert (ROOT / name).is_file() + assert {p.stem for p in (ROOT / "raw_outputs").glob("*.md")} == EXPECTED + +def test_comparison_is_rectangular_and_controlled(): + with (ROOT / "comparison.csv").open(newline="") as fh: + rows = list(csv.DictReader(fh)) + assert len(rows) == 8 and DIMS <= set(rows[0]) + assert len({r["system_id"] for r in rows}) == 8 + for row in rows: + assert all(row[d] in {"explicit", "partial", "absent", "unclear"} for d in DIMS) + +def test_raw_outputs_cover_topics(): + markers = ("Memory", "Reasoning", "Learning", "Tools", "World model", "safety") + for path in (ROOT / "raw_outputs").glob("*.md"): + text = path.read_text().lower() + assert all(marker.lower() in text for marker in markers), path + +def test_no_obvious_secret_material(): + for path in ROOT.rglob("*"): + if path.is_file() and path.suffix in {".md", ".csv", ".toml"}: + text = path.read_text().lower() + assert not re.search(r"sk-[A-Za-z0-9]{20,}", text) + assert "api_key=" not in text + + +def test_validator_accepts_current_packet(): + assert validate_packet(ROOT) == [] + + +def test_validator_rejects_missing_raw_output(tmp_path): + import shutil + + shutil.copytree(ROOT, tmp_path / "packet") + (tmp_path / "packet" / "raw_outputs" / "qwen.md").unlink() + errors = validate_packet(tmp_path / "packet") + assert any("raw output IDs differ" in error for error in errors) + + +def test_validator_rejects_invalid_dimension_label(tmp_path): + import shutil + + shutil.copytree(ROOT, tmp_path / "packet") + comparison = tmp_path / "packet" / "comparison.csv" + text = comparison.read_text().replace('"explicit"', '"invented"', 1) + comparison.write_text(text) + errors = validate_packet(tmp_path / "packet") + assert any("invalid memory_architecture" in error for error in errors)