From 7046e3446854c3cddbb0a96611ca422658982741 Mon Sep 17 00:00:00 2001 From: Bhumi Bhardwaj Date: Wed, 29 Jul 2026 16:57:17 +0530 Subject: [PATCH 1/3] get_day_summary: add include_patterns option (#17) --- CHANGELOG.md | 5 +++ README.md | 2 +- src/activity_frames/mcp_server.py | 61 ++++++++++++++++++++----------- tests/test_mcp.py | 19 ++++++++++ 4 files changed, 65 insertions(+), 22 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 32fdb24..4d9ecba 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,11 @@ All notable changes to this project are documented here. The format follows [Keep a Changelog](https://keepachangelog.com/), and versions follow semantic versioning. The document schema version is tracked separately in [SPEC.md](SPEC.md). +### Added +- `get_day_summary` MCP tool: optional `include_patterns` (with `pattern_days`) + appends the same repeated-workflow counts as `get_patterns` to the summary, + so an agent wanting both no longer needs two calls (#17). + ## [0.2.2] - 2026-07-28 ### Added diff --git a/README.md b/README.md index 0c53c92..3f95759 100644 --- a/README.md +++ b/README.md @@ -89,7 +89,7 @@ Passively-captured activity becomes **deterministic action** - and the cheapest claude mcp add activity-frames -- aframes mcp ``` -Any MCP client works: command `aframes`, args `["mcp"]`. Six tools: `get_context`, `get_activity`, `get_steps` (expand one activity frame into its ordered click-by-click script - the replay view of a demonstrated run, so an agent can repeat the task instead of re-deriving it), `get_day_summary`, `get_patterns` (repetitive-workflow detection: repeated clicks, action sequences, URL patterns, app-switching loops, daily habits), and `get_communications` (email/messaging surfaces with the window titles seen on each — for many clients the title carries the subject or conversation name; a client that doesn't title its windows with the conversation leaves only its presence to report. Titles only, measured tier: message bodies are never read). +Any MCP client works: command `aframes`, args `["mcp"]`. Six tools: `get_context`, `get_activity`, `get_steps` (expand one activity frame into its ordered click-by-click script - the replay view of a demonstrated run, so an agent can repeat the task instead of re-deriving it), `get_day_summary` (pass `include_patterns` to append the same repeated-workflow counts as `get_patterns`, so a caller can get both in one call), `get_patterns` (repetitive-workflow detection: repeated clicks, action sequences, URL patterns, app-switching loops, daily habits), and `get_communications` (email/messaging surfaces with the window titles seen on each — for many clients the title carries the subject or conversation name; a client that doesn't title its windows with the conversation leaves only its presence to report. Titles only, measured tier: message bodies are never read). ## Use it from Python diff --git a/src/activity_frames/mcp_server.py b/src/activity_frames/mcp_server.py index 880b8d1..7e2ad0a 100644 --- a/src/activity_frames/mcp_server.py +++ b/src/activity_frames/mcp_server.py @@ -117,22 +117,38 @@ }, }, { - "name": "get_day_summary", - "description": ( - "Get coverage and per-app usage for a local day: first/last " - "activity, active minutes, away gaps, minutes per app, session " - "counts. Lighter than get_activity." - ), - "inputSchema": { - "type": "object", - "properties": { - "day": { - "type": "string", - "description": "Local day YYYY-MM-DD (default today)", - } + "name": "get_day_summary", + "description": ( + "Get coverage and per-app usage for a local day: first/last " + "activity, active minutes, away gaps, minutes per app, session " + "counts. Lighter than get_activity. Set include_patterns to " + "also append repeated-workflow counts from get_patterns " + "(kind/label/count only, no prose) so both fit in one call." + ), + "inputSchema": { + "type": "object", + "properties": { + "day": { + "type": "string", + "description": "Local day YYYY-MM-DD (default today)", + }, + "include_patterns": { + "type": "boolean", + "description": ( + "Also return repeated-workflow counts " + "(same data as get_patterns; default false)" + ), + }, + "pattern_days": { + "type": "number", + "description": ( + "Lookback window for patterns when include_patterns " + "is true (default 7, same default as get_patterns)" + ), }, }, }, +}, { "name": "get_patterns", "description": ( @@ -244,15 +260,14 @@ def get_steps(self, frame: str = "", day: str | None = None, ensure_ascii=False, ) - def get_day_summary(self, day: str | None = None) -> str: + def get_day_summary(self, day: str | None = None, include_patterns: bool = False, pattern_days: int = 7) -> str: doc = self.log.day(day, min_minutes=1.0) d = doc.to_dict() from ._time import local_day_string, local_day_window_utc start, end = local_day_window_utc(day or local_day_string()) apps = app_ledger(self.log.db, start, end) - return json.dumps( - { + out = { "day": d["window"].get("day"), "coverage": d["coverage"], "apps": [ @@ -265,11 +280,15 @@ def get_day_summary(self, day: str | None = None) -> str: } for a in apps[:15] ], - }, - indent=2, - ensure_ascii=False, - ) - + } + if include_patterns: + pats = self.log.patterns(int(pattern_days)) + out["patterns"] = [ + {"kind": p.kind, "label": p.label, "count": p.count} + for p in pats[:40] + ] + return json.dumps(out, indent=2, ensure_ascii=False) + def get_patterns(self, days: int = 7) -> str: pats = self.log.patterns(int(days)) return json.dumps( diff --git a/tests/test_mcp.py b/tests/test_mcp.py index 32df747..988de13 100644 --- a/tests/test_mcp.py +++ b/tests/test_mcp.py @@ -80,7 +80,26 @@ def test_unknown_tool_and_method(fixture_db): unknown = _rpc(s, "no/such/method") assert unknown["error"]["code"] == -32601 +def test_tool_call_get_day_summary_include_patterns(fixture_db): + s = _make_server(fixture_db) + plain = _rpc(s, "tools/call", { + "name": "get_day_summary", + "arguments": {"day": "2026-07-04"}, + }) + plain_payload = json.loads(plain["result"]["content"][0]["text"]) + assert "patterns" not in plain_payload + with_patterns = _rpc(s, "tools/call", { + "name": "get_day_summary", + "arguments": {"day": "2026-07-04", "include_patterns": True}, + }) + payload = json.loads(with_patterns["result"]["content"][0]["text"]) + assert "coverage" in payload and "apps" in payload + assert isinstance(payload["patterns"], list) + for p in payload["patterns"]: + assert set(p.keys()) == {"kind", "label", "count"} + + def test_serve_loop_over_stdio(fixture_db): s = _make_server(fixture_db) lines = [ From 6439e294dbbe0d06e0ab137dcb8449d864dc59e2 Mon Sep 17 00:00:00 2001 From: Nossa Date: Sat, 1 Aug 2026 23:51:53 -0700 Subject: [PATCH 2/3] update readme --- README.md | 51 ++++++++++++++++++++++++++++++++++---------------- pyproject.toml | 2 +- 2 files changed, 36 insertions(+), 17 deletions(-) diff --git a/README.md b/README.md index 0c53c92..ed48d9c 100644 --- a/README.md +++ b/README.md @@ -12,11 +12,11 @@ > **[Download the desktop app](https://usenocta.app)** - Nocta uses activity-frames to watch how you work and brief you daily on what needs your attention. 100% local. -**Episodic memory for AI agents - and the routines they can replay.** +**Turn your workday into structured workflows agents can execute.** -Your agent can read your code, search the web, and call APIs - but it has no idea what you've been doing all day, so it starts every conversation blind. And when it runs a task for you, it works it out from scratch every time, even one you've done a hundred times. +Computer-use agents work every task out from scratch, even one you've done a hundred times. And between tasks, your agent has no idea what you've been doing all day, so it starts every conversation blind. -activity-frames fixes both. It records your screen locally and compiles what it sees into structured **activity frames**: bounded, deterministic episodes of tasks you actually did. The recurring ones compile into **routines a computer-use agent can use** instead of working them out again. So it does your repetitive computer tasks **cheaper** (enriching a compiled routine costs almost no tokens) and **more reliable** (the same steps, grounded the same way every time, instead of guessing from a screenshot). +activity-frames fixes both. It records your screen locally and compiles what it sees into structured **activity frames**: bounded, deterministic records of the tasks you actually did. The recurring ones become **workflows an agent can execute** instead of working out again. So your repetitive computer tasks get done **cheaper** (running a compiled workflow costs almost no tokens) and **more reliable** (the same steps, grounded the same way every time, instead of guessing from a screenshot) - and everything else becomes context your agent can use. ```bash pip install activity-frames @@ -28,7 +28,7 @@ aframes context # your last 2 hours, agent-ready Capture stores instants: thousands of snapshot rows a day, each one saying "at 22:53:05, Chrome showed linkedin.com/in/...". Useless to reason over. -activity-frames compiles those instants into episodes: +activity-frames compiles those instants into activity frames: ```yaml - id: f-0007 @@ -59,28 +59,47 @@ away: 18:47-20:24 (97m) Drop that into a prompt and your agent knows your day. A full day compiles in under a second and costs zero tokens. -## Episodic memory, done honestly +## Workflows agents can execute -Agent memory today means conversation memory: what you told the model. Episodic memory is what you actually *did* - and the hard part is representing it without lying. +Computer-use agents re-derive every task from scratch - screenshot, reason, act, repeat - even for a workflow they've run a hundred times. That re-derivation is where the token cost goes, and it's waste: the workflow hasn't changed. -activity-frames enforces a two-tier contract ([SPEC.md](SPEC.md)): +Because activity-frames compiles recurring activity deterministically, a task you've demonstrated becomes an executable script: -- **Tier 1, measured (this package):** everything is derivable by deterministic code from capture data - sessions, durations, typed page entities, input volume, coverage gaps. No interpretation, no intent labels. Same input, same output, every time. -- **Tier 2, inferred (optional extension):** tools that add interpretation must namespace it, tag confidence (`high | medium | speculative`), and link evidence. Facts and guesses can never silently mix. +```bash +aframes steps --find "message john doe" +``` -Every frame carries evidence pointers back to raw capture rows. Every document declares its blind spots. What the system did not see, it says it did not see. +```json +{ + "steps": [ + {"t": "20:24:09", "op": "focus", "target": "Google Chrome · LinkedIn", "n": 1}, + {"t": "20:24:14", "op": "click", "target": "Search", "role": "TextField", "url": "https://www.linkedin.com/feed/", "n": 2}, + {"t": "20:24:16", "op": "type", "chars": 8, "text": "john doe", "n": 3}, + {"t": "20:24:21", "op": "click", "target": "John Doe", "role": "Link", "url": "https://www.linkedin.com/search/results/people/", "n": 4}, + {"t": "20:24:29", "op": "click", "target": "Message", "role": "Button", "url": "https://www.linkedin.com/in/john-doe/", "n": 5}, + {"t": "20:24:35", "op": "type", "chars": 71, "text": "hey, loved your post on agent memory - open to a quick chat next week?", "n": 6} + ], + "step_count": 6, + "unresolved_clicks": 0 +} +``` -## Beyond memory: routines agents can replay +That's the replay view of a demonstrated run - ordered clicks grounded by element name and role, typed runs, focus changes. An agent repeats the task instead of re-deriving it: fill the slots with new values (a different name, the same steps) and execute. On the happy path it replays at zero model calls; anything unexpected halts and asks instead of guessing. -Episodic memory tells an agent what you did. The bigger result is what it lets an agent *do*. +We measured how much agents overpay to re-derive workflows they've already performed - the **Routine Overhead Ratio** - on weeks of real activity, replicated it on a public web-task dataset, and built a deterministic executor that replays a compiled workflow in a real browser. Instrument, measurements, and executor: [`research/`](research/). -Computer-use agents re-derive every task from scratch - screenshot, reason, act, repeat - even for a routine they've run a hundred times. That re-derivation is where the token cost goes, and it's waste: the routine hasn't changed. +Passively-captured activity becomes **deterministic action** - and the cheapest computer task is the one an agent never reasons through twice. -Because activity-frames compiles recurring activity deterministically, a routine you've done before becomes a **replayable script** - steps an agent executes directly, grounded by the accessibility tree, with no model in the loop. The agent only picks *which* routine and fills in what's new (message a different person, the same way); the replay itself costs essentially zero tokens. +## Measured, not guessed -We measured how much agents overpay to re-derive routines they've already performed - the **Routine Overhead Ratio** - on weeks of real activity, replicated it on a public web-task dataset, and built a deterministic executor that replays a compiled routine in a real browser. Instrument, measurements, and executor: [`research/`](research/). +Agent memory today means conversation memory: what you told the model. What you actually *did* is the missing half - and the hard part is representing it without lying. -Passively-captured activity becomes **deterministic action** - and the cheapest computer task is the one an agent never reasons through twice. +activity-frames enforces a two-tier contract ([SPEC.md](SPEC.md)): + +- **Tier 1, measured (this package):** everything is derivable by deterministic code from capture data - sessions, durations, typed page entities, input volume, coverage gaps. No interpretation, no intent labels. Same input, same output, every time. +- **Tier 2, inferred (optional extension):** tools that add interpretation must namespace it, tag confidence (`high | medium | speculative`), and link evidence. Facts and guesses can never silently mix. + +Every frame carries evidence pointers back to raw capture rows. Every document declares its blind spots. What the system did not see, it says it did not see. ## Use it from an agent (MCP) diff --git a/pyproject.toml b/pyproject.toml index a4ffc60..4081773 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,7 +5,7 @@ build-backend = "hatchling.build" [project] name = "activity-frames" dynamic = ["version"] -description = "Episodic memory for AI agents: compile raw screen capture into structured, deterministic activity frames." +description = "Turn your workday into structured workflows agents can execute. 100% local, served over MCP." readme = "README.md" license = "MIT" requires-python = ">=3.9" From 972b4d917c7f6c290a73138726f201a5f586eb70 Mon Sep 17 00:00:00 2001 From: Nossa Date: Sat, 1 Aug 2026 23:51:53 -0700 Subject: [PATCH 3/3] update readme --- README.md | 54 +++++++++++++++++++++++++++++++++----------------- pyproject.toml | 2 +- 2 files changed, 37 insertions(+), 19 deletions(-) diff --git a/README.md b/README.md index 0c53c92..c39609b 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,7 @@ # activity-frames powering Nocta [![Downloads](https://static.pepy.tech/badge/activity-frames)](https://pepy.tech/projects/activity-frames) +[![GitHub stars](https://img.shields.io/github/stars/nossa-y/activity-frames)](https://github.com/nossa-y/activity-frames/stargazers) [![Paper](https://img.shields.io/badge/paper-PDF-b31b1b)](https://github.com/nossa-y/activity-frames/blob/main/paper/activity-frames-paper.pdf) [![HackerNoon](https://img.shields.io/badge/HackerNoon-top%20story-00E980?logo=hackernoon&logoColor=white)](https://hackernoon.com/i-compiled-55-days-of-screen-activity-into-episodic-memory-for-my-ai-agent) [![Python](https://img.shields.io/pypi/pyversions/activity-frames)](https://pypi.org/project/activity-frames/) @@ -10,13 +11,11 @@ [![PyPI](https://img.shields.io/pypi/v/activity-frames)](https://pypi.org/project/activity-frames/) -> **[Download the desktop app](https://usenocta.app)** - Nocta uses activity-frames to watch how you work and brief you daily on what needs your attention. 100% local. +**Turn your workday into structured workflows agents can execute.** -**Episodic memory for AI agents - and the routines they can replay.** +Computer-use agents work every task out from scratch, even one you've done a hundred times. And between tasks, your agent has no idea what you've been doing all day, so it starts every conversation blind. -Your agent can read your code, search the web, and call APIs - but it has no idea what you've been doing all day, so it starts every conversation blind. And when it runs a task for you, it works it out from scratch every time, even one you've done a hundred times. - -activity-frames fixes both. It records your screen locally and compiles what it sees into structured **activity frames**: bounded, deterministic episodes of tasks you actually did. The recurring ones compile into **routines a computer-use agent can use** instead of working them out again. So it does your repetitive computer tasks **cheaper** (enriching a compiled routine costs almost no tokens) and **more reliable** (the same steps, grounded the same way every time, instead of guessing from a screenshot). +activity-frames fixes both. It records your screen locally and compiles what it sees into structured **activity frames**: bounded, deterministic records of the tasks you actually did. The recurring ones become **workflows an agent can execute** instead of working out again. So your repetitive computer tasks get done **cheaper** (running a compiled workflow costs almost no tokens) and **more reliable** (the same steps, grounded the same way every time, instead of guessing from a screenshot) - and everything else becomes context your agent can use. ```bash pip install activity-frames @@ -28,7 +27,7 @@ aframes context # your last 2 hours, agent-ready Capture stores instants: thousands of snapshot rows a day, each one saying "at 22:53:05, Chrome showed linkedin.com/in/...". Useless to reason over. -activity-frames compiles those instants into episodes: +activity-frames compiles those instants into activity frames: ```yaml - id: f-0007 @@ -59,28 +58,47 @@ away: 18:47-20:24 (97m) Drop that into a prompt and your agent knows your day. A full day compiles in under a second and costs zero tokens. -## Episodic memory, done honestly +## Workflows agents can execute -Agent memory today means conversation memory: what you told the model. Episodic memory is what you actually *did* - and the hard part is representing it without lying. +Computer-use agents re-derive every task from scratch - screenshot, reason, act, repeat - even for a workflow they've run a hundred times. That re-derivation is where the token cost goes, and it's waste: the workflow hasn't changed. -activity-frames enforces a two-tier contract ([SPEC.md](SPEC.md)): +Because activity-frames compiles recurring activity deterministically, a task you've demonstrated becomes an executable script: -- **Tier 1, measured (this package):** everything is derivable by deterministic code from capture data - sessions, durations, typed page entities, input volume, coverage gaps. No interpretation, no intent labels. Same input, same output, every time. -- **Tier 2, inferred (optional extension):** tools that add interpretation must namespace it, tag confidence (`high | medium | speculative`), and link evidence. Facts and guesses can never silently mix. +```bash +aframes steps --find "message john doe" +``` -Every frame carries evidence pointers back to raw capture rows. Every document declares its blind spots. What the system did not see, it says it did not see. +```json +{ + "steps": [ + {"t": "20:24:09", "op": "focus", "target": "Google Chrome · LinkedIn", "n": 1}, + {"t": "20:24:14", "op": "click", "target": "Search", "role": "TextField", "url": "https://www.linkedin.com/feed/", "n": 2}, + {"t": "20:24:16", "op": "type", "chars": 8, "text": "john doe", "n": 3}, + {"t": "20:24:21", "op": "click", "target": "John Doe", "role": "Link", "url": "https://www.linkedin.com/search/results/people/", "n": 4}, + {"t": "20:24:29", "op": "click", "target": "Message", "role": "Button", "url": "https://www.linkedin.com/in/john-doe/", "n": 5}, + {"t": "20:24:35", "op": "type", "chars": 71, "text": "hey, loved your post on agent memory - open to a quick chat next week?", "n": 6} + ], + "step_count": 6, + "unresolved_clicks": 0 +} +``` -## Beyond memory: routines agents can replay +That's the replay view of a demonstrated run - ordered clicks grounded by element name and role, typed runs, focus changes. An agent repeats the task instead of re-deriving it: fill the slots with new values (a different name, the same steps) and execute. On the happy path it replays at zero model calls; anything unexpected halts and asks instead of guessing. -Episodic memory tells an agent what you did. The bigger result is what it lets an agent *do*. +We measured how much agents overpay to re-derive workflows they've already performed - the **Routine Overhead Ratio** - on weeks of real activity, replicated it on a public web-task dataset, and built a deterministic executor that replays a compiled workflow in a real browser. Instrument, measurements, and executor: [`research/`](research/). -Computer-use agents re-derive every task from scratch - screenshot, reason, act, repeat - even for a routine they've run a hundred times. That re-derivation is where the token cost goes, and it's waste: the routine hasn't changed. +Passively-captured activity becomes **deterministic action** - and the cheapest computer task is the one an agent never reasons through twice. -Because activity-frames compiles recurring activity deterministically, a routine you've done before becomes a **replayable script** - steps an agent executes directly, grounded by the accessibility tree, with no model in the loop. The agent only picks *which* routine and fills in what's new (message a different person, the same way); the replay itself costs essentially zero tokens. +## Measured, not guessed -We measured how much agents overpay to re-derive routines they've already performed - the **Routine Overhead Ratio** - on weeks of real activity, replicated it on a public web-task dataset, and built a deterministic executor that replays a compiled routine in a real browser. Instrument, measurements, and executor: [`research/`](research/). +Agent memory today means conversation memory: what you told the model. What you actually *did* is the missing half - and the hard part is representing it without lying. -Passively-captured activity becomes **deterministic action** - and the cheapest computer task is the one an agent never reasons through twice. +activity-frames enforces a two-tier contract ([SPEC.md](SPEC.md)): + +- **Tier 1, measured (this package):** everything is derivable by deterministic code from capture data - sessions, durations, typed page entities, input volume, coverage gaps. No interpretation, no intent labels. Same input, same output, every time. +- **Tier 2, inferred (optional extension):** tools that add interpretation must namespace it, tag confidence (`high | medium | speculative`), and link evidence. Facts and guesses can never silently mix. + +Every frame carries evidence pointers back to raw capture rows. Every document declares its blind spots. What the system did not see, it says it did not see. ## Use it from an agent (MCP) diff --git a/pyproject.toml b/pyproject.toml index a4ffc60..4081773 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,7 +5,7 @@ build-backend = "hatchling.build" [project] name = "activity-frames" dynamic = ["version"] -description = "Episodic memory for AI agents: compile raw screen capture into structured, deterministic activity frames." +description = "Turn your workday into structured workflows agents can execute. 100% local, served over MCP." readme = "README.md" license = "MIT" requires-python = ">=3.9"