Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion docs/MCP-ARCHITECTURE.md
Original file line number Diff line number Diff line change
Expand Up @@ -127,7 +127,7 @@ The CLI (`rosetta-cli`, published on PyPI) publishes instructions from the instr

**Critical rule:** Always publish the entire `/instructions` folder. Never subfolders or single files (breaks tag extraction).

**Change detection:** MD5 hash of content. Only modified files publish (~77% time savings). Use `--force` to bypass.
**Change detection:** MD5 hash of content and metadata. Text-suffix files use the text path only when strict UTF-8 decoding succeeds; other content contributes a SHA-256 digest of its raw bytes, so equal-size replacements are detected without lossy decoding. Existing length-based or lossy binary hashes cause a one-time republish; valid UTF-8 text hashes are unchanged. Use `--force` to bypass.

**Auto-tagging and metadata extraction.** The CLI reads each file during publishing and extracts everything MCP needs to serve it efficiently:
- **Tags:** all folder names + filename + composite pairs/triples (`core/skills`, `r3/core/skills`, etc.). These are what the typed load aliases query against.
Expand Down
2 changes: 1 addition & 1 deletion docs/web/docs/mcp-architecture.md
Original file line number Diff line number Diff line change
Expand Up @@ -132,7 +132,7 @@ The CLI (`rosetta-cli`, published on PyPI) publishes instructions from the instr

**Critical rule:** Always publish the entire `/instructions` folder. Never subfolders or single files (breaks tag extraction).

**Change detection:** MD5 hash of content. Only modified files publish (~77% time savings). Use `--force` to bypass.
**Change detection:** MD5 hash of content and metadata. Text-suffix files use the text path only when strict UTF-8 decoding succeeds; other content contributes a SHA-256 digest of its raw bytes, so equal-size replacements are detected without lossy decoding. Existing length-based or lossy binary hashes cause a one-time republish; valid UTF-8 text hashes are unchanged. Use `--force` to bypass.

**Auto-tagging and metadata extraction.** The CLI reads each file during publishing and extracts everything MCP needs to serve it efficiently:
- **Tags:** all folder names + filename + composite pairs/triples (`core/skills`, `r3/core/skills`, etc.). These are what the typed load aliases query against.
Expand Down
2 changes: 0 additions & 2 deletions src/rosetta-cli/rosetta_cli/rosetta_publisher.py
Original file line number Diff line number Diff line change
Expand Up @@ -178,7 +178,6 @@ def publish_folder(
cache = DocumentData.from_file(
file,
self.workspace_root,
self.file_extensions,
publish_root=folder,
)
all_caches.append(cache)
Expand Down Expand Up @@ -350,7 +349,6 @@ def publish_file(
cache = DocumentData.from_file(
file,
self.workspace_root,
self.file_extensions,
publish_root=file.parent,
)

Expand Down
4 changes: 2 additions & 2 deletions src/rosetta-cli/rosetta_cli/services/document_data.py
Original file line number Diff line number Diff line change
Expand Up @@ -86,7 +86,7 @@ def from_file(
try:
content_str = content.decode("utf-8")
except UnicodeDecodeError:
content_str = content.decode("utf-8", errors="ignore")
is_text = False

# Count lines platform-independently: \r\n, \n\r, \r, \n all count as separators
line_count = None
Expand All @@ -111,7 +111,7 @@ def from_file(
doc_title = cls._compute_doc_title(parsed_path, file_path.name)

content_hash = cls._calculate_hash(
content_str if content_str is not None else str(len(content)),
content_str if content_str is not None else hashlib.sha256(content).hexdigest(),
tags,
domain,
release,
Expand Down
69 changes: 69 additions & 0 deletions src/rosetta-cli/tests/test_document_data.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
from pathlib import Path

import pytest

from rosetta_cli.services.document_data import DocumentData


Expand Down Expand Up @@ -107,3 +109,70 @@ def test_hash_changes_when_doc_name_changes():
)

assert hash_a != hash_b


@pytest.mark.parametrize("extension", [".bin", ".pdf"])
def test_binary_hash_changes_for_same_size_replacement(tmp_path: Path, extension: str):
path = tmp_path / f"asset{extension}"
path.write_bytes(b"\x00\xff\x01\x02")
original = DocumentData.from_file(path, workspace_root=tmp_path)
path.write_bytes(b"\x00\xfe\x01\x02")
replacement = DocumentData.from_file(path, workspace_root=tmp_path)

assert not original.is_text and not replacement.is_text
assert original.content_str is None and replacement.content_str is None
assert len(original.content) == len(replacement.content)
assert original.ims_doc_id == replacement.ims_doc_id
assert original.tags == replacement.tags
assert original.content_hash != replacement.content_hash


@pytest.mark.parametrize("content", [b"", b"\x00\xff\x01", b"text in a binary extension"])
def test_binary_hash_is_stable_for_identical_bytes(tmp_path: Path, content: bytes):
path = tmp_path / "asset.bin"
path.write_bytes(content)
original = DocumentData.from_file(path, workspace_root=tmp_path)
path.write_bytes(content)
unchanged = DocumentData.from_file(path, workspace_root=tmp_path)

assert original.content == unchanged.content == content
assert original.content_hash == unchanged.content_hash


def test_text_hash_preserves_existing_format(tmp_path: Path):
path = tmp_path / "example.md"
content = "# Example\r\n\nUnicode: café\n"
path.write_bytes(content.encode("utf-8"))
data = DocumentData.from_file(path, workspace_root=tmp_path)

assert data.is_text
assert data.content_hash == DocumentData._calculate_hash(
content, data.tags, data.domain, data.release, data.doc_title,
data.doc_title, data.sort_order, data.original_path, data.resource_path,
)


def test_invalid_utf8_text_suffix_uses_binary_content_state(tmp_path: Path):
path = tmp_path / "example.md"
path.write_bytes(b"\xff---\ntags: [ignored]\n---\nbody")
original = DocumentData.from_file(path, workspace_root=tmp_path)
path.write_bytes(b"\xfe---\ntags: [ignored]\n---\nbody")
replacement = DocumentData.from_file(path, workspace_root=tmp_path)

assert not original.is_text and not replacement.is_text
assert original.content_str is None and replacement.content_str is None
assert original.line_count is None and replacement.line_count is None
assert original.frontmatter is None and replacement.frontmatter is None
assert "ignored" not in original.tags and "ignored" not in replacement.tags
assert original.content_hash != replacement.content_hash


def test_binary_hash_still_includes_metadata(tmp_path: Path):
path = tmp_path / "asset.bin"
path.write_bytes(b"\x00\xff")
original = DocumentData.from_file(path, workspace_root=tmp_path)
renamed_path = path.rename(tmp_path / "renamed.bin")
renamed = DocumentData.from_file(renamed_path, workspace_root=tmp_path)

assert original.content == renamed.content
assert original.content_hash != renamed.content_hash
128 changes: 128 additions & 0 deletions src/rosetta-cli/tests/test_publish_binary_changes.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,128 @@
from dataclasses import asdict
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import Mock

import pytest

from rosetta_cli.ragflow_client import RAGFlowClient
from rosetta_cli.rosetta_publisher import ContentPublisher
from rosetta_cli.services.document_data import DocumentData


@pytest.mark.parametrize("publish_folder", [False, True], ids=["file", "folder"])
@pytest.mark.parametrize("legacy_hash", [False, True], ids=["current", "legacy"])
def test_binary_publish_updates_changed_content_and_then_skips(
tmp_path: Path, publish_folder: bool, legacy_hash: bool
):
instructions = tmp_path / "instructions"
path = instructions / "r3" / "core" / "assets" / "sample.bin"
path.parent.mkdir(parents=True)
original_bytes = b"\x00\xff\x01\x02"
replacement_bytes = b"\x00\xfe\x01\x02"
path.write_bytes(original_bytes)
original = DocumentData.from_file(path, workspace_root=tmp_path)
existing = SimpleNamespace(
id="stored-document", name=original.doc_title,
meta_fields=original.to_metadata_dict(),
)
if legacy_hash:
existing.meta_fields["content_hash"] = DocumentData._calculate_hash(
str(len(original_bytes)), original.tags, original.domain, original.release,
original.doc_title, original.doc_title, original.sort_order,
original.original_path, original.resource_path,
)

# Keep the real publisher/change detector; replace only the remote client.
client = Mock(spec=RAGFlowClient)
client.page_size = 1000
client.get_dataset.return_value = SimpleNamespace(id="dataset")
client.get_existing_doc.return_value = existing
client.list_documents.return_value = [existing]

def upload_document(**kwargs):
existing.meta_fields = asdict(kwargs["metadata"])
return existing, "dataset"

client.upload_document.side_effect = upload_document
publisher = ContentPublisher(client, str(tmp_path))

def publish():
if publish_folder:
[result] = publisher.publish_folder(str(instructions), parse_documents=False)
return result
return publisher.publish_file(str(path), parse_documents=False)

first = publish()
assert first.success
assert first.skipped is not legacy_hash
assert client.upload_document.call_count == int(legacy_hash)
client.upload_document.reset_mock()

path.write_bytes(replacement_bytes)
changed = publish()
assert changed.success and not changed.skipped
client.upload_document.assert_called_once()
uploaded = client.upload_document.call_args.kwargs
assert uploaded["content"] == replacement_bytes
assert uploaded["metadata"].ims_doc_id == original.ims_doc_id
assert uploaded["metadata"].tags == original.tags
assert uploaded["metadata"].content_hash != original.content_hash

client.upload_document.reset_mock()
unchanged = publish()
assert unchanged.success and unchanged.skipped
client.upload_document.assert_not_called()


def test_configured_pdf_filter_keeps_binary_change_detection(tmp_path: Path):
instructions = tmp_path / "instructions"
path = instructions / "r3" / "core" / "assets" / "sample.pdf"
path.parent.mkdir(parents=True)
original_bytes = b"\xffPDF"
replacement_bytes = b"\xfePDF"
path.write_bytes(original_bytes)
(path.parent / "excluded.bin").write_bytes(b"ignored")

legacy = DocumentData.from_file(path, workspace_root=tmp_path)
# Reproduce metadata written before strict decoding and filter separation.
legacy_hash = DocumentData._calculate_hash(
original_bytes.decode("utf-8", errors="ignore"), legacy.tags,
legacy.domain, legacy.release, legacy.doc_title, legacy.doc_title,
legacy.sort_order, legacy.original_path, legacy.resource_path,
)
assert legacy_hash != legacy.content_hash
existing = SimpleNamespace(
id="stored-document", name=legacy.doc_title,
meta_fields=legacy.to_metadata_dict(),
)
existing.meta_fields["content_hash"] = legacy_hash
client = Mock(spec=RAGFlowClient)
client.page_size = 1000
client.get_dataset.return_value = SimpleNamespace(id="dataset")
client.get_existing_doc.return_value = existing
client.list_documents.return_value = [existing]

def upload_document(**kwargs):
existing.meta_fields = asdict(kwargs["metadata"])
return existing, "dataset"

client.upload_document.side_effect = upload_document
publisher = ContentPublisher(client, str(tmp_path), file_extensions=[".pdf"])

[migrated] = publisher.publish_folder(str(instructions), parse_documents=False)
assert migrated.success and not migrated.skipped
assert Path(migrated.file_path) == path
assert client.upload_document.call_args.kwargs["content"] == original_bytes
assert client.upload_document.call_args.kwargs["metadata"].line_count is None
client.upload_document.reset_mock()

path.write_bytes(replacement_bytes)
[changed] = publisher.publish_folder(str(instructions), parse_documents=False)
assert changed.success and not changed.skipped
assert client.upload_document.call_args.kwargs["content"] == replacement_bytes
client.upload_document.reset_mock()

[unchanged] = publisher.publish_folder(str(instructions), parse_documents=False)
assert unchanged.success and unchanged.skipped
client.upload_document.assert_not_called()