Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
22 commits
Select commit Hold shift + click to select a range
155ee56
20260323/Implementation 1 - Lock Step
duck-lint Mar 23, 2026
cad3039
Initial plan
Copilot Mar 23, 2026
9ff0344
Initial plan
Copilot Mar 23, 2026
3ebd0f0
Initial plan
Copilot Mar 23, 2026
9b86acd
Persist grounded retrieval diagnostics
Copilot Mar 23, 2026
c34de89
fix: clarify duplicate document identity failures
Copilot Mar 23, 2026
b259f1a
fix: align metadata projection and rerank contract
Copilot Mar 23, 2026
1f74637
Align grounded retrieval test with runtime path
Copilot Mar 23, 2026
16e6975
Update agent/corpus_db.py
duck-lint Mar 23, 2026
de5cb45
Refine grounded retrieval capture test
Copilot Mar 23, 2026
9863e73
fix: preserve uuid identity in chunk keys
Copilot Mar 23, 2026
f23dd17
Merge pull request #35 from duck-lint/copilot/sim-2026-03-23-fix-dupl…
duck-lint Mar 23, 2026
1804c76
Merge pull request #36 from duck-lint/copilot/sim-2026-03-23-fix-002-…
duck-lint Mar 23, 2026
2ef0cc7
Merge pull request #37 from duck-lint/copilot/sim-2026-03-23-fix-003-…
duck-lint Mar 23, 2026
27b23a5
Initial plan
Copilot Mar 23, 2026
a11633d
Update configs/default.yaml
duck-lint Mar 23, 2026
9aa6878
Update configs/default.yaml
duck-lint Mar 23, 2026
f24d369
Update agent/corpus.py
duck-lint Mar 23, 2026
4c661ce
fix: add LIMIT to _query_chunk_search_fallback to bound result sets
Copilot Mar 23, 2026
a2a3bf6
Update agent/corpus_db.py
duck-lint Mar 23, 2026
0c8fc41
test: add regression tests for _query_chunk_search_fallback LIMIT enf…
Copilot Mar 23, 2026
ce09819
Merge pull request #39 from duck-lint/copilot/sub-pr-38
duck-lint Mar 23, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions OPERATOR_QUICKREF.md
Original file line number Diff line number Diff line change
Expand Up @@ -158,6 +158,10 @@ Check these fields first:
- rebuild corpus and embeddings, then rerun `doctor --require-grounding`
- `DOCTOR_MEMORY_DANGLING_EVIDENCE`
- reset or repair durable memory entries that cite removed chunk keys
- `DUPLICATE_DOCUMENT_IDENTITY`
- `python -m agent index --json` found two configured sources that resolve to the same global document identity
- if the notes are missing UUIDs, add explicit `uuid` frontmatter to one or both notes, or rename one note so the relative path changes
- if the notes already have UUIDs, assign distinct UUIDs unless both files intentionally represent the same document

## Security Checks

Expand Down
7 changes: 7 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,13 @@ The corpus ingester ports the vault-aware logic that matters for retrieval quali
- wikilink extraction
- date extraction from paths and metadata

Document identity is intentionally strict and global across configured sources:

- frontmatter `uuid` is the authoritative document identity when present
- notes without `uuid` fall back to a stable `doc_key` derived from the note's relative path
- fallback `doc_key` values are still globally unique, so two configured sources cannot contain the same relative-path note without UUIDs
- if that happens, corpus ingest fails fast with a duplicate document identity error; the supported fixes are to add UUIDs or rename one of the notes

The only supported chunking and embedding-preprocess profile is `obsidian_v1`.

## Storage
Expand Down
1 change: 1 addition & 0 deletions agent/app_types.py
Original file line number Diff line number Diff line change
Expand Up @@ -152,6 +152,7 @@ class DocumentRecord:
class ChunkRecord:
chunk_key: str
doc_key: str
chunk_kind: str
chunk_index: int
section_index: int
heading_path: str
Expand Down
115 changes: 114 additions & 1 deletion agent/chunking.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,16 @@
WIKILINK_RE = re.compile(r"\[\[([^\]|]+)(\|([^\]]+))?\]\]")
HEADING_RE = re.compile(r"^\s{0,3}(#{1,6})\s+(.*)$")

CHUNK_KIND_CONTENT = "content"
CHUNK_KIND_METADATA = "metadata"
METADATA_HEADING_PATH = "META: frontmatter"
METADATA_CHUNK_ANCHOR = "frontmatter"
METADATA_CHUNK_TITLE = "frontmatter"
METADATA_CHUNK_INDEX = -1
METADATA_SECTION_INDEX = -1
METADATA_PROJECTION_VERSION = "metadata_v2"
LEXICAL_PROJECTION_VERSION = "lexical_v2"


@dataclass(frozen=True)
class ChunkDraft:
Expand All @@ -35,6 +45,18 @@ class Section:
start_char: int


@dataclass(frozen=True)
class MetadataProjection:
note_type: str
aliases: list[str]
tags: list[str]
journal_entry_date: Optional[str]
canonical_name: str
layer: str
register: str
text: str


def sha256_text(value: str) -> str:
return hashlib.sha256(value.encode("utf-8", errors="replace")).hexdigest()

Expand Down Expand Up @@ -76,13 +98,15 @@ def canonicalize_heading_path(heading_path: Any) -> list[str]:
def stable_chunk_key(
*,
source_uri: str,
chunk_kind: str,
heading_path: list[str],
section_index: int,
chunk_index: int,
) -> str:
canonical_source = canonicalize_source_uri(source_uri)
canonical_heading = " > ".join(canonicalize_heading_path(heading_path))
return sha256_text(f"{canonical_source}|{canonical_heading}|{section_index}|{chunk_index}")[:32]
kind = str(chunk_kind or CHUNK_KIND_CONTENT).strip().lower() or CHUNK_KIND_CONTENT
return sha256_text(f"{canonical_source}|{kind}|{canonical_heading}|{section_index}|{chunk_index}")[:32]


def split_frontmatter(text: str) -> tuple[str, str]:
Expand Down Expand Up @@ -146,6 +170,95 @@ def parse_source_date(meta: dict[str, Any], filename: str) -> Optional[str]:
return None


def parse_string_list_field(meta: dict[str, Any], key: str) -> list[str]:
raw = meta.get(key)
if raw is None:
return []
if isinstance(raw, list):
items = raw
else:
items = [raw]
seen: set[str] = set()
out: list[str] = []
for item in items:
text = str(item or "").strip()
if not text:
continue
if text in seen:
continue
seen.add(text)
out.append(text)
return out


def parse_string_field(meta: dict[str, Any], key: str) -> str:
raw = meta.get(key)
if raw is None:
return ""
return str(raw).strip()


def normalize_doc_type(
meta: dict[str, Any],
*,
folder: str,
entry_date: Optional[str],
) -> str:
explicit = str(meta.get("doc_type") or "").strip().lower()
if explicit:
return explicit
note_type = str(meta.get("note_type") or "").strip().lower()
if note_type:
if note_type in {"journal", "journal_entry", "journal-entry"}:
return "journal"
return note_type
note_status = str(meta.get("note_status") or "").strip().lower()
if note_status in {"journal", "journal_entry", "journal-entry"}:
return "journal"
if entry_date:
return "journal"
return folder.lower() if folder else "note"


def build_metadata_projection(
*,
meta: dict[str, Any],
document_title: str,
entry_date: Optional[str],
) -> MetadataProjection:
note_type = parse_string_field(meta, "note_type")
aliases = parse_string_list_field(meta, "aliases")
tags = parse_string_list_field(meta, "tags")
canonical_name = parse_string_field(meta, "canonical_name") or document_title
layer = parse_string_field(meta, "layer")
register = parse_string_field(meta, "register")
lines: list[str] = []
if note_type:
lines.append(f"note_type: {note_type}")
if aliases:
lines.append(f"aliases: {', '.join(aliases)}")
if tags:
lines.append(f"tags: {', '.join(tags)}")
if entry_date:
lines.append(f"journal_entry_date: {entry_date}")
if canonical_name:
lines.append(f"canonical_name: {canonical_name}")
if layer:
lines.append(f"layer: {layer}")
if register:
lines.append(f"register: {register}")
return MetadataProjection(
note_type=note_type,
aliases=aliases,
tags=tags,
journal_entry_date=entry_date,
canonical_name=canonical_name,
layer=layer,
register=register,
text="\n".join(lines).strip() + "\n",
)


def normalize_markdown_light(markdown: str) -> str:
lines = markdown.splitlines()
out: list[str] = []
Expand Down
Loading
Loading