Resolve fuzzy org link targets across the full search ladder

Fuzzy link targets previously resolved only to an exact headline title or
a slug id, always with offset 0. Extend resolve_in_doc to follow the
Emacs org-link-search precedence:

  1. a <<target>> dedicated target (or <<<radio>>> radio target)
  2. an element named with #+NAME: target
  3. a headline whose title matches
  4. (compat) a section whose slug id matches
  5. the target text appearing literally in a section's content
     (case-insensitive)

Each branch returns the owning section plus a byte offset relative to the
start of that section's raw content, so a caller can later scroll within
/body. Dedicated-target and plain-text matches use lightweight scans over
the raw section slice (find_org_target, find_plain_text); #+NAME uses the
parsed affiliated keywords. The new section_content helper yields a
section's raw content region and its starting offset.

Fuzzy matching inherits the cross-document fallback already wrapping
resolve_in_doc, so a bare target can reach a <<target>>, named element,
or body text in another file.
This commit is contained in:
Levi Neely 2026-10-02 09:34:33 +02:00
parent 27e63f7a6f
commit c6bb21800a
1 changed files with 175 additions and 4 deletions

View File

@ -1977,23 +1977,81 @@ impl OrkState {
return Ok("kind: notfound\n".to_string()); return Ok("kind: notfound\n".to_string());
} }
// Fuzzy: first try an exact headline-title match (a bare target that // Fuzzy resolution follows the Emacs `org-link-search` ladder:
// happens to name a heading), then fall back to the slug id. Phase 2 // 1. a `<<target>>` dedicated target (or `<<<radio>>>`)
// will extend this with <<target>>/#+NAME/plain-text search returning a // 2. an element named with `#+NAME: target`
// byte offset within the owning section. // 3. a headline whose title matches
// 4. (compat) a section whose slug id matches
// 5. the target text appearing literally in a section's content
// Each branch returns the owning section plus a byte offset relative to
// the start of that section's raw content (headline end), so a caller
// can scroll within `/body`. Headline/slug matches carry offset 0.
let want = normalize_search(target); let want = normalize_search(target);
// 1. Dedicated target: `<<target>>` or radio `<<<target>>>`.
for section in d.doc.all_sections() {
let (_content_start, raw) = self.section_content(&d.content, section);
if let Some(off) = find_org_target(raw, &want) {
return Ok(fuzzy_result(doc_name, &self.section_id(section), off));
}
}
// 2. Named element: `#+NAME: target`.
for section in d.doc.all_sections() {
let (content_start, _raw) = self.section_content(&d.content, section);
if let Some(elem) = section.content.iter().find(|e| {
e.affiliated.iter().any(|a| {
a.key.eq_ignore_ascii_case("NAME")
&& matches!(&a.value,
org_ast::AffiliatedValue::Text(v) if normalize_search(v) == want)
})
}) {
let off = elem.span.start.offset.saturating_sub(content_start);
return Ok(fuzzy_result(doc_name, &self.section_id(section), off));
}
}
// 3. Headline title match.
if let Some(section) = d.doc.all_sections() if let Some(section) = d.doc.all_sections()
.find(|s| normalize_search(&s.headline.title_text()) == want) .find(|s| normalize_search(&s.headline.title_text()) == want)
{ {
return Ok(fuzzy_result(doc_name, &self.section_id(section), 0)); return Ok(fuzzy_result(doc_name, &self.section_id(section), 0));
} }
// 4. Compat: a section whose slug id matches the bare target.
if let Some(section) = self.find_section(&d.doc, target) { if let Some(section) = self.find_section(&d.doc, target) {
return Ok(fuzzy_result(doc_name, &self.section_id(section), 0)); return Ok(fuzzy_result(doc_name, &self.section_id(section), 0));
} }
// 5. Plain text: the target appears literally in a section's content.
for section in d.doc.all_sections() {
let (_content_start, raw) = self.section_content(&d.content, section);
if let Some(off) = find_plain_text(raw, target) {
return Ok(fuzzy_result(doc_name, &self.section_id(section), off));
}
}
Ok("kind: notfound\n".to_string()) Ok("kind: notfound\n".to_string())
} }
/// Return a section's raw content region: the slice from the end of its
/// headline to the start of its first child (or the section end), together
/// with that region's starting byte offset in the document. The offset lets
/// callers translate an absolute document position into a section-relative
/// one.
fn section_content<'a>(&self, content: &'a str, section: &Section) -> (usize, &'a str) {
let start = section.headline.span.end.offset;
let end = section
.children
.first()
.map(|c| c.headline.span.start.offset)
.unwrap_or(section.span.end.offset);
if start >= end || end > content.len() {
return (start, "");
}
(start, &content[start..end])
}
/// Map a `file:` link's file component to a served document name, if any. /// Map a `file:` link's file component to a served document name, if any.
/// Matches on the file stem (e.g. `notes.org`, `./notes.org`, `notes` /// Matches on the file stem (e.g. `notes.org`, `./notes.org`, `notes`
/// all map to the document `notes`). /// all map to the document `notes`).
@ -2046,6 +2104,52 @@ fn normalize_search(s: &str) -> String {
s.split_whitespace().collect::<Vec<_>>().join(" ") s.split_whitespace().collect::<Vec<_>>().join(" ")
} }
/// Find a `<<target>>` dedicated target (or `<<<target>>>` radio target) in a
/// section's raw content. Matching is whitespace-insensitive on the target
/// name (`want` must already be normalized). Returns the byte offset of the
/// first `<` relative to the start of `content`.
fn find_org_target(content: &str, want: &str) -> Option<usize> {
let bytes = content.as_bytes();
let mut i = 0;
while let Some(rel) = content[i..].find("<<") {
let open = i + rel;
// Count the full run of leading '<' so `<<<radio>>>` works too.
let mut run = 0;
while open + run < bytes.len() && bytes[open + run] == b'<' {
run += 1;
}
let inner_start = open + run;
if let Some(close_rel) = content[inner_start..].find(">>") {
let inner = &content[inner_start..inner_start + close_rel];
// A target name is non-empty and contains no newline.
if !inner.is_empty() && !inner.contains('\n')
&& normalize_search(inner) == want
{
return Some(open);
}
// Advance past this candidate's closing delimiter.
i = inner_start + close_rel + 2;
} else {
break;
}
}
None
}
/// Find the first literal, case-insensitive occurrence of `needle` in a
/// section's raw `content`. Returns the byte offset relative to the start of
/// `content`. Whitespace is not normalized — this is the lowest-priority
/// fuzzy fallback.
fn find_plain_text(content: &str, needle: &str) -> Option<usize> {
let needle = needle.trim();
if needle.is_empty() {
return None;
}
let hay = content.to_lowercase();
let want = needle.to_lowercase();
hay.find(&want)
}
/// Format a resolution result that names only the owning document (used for a /// Format a resolution result that names only the owning document (used for a
/// bare `file:DOC` link with no in-document target). /// bare `file:DOC` link with no in-document target).
fn doc_result(doc: &str) -> String { fn doc_result(doc: &str) -> String {
@ -2438,6 +2542,73 @@ mod tests {
assert!(r.contains("doc: a"), "{}", r); assert!(r.contains("doc: a"), "{}", r);
} }
#[test]
fn test_resolve_fuzzy_dedicated_target() {
// `<<target>>` resolves to the owning section with a byte offset into
// the section's raw content.
let (_d, state) = state_with(
"#+TITLE: W\n* Alpha\nIntro line.\nHere is a <<my-anchor>> target.\n");
let r = state.resolve_target("w", "my-anchor").unwrap();
assert!(r.contains("kind: fuzzy"), "{}", r);
assert!(r.contains("section: alpha"), "{}", r);
// Offset is relative to section content; it points past the headline.
assert!(!r.contains("offset: 0\n"), "offset should be non-zero: {}", r);
}
#[test]
fn test_resolve_fuzzy_radio_target() {
// `<<<radio>>>` radio targets resolve like dedicated targets.
let (_d, state) = state_with(
"#+TITLE: W\n* Alpha\nA <<<radio-tgt>>> here.\n");
let r = state.resolve_target("w", "radio-tgt").unwrap();
assert!(r.contains("kind: fuzzy"), "{}", r);
assert!(r.contains("section: alpha"), "{}", r);
}
#[test]
fn test_resolve_fuzzy_named_element() {
// `#+NAME: foo` makes `[[foo]]` resolve to the named element's section.
let (_d, state) = state_with(
"#+TITLE: W\n* Alpha\nText.\n#+NAME: fig-one\n| a | b |\n");
let r = state.resolve_target("w", "fig-one").unwrap();
assert!(r.contains("kind: fuzzy"), "{}", r);
assert!(r.contains("section: alpha"), "{}", r);
}
#[test]
fn test_resolve_fuzzy_plain_text() {
// A bare target that is neither a target, name, nor headline resolves
// to the first section whose content contains it (case-insensitive).
let (_d, state) = state_with(
"#+TITLE: W\n* Alpha\nThe quick brown FOX jumps.\n");
let r = state.resolve_target("w", "brown fox").unwrap();
assert!(r.contains("kind: fuzzy"), "{}", r);
assert!(r.contains("section: alpha"), "{}", r);
}
#[test]
fn test_resolve_fuzzy_target_beats_headline() {
// When a `<<name>>` target and a same-named heading both exist, the
// dedicated target wins (Emacs precedence).
let (_d, state) = state_with(
"#+TITLE: W\n* Intro\nSee <<spot>> here.\n* Spot\n");
let r = state.resolve_target("w", "spot").unwrap();
assert!(r.contains("section: intro"), "target in Intro wins: {}", r);
}
#[test]
fn test_resolve_fuzzy_cross_document() {
// Fuzzy targets fall back across the org path like other link kinds.
let (_d, state) = state_with_docs(&[
("a", "#+TITLE: A\n* Start\n"),
("b", "#+TITLE: B\n* Beta\nContains a <<faraway-target>>.\n"),
]);
let r = state.resolve_target("a", "faraway-target").unwrap();
assert!(r.contains("kind: fuzzy"), "{}", r);
assert!(r.contains("doc: b"), "{}", r);
assert!(r.contains("section: beta"), "{}", r);
}
#[test] #[test]
fn test_relevel_section() { fn test_relevel_section() {
let (_d, state) = state_with("#+TITLE: W\n* Parent\n** Child\n* Other\n"); let (_d, state) = state_with("#+TITLE: W\n* Parent\n** Child\n* Other\n");