Resolve fuzzy org link targets across the full search ladder
Fuzzy link targets previously resolved only to an exact headline title or
a slug id, always with offset 0. Extend resolve_in_doc to follow the
Emacs org-link-search precedence:
1. a <<target>> dedicated target (or <<<radio>>> radio target)
2. an element named with #+NAME: target
3. a headline whose title matches
4. (compat) a section whose slug id matches
5. the target text appearing literally in a section's content
(case-insensitive)
Each branch returns the owning section plus a byte offset relative to the
start of that section's raw content, so a caller can later scroll within
/body. Dedicated-target and plain-text matches use lightweight scans over
the raw section slice (find_org_target, find_plain_text); #+NAME uses the
parsed affiliated keywords. The new section_content helper yields a
section's raw content region and its starting offset.
Fuzzy matching inherits the cross-document fallback already wrapping
resolve_in_doc, so a bare target can reach a <<target>>, named element,
or body text in another file.
This commit is contained in:
parent
27e63f7a6f
commit
c6bb21800a
|
|
@ -1977,23 +1977,81 @@ impl OrkState {
|
|||
return Ok("kind: notfound\n".to_string());
|
||||
}
|
||||
|
||||
// Fuzzy: first try an exact headline-title match (a bare target that
|
||||
// happens to name a heading), then fall back to the slug id. Phase 2
|
||||
// will extend this with <<target>>/#+NAME/plain-text search returning a
|
||||
// byte offset within the owning section.
|
||||
// Fuzzy resolution follows the Emacs `org-link-search` ladder:
|
||||
// 1. a `<<target>>` dedicated target (or `<<<radio>>>`)
|
||||
// 2. an element named with `#+NAME: target`
|
||||
// 3. a headline whose title matches
|
||||
// 4. (compat) a section whose slug id matches
|
||||
// 5. the target text appearing literally in a section's content
|
||||
// Each branch returns the owning section plus a byte offset relative to
|
||||
// the start of that section's raw content (headline end), so a caller
|
||||
// can scroll within `/body`. Headline/slug matches carry offset 0.
|
||||
let want = normalize_search(target);
|
||||
|
||||
// 1. Dedicated target: `<<target>>` or radio `<<<target>>>`.
|
||||
for section in d.doc.all_sections() {
|
||||
let (_content_start, raw) = self.section_content(&d.content, section);
|
||||
if let Some(off) = find_org_target(raw, &want) {
|
||||
return Ok(fuzzy_result(doc_name, &self.section_id(section), off));
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Named element: `#+NAME: target`.
|
||||
for section in d.doc.all_sections() {
|
||||
let (content_start, _raw) = self.section_content(&d.content, section);
|
||||
if let Some(elem) = section.content.iter().find(|e| {
|
||||
e.affiliated.iter().any(|a| {
|
||||
a.key.eq_ignore_ascii_case("NAME")
|
||||
&& matches!(&a.value,
|
||||
org_ast::AffiliatedValue::Text(v) if normalize_search(v) == want)
|
||||
})
|
||||
}) {
|
||||
let off = elem.span.start.offset.saturating_sub(content_start);
|
||||
return Ok(fuzzy_result(doc_name, &self.section_id(section), off));
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Headline title match.
|
||||
if let Some(section) = d.doc.all_sections()
|
||||
.find(|s| normalize_search(&s.headline.title_text()) == want)
|
||||
{
|
||||
return Ok(fuzzy_result(doc_name, &self.section_id(section), 0));
|
||||
}
|
||||
|
||||
// 4. Compat: a section whose slug id matches the bare target.
|
||||
if let Some(section) = self.find_section(&d.doc, target) {
|
||||
return Ok(fuzzy_result(doc_name, &self.section_id(section), 0));
|
||||
}
|
||||
|
||||
// 5. Plain text: the target appears literally in a section's content.
|
||||
for section in d.doc.all_sections() {
|
||||
let (_content_start, raw) = self.section_content(&d.content, section);
|
||||
if let Some(off) = find_plain_text(raw, target) {
|
||||
return Ok(fuzzy_result(doc_name, &self.section_id(section), off));
|
||||
}
|
||||
}
|
||||
|
||||
Ok("kind: notfound\n".to_string())
|
||||
}
|
||||
|
||||
/// Return a section's raw content region: the slice from the end of its
|
||||
/// headline to the start of its first child (or the section end), together
|
||||
/// with that region's starting byte offset in the document. The offset lets
|
||||
/// callers translate an absolute document position into a section-relative
|
||||
/// one.
|
||||
fn section_content<'a>(&self, content: &'a str, section: &Section) -> (usize, &'a str) {
|
||||
let start = section.headline.span.end.offset;
|
||||
let end = section
|
||||
.children
|
||||
.first()
|
||||
.map(|c| c.headline.span.start.offset)
|
||||
.unwrap_or(section.span.end.offset);
|
||||
if start >= end || end > content.len() {
|
||||
return (start, "");
|
||||
}
|
||||
(start, &content[start..end])
|
||||
}
|
||||
|
||||
/// Map a `file:` link's file component to a served document name, if any.
|
||||
/// Matches on the file stem (e.g. `notes.org`, `./notes.org`, `notes`
|
||||
/// all map to the document `notes`).
|
||||
|
|
@ -2046,6 +2104,52 @@ fn normalize_search(s: &str) -> String {
|
|||
s.split_whitespace().collect::<Vec<_>>().join(" ")
|
||||
}
|
||||
|
||||
/// Find a `<<target>>` dedicated target (or `<<<target>>>` radio target) in a
|
||||
/// section's raw content. Matching is whitespace-insensitive on the target
|
||||
/// name (`want` must already be normalized). Returns the byte offset of the
|
||||
/// first `<` relative to the start of `content`.
|
||||
fn find_org_target(content: &str, want: &str) -> Option<usize> {
|
||||
let bytes = content.as_bytes();
|
||||
let mut i = 0;
|
||||
while let Some(rel) = content[i..].find("<<") {
|
||||
let open = i + rel;
|
||||
// Count the full run of leading '<' so `<<<radio>>>` works too.
|
||||
let mut run = 0;
|
||||
while open + run < bytes.len() && bytes[open + run] == b'<' {
|
||||
run += 1;
|
||||
}
|
||||
let inner_start = open + run;
|
||||
if let Some(close_rel) = content[inner_start..].find(">>") {
|
||||
let inner = &content[inner_start..inner_start + close_rel];
|
||||
// A target name is non-empty and contains no newline.
|
||||
if !inner.is_empty() && !inner.contains('\n')
|
||||
&& normalize_search(inner) == want
|
||||
{
|
||||
return Some(open);
|
||||
}
|
||||
// Advance past this candidate's closing delimiter.
|
||||
i = inner_start + close_rel + 2;
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Find the first literal, case-insensitive occurrence of `needle` in a
|
||||
/// section's raw `content`. Returns the byte offset relative to the start of
|
||||
/// `content`. Whitespace is not normalized — this is the lowest-priority
|
||||
/// fuzzy fallback.
|
||||
fn find_plain_text(content: &str, needle: &str) -> Option<usize> {
|
||||
let needle = needle.trim();
|
||||
if needle.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let hay = content.to_lowercase();
|
||||
let want = needle.to_lowercase();
|
||||
hay.find(&want)
|
||||
}
|
||||
|
||||
/// Format a resolution result that names only the owning document (used for a
|
||||
/// bare `file:DOC` link with no in-document target).
|
||||
fn doc_result(doc: &str) -> String {
|
||||
|
|
@ -2438,6 +2542,73 @@ mod tests {
|
|||
assert!(r.contains("doc: a"), "{}", r);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_fuzzy_dedicated_target() {
|
||||
// `<<target>>` resolves to the owning section with a byte offset into
|
||||
// the section's raw content.
|
||||
let (_d, state) = state_with(
|
||||
"#+TITLE: W\n* Alpha\nIntro line.\nHere is a <<my-anchor>> target.\n");
|
||||
let r = state.resolve_target("w", "my-anchor").unwrap();
|
||||
assert!(r.contains("kind: fuzzy"), "{}", r);
|
||||
assert!(r.contains("section: alpha"), "{}", r);
|
||||
// Offset is relative to section content; it points past the headline.
|
||||
assert!(!r.contains("offset: 0\n"), "offset should be non-zero: {}", r);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_fuzzy_radio_target() {
|
||||
// `<<<radio>>>` radio targets resolve like dedicated targets.
|
||||
let (_d, state) = state_with(
|
||||
"#+TITLE: W\n* Alpha\nA <<<radio-tgt>>> here.\n");
|
||||
let r = state.resolve_target("w", "radio-tgt").unwrap();
|
||||
assert!(r.contains("kind: fuzzy"), "{}", r);
|
||||
assert!(r.contains("section: alpha"), "{}", r);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_fuzzy_named_element() {
|
||||
// `#+NAME: foo` makes `[[foo]]` resolve to the named element's section.
|
||||
let (_d, state) = state_with(
|
||||
"#+TITLE: W\n* Alpha\nText.\n#+NAME: fig-one\n| a | b |\n");
|
||||
let r = state.resolve_target("w", "fig-one").unwrap();
|
||||
assert!(r.contains("kind: fuzzy"), "{}", r);
|
||||
assert!(r.contains("section: alpha"), "{}", r);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_fuzzy_plain_text() {
|
||||
// A bare target that is neither a target, name, nor headline resolves
|
||||
// to the first section whose content contains it (case-insensitive).
|
||||
let (_d, state) = state_with(
|
||||
"#+TITLE: W\n* Alpha\nThe quick brown FOX jumps.\n");
|
||||
let r = state.resolve_target("w", "brown fox").unwrap();
|
||||
assert!(r.contains("kind: fuzzy"), "{}", r);
|
||||
assert!(r.contains("section: alpha"), "{}", r);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_fuzzy_target_beats_headline() {
|
||||
// When a `<<name>>` target and a same-named heading both exist, the
|
||||
// dedicated target wins (Emacs precedence).
|
||||
let (_d, state) = state_with(
|
||||
"#+TITLE: W\n* Intro\nSee <<spot>> here.\n* Spot\n");
|
||||
let r = state.resolve_target("w", "spot").unwrap();
|
||||
assert!(r.contains("section: intro"), "target in Intro wins: {}", r);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_fuzzy_cross_document() {
|
||||
// Fuzzy targets fall back across the org path like other link kinds.
|
||||
let (_d, state) = state_with_docs(&[
|
||||
("a", "#+TITLE: A\n* Start\n"),
|
||||
("b", "#+TITLE: B\n* Beta\nContains a <<faraway-target>>.\n"),
|
||||
]);
|
||||
let r = state.resolve_target("a", "faraway-target").unwrap();
|
||||
assert!(r.contains("kind: fuzzy"), "{}", r);
|
||||
assert!(r.contains("doc: b"), "{}", r);
|
||||
assert!(r.contains("section: beta"), "{}", r);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_relevel_section() {
|
||||
let (_d, state) = state_with("#+TITLE: W\n* Parent\n** Child\n* Other\n");
|
||||
|
|
|
|||
Loading…
Reference in New Issue