diff --git a/crates/neuromesh-cli/src/commands/packet.rs b/crates/neuromesh-cli/src/commands/packet.rs index a74d417..f793557 100644 --- a/crates/neuromesh-cli/src/commands/packet.rs +++ b/crates/neuromesh-cli/src/commands/packet.rs @@ -44,6 +44,10 @@ struct PacketJsonOut { /// Packet files, then the next source files of the whole-question ranking /// (20 deep): where to look, best first. localization: Vec, + /// For a long report: definitions ranked by the report (path, name, + /// lines), 20 deep — function-level localisation. + #[serde(skip_serializing_if = "Vec::is_empty")] + definitions: Vec, /// The whole-question file ranking (BM25F), top 10, for evaluation. ranked_paths: Vec, identifiers: Vec, @@ -130,6 +134,19 @@ pub fn execute(args: &[String]) -> Result<()> { selected_files: files.clone(), selected_paths, localization, + definitions: if prompt.split_whitespace().count() + >= neuromesh_parser::text_normalize::REPORT_WORDS + { + graph + .rank_definitions(&prompt, def_depth()) + .into_iter() + .filter(|d| { + !neuromesh_context::selector::is_noise_path(std::path::Path::new(&d.path)) + }) + .collect() + } else { + Vec::new() + }, ranked_paths: graph .file_rank(&prompt, 10) .into_iter() @@ -397,6 +414,15 @@ fn apply_client_signals(signature: &mut TaskSignature, args: &PacketArgs) { apply_auto_extract_keywords(signature, prompt, enabled); } +/// How many ranked definitions `packet --json` lists for a report +/// (`NM_DEFINITIONS`, default 20; research runs ask for more). +fn def_depth() -> usize { + std::env::var("NM_DEFINITIONS") + .ok() + .and_then(|v| v.parse().ok()) + .unwrap_or(20) +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/neuromesh-graph/src/chunk_rank.rs b/crates/neuromesh-graph/src/chunk_rank.rs index 4445b95..1930533 100644 --- a/crates/neuromesh-graph/src/chunk_rank.rs +++ b/crates/neuromesh-graph/src/chunk_rank.rs @@ -34,10 +34,25 @@ pub(crate) struct ChunkSource { pub path: PathBuf, pub source: String, pub spans: Vec>, + /// The definition each span is (`Class.method`), same order as `spans`. + pub names: Vec, +} + +/// One definition ranked for a report: where it is and how well it matched. +#[derive(Debug, Clone, serde::Serialize)] +pub struct RankedDefinition { + pub path: String, + pub name: String, + pub lines: (usize, usize), + pub score: f32, } /// A file's chunks as term counts, before they are numbered. -type FileChunks = (NodeId, PathBuf, Vec>); +type FileChunks = ( + NodeId, + PathBuf, + Vec<(HashMap, String, usize, usize)>, +); #[derive(Default)] pub(crate) struct ChunkRankIndex { @@ -46,6 +61,8 @@ pub(crate) struct ChunkRankIndex { len: Vec, avg_len: f32, post: HashMap>, + /// Per chunk: definition name (`` for the file head) and lines. + label: Vec<(String, usize, usize)>, } impl ChunkRankIndex { @@ -62,9 +79,10 @@ impl ChunkRankIndex { }; let path_terms = terms_of(&stem_path, &st, true); let lines: Vec<&str> = f.source.lines().collect(); - let head = 1..HEAD_LINES.min(lines.len()); + let head = (1..HEAD_LINES.min(lines.len()), "".to_string()); let mut chunks = Vec::new(); - for span in std::iter::once(head).chain(f.spans.iter().cloned()) { + let named = f.spans.iter().cloned().zip(f.names.iter().cloned()); + for (span, name) in std::iter::once(head).chain(named) { let start = span.start.max(1) - 1; let end = span.end.min(lines.len()).min(start + MAX_CHUNK_LINES); if start >= end { @@ -79,7 +97,7 @@ impl ChunkRankIndex { *c = c.saturating_add(1); } if !counts.is_empty() { - chunks.push(counts); + chunks.push((counts, name, start + 1, end)); } } (f.id, f.path, chunks) @@ -89,9 +107,10 @@ impl ChunkRankIndex { for (id, path, chunks) in per_file { let file = idx.files.len() as u32; idx.files.push((id, path)); - for counts in chunks { + for (counts, name, start, end) in chunks { let chunk = idx.file_of.len() as u32; idx.file_of.push(file); + idx.label.push((name, start, end)); idx.len.push(counts.values().map(|&c| c as u32).sum()); for (t, c) in counts { idx.post.entry(t).or_default().push((chunk, c)); @@ -103,11 +122,71 @@ impl ChunkRankIndex { idx } + /// Definitions by score for `prompt`, best first (file heads left out). + pub(crate) fn rank_definitions(&self, prompt: &str, limit: usize) -> Vec { + let mut scored: Vec<(u32, f32)> = self + .scores(prompt) + .into_iter() + .filter(|(c, _)| self.label[*c as usize].0 != "") + .collect(); + scored.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0))); + scored + .into_iter() + .take(limit) + .map(|(c, score)| { + let (name, start, end) = &self.label[c as usize]; + RankedDefinition { + path: self.files[self.file_of[c as usize] as usize] + .1 + .to_string_lossy() + .replace('\\', "/"), + name: name.clone(), + lines: (*start, *end), + score, + } + }) + .collect() + } + /// Files by their best chunk for `prompt`, best first. pub(crate) fn rank(&self, prompt: &str, limit: usize) -> Vec { + let mut best: HashMap = HashMap::new(); + for (chunk, s) in self.scores(prompt) { + let f = self.file_of[chunk as usize]; + let e = best.entry(f).or_insert(0.0); + if s > *e { + *e = s; + } + } + let mut ranked: Vec = best + .into_iter() + .map(|(f, s)| { + let (id, path) = &self.files[f as usize]; + RankedFile { + id: id.clone(), + path: path.clone(), + score: s, + matched: 0, + } + }) + .collect(); + ranked.sort_by(|a, b| { + b.score + .partial_cmp(&a.score) + .unwrap_or(std::cmp::Ordering::Equal) + .then_with(|| a.path.cmp(&b.path)) + }); + ranked.truncate(limit); + ranked + } + + /// BM25 score of every chunk sharing a term with `prompt` (boilerplate + /// stripped, the title counted [`TITLE_EXTRA`] more times). + fn scores(&self, prompt: &str) -> HashMap { let n = self.len.len() as f32; + let mut score: HashMap = HashMap::new(); if n == 0.0 { - return Vec::new(); + return score; } let st = stemmer(); let body = neuromesh_parser::strip_issue_boilerplate(prompt); @@ -126,7 +205,6 @@ impl ChunkRankIndex { let mut terms: Vec<(String, f32)> = q.into_iter().collect(); terms.sort_by(|a, b| a.0.cmp(&b.0)); terms.truncate(256); - let mut score: HashMap = HashMap::new(); for (t, w) in &terms { let Some(list) = self.post.get(t) else { continue; @@ -139,34 +217,7 @@ impl ChunkRankIndex { *score.entry(chunk).or_insert(0.0) += w * idf * tf * (K1 + 1.0) / (tf + K1 * norm); } } - let mut best: HashMap = HashMap::new(); - for (chunk, s) in score { - let f = self.file_of[chunk as usize]; - let e = best.entry(f).or_insert(0.0); - if s > *e { - *e = s; - } - } - let mut ranked: Vec = best - .into_iter() - .map(|(f, s)| { - let (id, path) = &self.files[f as usize]; - RankedFile { - id: id.clone(), - path: path.clone(), - score: s, - matched: 0, - } - }) - .collect(); - ranked.sort_by(|a, b| { - b.score - .partial_cmp(&a.score) - .unwrap_or(std::cmp::Ordering::Equal) - .then_with(|| a.path.cmp(&b.path)) - }); - ranked.truncate(limit); - ranked + score } } @@ -179,6 +230,7 @@ mod tests { id: NodeId::new(id), path: PathBuf::from(path), source: source.to_string(), + names: spans.iter().map(|s| format!("def_{}", s.start)).collect(), spans, } } diff --git a/crates/neuromesh-graph/src/graph.rs b/crates/neuromesh-graph/src/graph.rs index 5f8bc4d..1a841d6 100644 --- a/crates/neuromesh-graph/src/graph.rs +++ b/crates/neuromesh-graph/src/graph.rs @@ -776,6 +776,21 @@ impl NeuralProjectGraph { self.chunk_rank_index().rank(&prompt[..end], limit) } + /// Definitions (functions, methods, classes) ranked for a long report — + /// the function-level counterpart of [`Self::chunk_rank`]. + pub fn rank_definitions( + &self, + prompt: &str, + limit: usize, + ) -> Vec { + let mut end = prompt.len().min(32 * 1024); + while !prompt.is_char_boundary(end) { + end -= 1; + } + self.chunk_rank_index() + .rank_definitions(&prompt[..end], limit) + } + fn chunk_rank_index(&self) -> Arc { let (key, files) = { let data = self.inner.read(); @@ -793,6 +808,7 @@ impl NeuralProjectGraph { } let mut file_id = None; let mut spans = Vec::new(); + let mut names = Vec::new(); for id in ids { let Some(node) = data.mesh.node(id) else { continue; @@ -802,11 +818,15 @@ impl NeuralProjectGraph { } else if let Some(r) = &node.line_range { if r.end > r.start { spans.push(r.clone()); + names.push(match &node.parent { + Some(p) if !p.is_empty() => format!("{p}.{}", node.name), + _ => node.name.clone(), + }); } } } if let Some(id) = file_id { - files.push((id, path.clone(), spans)); + files.push((id, path.clone(), spans, names)); } } (key, files) @@ -814,13 +834,14 @@ impl NeuralProjectGraph { let t0 = std::time::Instant::now(); let sources: Vec = files .into_par_iter() - .filter_map(|(id, path, spans)| { + .filter_map(|(id, path, spans, names)| { let source = self.read_source(&path)?; Some(crate::chunk_rank::ChunkSource { id, path, source, spans, + names, }) }) .collect(); diff --git a/crates/neuromesh-graph/src/lib.rs b/crates/neuromesh-graph/src/lib.rs index 7d8b03e..2f57dc7 100644 --- a/crates/neuromesh-graph/src/lib.rs +++ b/crates/neuromesh-graph/src/lib.rs @@ -1,5 +1,6 @@ pub mod activation; mod chunk_rank; +pub use chunk_rank::RankedDefinition; pub mod concept_index; pub mod edge; pub mod embeddings; diff --git a/docs/research/contributions-log.md b/docs/research/contributions-log.md index be6c57a..59f96c7 100644 --- a/docs/research/contributions-log.md +++ b/docs/research/contributions-log.md @@ -330,3 +330,34 @@ SWE-bench dev, 50 single-file issues name the gold file in their text, yet only reproduces the prototype exactly (0.271/0.525/0.576/0.695). Fourteen sets unchanged. - **Error-message grep** (`error_grep_prior.py`): fires on 15 of 225 dev issues, fixes 2 (+0.009 @1). Small and positive; not shipped. + +### 8.9 Local LLM stage (D1, Agentless-style) — pilot on a laptop CPU + +Setup: llama.cpp b11172 (official release), Qwen2.5-Coder-3B-Instruct Q4_K_M (official GGUF, +sha256 verified), one call per issue over the engine's top-10 localisation candidates with a +signature skeleton of each (`scripts/research/llm_localize.py`): ~2.2k prompt tokens per issue +(Agentless sends the repository structure). Hardware: Ryzen 5 7530U (6 cores), no discrete GPU; +the Vulkan iGPU path was slower than CPU (18 vs 20 prompt tok/s) → ~105 s per issue. + +Pilot, 15 random dev-fast issues, candidates in engine order: Acc@1/3/5 0.200/0.467/0.600 → +identical. The 3B model returns the list in the order shown (position bias). Next: candidates in +alphabetical order (`--shuffle`). + +Shuffle variant (candidates alphabetical, same 15 issues): Acc@1/3/5 0.200/0.467/0.600 → +**0.067/0.400/0.600**. Without the engine's order the 3B model judges worse than the lexical + +structural ranking; with it, it copies it. **Rejected** for a 3B model on CPU: the LLM stage needs a +stronger judge (an API model, or a 7B fine-tuned one as LocAgent's Qwen2.5-7B(ft) at 0.708 Acc@1), +which this laptop runs at ~5 min per issue. Finding for the paper: the engine's ranking already +beats a small local LLM as a judge, so the LLM budget is better spent on a strong model at one call +per issue over `where_to_look` than on a local small one. + +### 8.10 Function-level output (C14) + +`chunk_rank` already scores definitions; `packet --json` now lists them for a report +(`definitions`: path, `Class.method`, lines; 20 deep, `NM_DEFINITIONS` for research runs). +Function gold = innermost function containing a removed line / insertion point of the reference +patch at the base commit (`scripts/research/func_eval.py`, LocAgent's metric: all edited +functions in the top k). dev-fast (48 issues with function gold): func Acc@1/5/10/20 +0.062/0.229/0.292/0.396 (raw definition ranking, no file prior). Reference (LocAgent Table 4, Lite, +function level Acc@5/@10): BM25 0.318/0.369, CodeRankEmbed 0.518/0.588, Agentless+Claude 0.588, +LocAgent+Claude 0.734/0.774. File-prior orderings (`func_hier.py`) pending the full-dev run. diff --git a/scripts/research/func_eval.py b/scripts/research/func_eval.py new file mode 100644 index 0000000..a26a5d0 --- /dev/null +++ b/scripts/research/func_eval.py @@ -0,0 +1,98 @@ +"""Function-level Acc@k for the engine's `definitions` list (LocAgent's +function-level metric: every edited function in the top k). + +Gold = the innermost function or method (as `Class.method` or `func`) that +contains a removed line, or the insertion point of an added line, of the +reference patch, in the file at `base_commit` (Python `ast`). Instances whose +patch edits no existing function have no function gold and are skipped (that is +how LocAgent arrives at 274 of Lite's 300). + + py -3 scripts/research/func_eval.py --results run.jsonl --data rows.json --repos [--subset ids.json] +""" +import argparse +import ast +import json +import os +import subprocess +import sys + +sys.path.insert(0, os.path.dirname(__file__)) +from locagent_subset import touched_old_lines # noqa: E402 + + +def defs_with_names(src): + out = [] + try: + tree = ast.parse(src) + except (SyntaxError, ValueError): + return out + + def walk(node, stack): + for child in ast.iter_child_nodes(node): + if isinstance(child, ast.ClassDef): + walk(child, stack + [child.name]) + elif isinstance(child, (ast.FunctionDef, ast.AsyncFunctionDef)): + out.append((child.lineno, child.end_lineno, ".".join(stack + [child.name]))) + walk(child, stack + [child.name]) + else: + walk(child, stack) + + walk(tree, []) + return out + + +def gold_functions(clone, commit, patch): + gold = set() + for path, lines in touched_old_lines(patch).items(): + src = subprocess.run(["git", "-C", clone, "show", f"{commit}:{path}"], capture_output=True, + text=True, encoding="utf-8", errors="ignore").stdout + defs = defs_with_names(src) + for line in lines: + inside = [d for d in defs if d[0] <= line <= d[1]] + if inside: + a, b, name = min(inside, key=lambda d: d[1] - d[0]) + gold.add((path, name)) + return gold + + +def same(pred, gold): + (pp, pn), (gp, gn) = pred, gold + if not (pp == gp or pp.endswith("/" + gp) or gp.endswith("/" + pp)): + return False + return pn == gn or pn.split(".")[-1] == gn.split(".")[-1] and ( + "." not in pn or "." not in gn or pn == gn) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--results", required=True) + ap.add_argument("--data", required=True) + ap.add_argument("--repos", required=True) + ap.add_argument("--subset", default="") + args = ap.parse_args() + rows = {r["instance_id"]: r for r in json.load(open(args.data, encoding="utf-8"))} + keep = None + if args.subset: + keep = {r["instance_id"] for r in json.load(open(args.subset, encoding="utf-8"))} + res = [json.loads(l) for l in open(args.results, encoding="utf-8") if l.strip()] + res = [r for r in res if "definitions" in r and (keep is None or r["instance_id"] in keep)] + ks = (1, 5, 10, 20) + hits = {k: 0 for k in ks} + n = 0 + for r in res: + row = rows[r["instance_id"]] + clone = os.path.join(args.repos, row["repo"].split("/")[1]) + gold = gold_functions(clone, row["base_commit"], row["patch"]) + if not gold: + continue + n += 1 + preds = [(d["path"], d["name"]) for d in r["definitions"]] + for k in ks: + top = preds[:k] + hits[k] += all(any(same(p, g) for p in top) for g in gold) + print(f"instances with function gold: {n}") + print(" ".join(f"func Acc@{k} {hits[k] / max(n, 1):.3f}" for k in ks)) + + +if __name__ == "__main__": + main() diff --git a/scripts/research/func_hier.py b/scripts/research/func_hier.py new file mode 100644 index 0000000..b8caf9e --- /dev/null +++ b/scripts/research/func_hier.py @@ -0,0 +1,78 @@ +"""Function-level orderings built from the file list (Agentless localises files +first, then functions inside them). Offline over a harness run made with +NM_DEFINITIONS=300 (definitions + localization stored per instance). + + global definitions as ranked (no file prior) + file-major best files first (loc order), each file's definitions by score + interleave round-robin: best definition of file 1, of file 2, ... then seconds + rrf definition rank fused with its file's rank (k = 60) + + py -3 scripts/research/func_hier.py --results run.jsonl --data rows.json --repos [--files 5] +""" +import argparse +import json +import os +import sys + +sys.path.insert(0, os.path.dirname(__file__)) +from func_eval import gold_functions, same # noqa: E402 + + +def orderings(r, nfiles): + defs = [(d["path"], d["name"]) for d in r["definitions"]] + loc = r["loc_files"][:nfiles] + by_file = {f: [d for d in defs if d[0] == f] for f in loc} + file_major = [d for f in loc for d in by_file[f]] + inter = [] + depth = max((len(v) for v in by_file.values()), default=0) + for i in range(depth): + for f in loc: + if i < len(by_file[f]): + inter.append(by_file[f][i]) + frank = {f: i for i, f in enumerate(r["loc_files"])} + scored = [] + for i, d in enumerate(defs): + s = 1 / (60 + i + 1) + if d[0] in frank: + s += 1 / (60 + frank[d[0]] + 1) + scored.append((s, i, d)) + rrf = [d for _, _, d in sorted(scored, key=lambda x: (-x[0], x[1]))] + return {"global": defs, "file-major": file_major, "interleave": inter, "rrf": rrf} + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--results", required=True) + ap.add_argument("--data", required=True) + ap.add_argument("--repos", required=True) + ap.add_argument("--files", type=int, default=5) + ap.add_argument("--subset", default="") + args = ap.parse_args() + rows = {r["instance_id"]: r for r in json.load(open(args.data, encoding="utf-8"))} + keep = None + if args.subset: + keep = {r["instance_id"] for r in json.load(open(args.subset, encoding="utf-8"))} + res = [json.loads(l) for l in open(args.results, encoding="utf-8") if l.strip()] + res = [r for r in res if r.get("definitions") and r.get("loc_files") + and (keep is None or r["instance_id"] in keep)] + ks = (1, 5, 10) + score = {} + n = 0 + for r in res: + row = rows[r["instance_id"]] + gold = gold_functions(os.path.join(args.repos, row["repo"].split("/")[1]), + row["base_commit"], row["patch"]) + if not gold: + continue + n += 1 + for name, lst in orderings(r, args.files).items(): + s = score.setdefault(name, {k: 0 for k in ks}) + for k in ks: + s[k] += all(any(same(p, g) for p in lst[:k]) for g in gold) + print(f"instances with function gold: {n}") + for name, s in score.items(): + print(f"{name:11s}" + " ".join(f"@{k} {v / max(n, 1):.3f}" for k, v in s.items())) + + +if __name__ == "__main__": + main() diff --git a/scripts/research/llm_localize.py b/scripts/research/llm_localize.py index 46cdb01..93d2b82 100644 --- a/scripts/research/llm_localize.py +++ b/scripts/research/llm_localize.py @@ -102,6 +102,8 @@ def main(): ap.add_argument("--k", type=int, default=5) ap.add_argument("--issue-chars", type=int, default=6000) ap.add_argument("--subset", default="") + ap.add_argument("--shuffle", action="store_true", + help="list candidates alphabetically, hiding the engine's order from the model") args = ap.parse_args() rows = {r["instance_id"]: r for r in json.load(open(args.data, encoding="utf-8"))} keep = None @@ -123,8 +125,9 @@ def main(): checkout(os.path.join(args.repos, name), row["base_commit"], dest) cands = r["loc_files"][: args.top] issue = strip_boilerplate(row["problem_statement"])[: args.issue_chars] + shown = sorted(cands) if args.shuffle else cands prompt = PROMPT.format(issue=issue, k=args.k, - skeletons="\n\n".join(skeleton(dest, c) for c in cands)) + skeletons="\n\n".join(skeleton(dest, c) for c in shown)) t0 = time.time() try: answer, usage = chat(args.endpoint, prompt) diff --git a/scripts/swebench_localize.py b/scripts/swebench_localize.py index 6e959ec..eee1b4a 100644 --- a/scripts/swebench_localize.py +++ b/scripts/swebench_localize.py @@ -162,6 +162,8 @@ def _main(): ours_hit=hit(files, gold), **{f"ours_hit@{k}": hit(files, gold, k) for k in (1, 3, 5)}, ) + if pkt.get("definitions"): + rec["definitions"] = pkt["definitions"] loc = pkt.get("localization") or [] if loc: rec["loc_files"] = loc