Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 7 additions & 13 deletions crates/neuromesh-cli/src/commands/packet.rs
Original file line number Diff line number Diff line change
Expand Up @@ -104,6 +104,12 @@ pub fn execute(args: &[String]) -> Result<()> {

let selected_paths = neuromesh_context::gold::packet_file_order(&graph, &prompt, &view);
let localization = neuromesh_context::gold::localization_order(&graph, &prompt, &view, 20);
let definitions =
if prompt.split_whitespace().count() >= neuromesh_parser::text_normalize::REPORT_WORDS {
neuromesh_context::gold::definition_order(&graph, &prompt, &localization, def_depth())
} else {
Vec::new()
};
let mut files: Vec<String> = packet_file_names(&view).into_iter().collect();
files.sort();
let reduction = if workspace_tokens > 0 {
Expand Down Expand Up @@ -134,19 +140,7 @@ pub fn execute(args: &[String]) -> Result<()> {
selected_files: files.clone(),
selected_paths,
localization,
definitions: if prompt.split_whitespace().count()
>= neuromesh_parser::text_normalize::REPORT_WORDS
{
graph
.rank_definitions(&prompt, def_depth())
.into_iter()
.filter(|d| {
!neuromesh_context::selector::is_noise_path(std::path::Path::new(&d.path))
})
.collect()
} else {
Vec::new()
},
definitions,
ranked_paths: graph
.file_rank(&prompt, 10)
.into_iter()
Expand Down
34 changes: 34 additions & 0 deletions crates/neuromesh-context/src/gold.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1135,6 +1135,40 @@ pub fn localization_order(
out
}

/// Functions and methods to look at for a report, best first, `k` deep: the
/// definition ranking fused (RRF, k = 60) with the rank of each definition's
/// file in [`localization_order`] — the function step Agentless takes inside
/// the files it chose. SWE-bench dev (190 issues with function gold), func
/// Acc@5/@10: 0.311/0.374 without the file prior → 0.321/0.400 with it.
pub fn definition_order(
graph: &neuromesh_graph::NeuralProjectGraph,
prompt: &str,
files: &[String],
k: usize,
) -> Vec<neuromesh_graph::RankedDefinition> {
const K: f32 = 60.0;
let file_rank: std::collections::HashMap<&str, usize> = files
.iter()
.enumerate()
.map(|(i, f)| (f.as_str(), i))
.collect();
let mut scored: Vec<(f32, usize, neuromesh_graph::RankedDefinition)> = graph
.rank_definitions(prompt, 300)
.into_iter()
.filter(|d| !crate::selector::is_noise_path(std::path::Path::new(&d.path)))
.enumerate()
.map(|(i, d)| {
let mut s = 1.0 / (K + i as f32 + 1.0);
if let Some(r) = file_rank.get(d.path.as_str()) {
s += 1.0 / (K + *r as f32 + 1.0);
}
(s, i, d)
})
.collect();
scored.sort_by(|a, b| b.0.total_cmp(&a.0).then_with(|| a.1.cmp(&b.1)));
scored.into_iter().take(k).map(|(_, _, d)| d).collect()
}

/// Indexed files a report names, best pointer first: traceback frames
/// (`File "…/pkg/mod.py", line 12`) from the deepest up, then any other path
/// or file name in order of mention. A path resolves by its longest suffix
Expand Down
40 changes: 25 additions & 15 deletions crates/neuromesh-graph/src/chunk_rank.rs
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,9 @@ pub(crate) struct ChunkSource {
pub spans: Vec<std::ops::Range<usize>>,
/// The definition each span is (`Class.method`), same order as `spans`.
pub names: Vec<String>,
/// Whether each span holds other definitions (a class): it still votes
/// for its file, but a function-level list names what is inside it.
pub containers: Vec<bool>,
}

/// One definition ranked for a report: where it is and how well it matched.
Expand All @@ -48,11 +51,10 @@ pub struct RankedDefinition {
}

/// A file's chunks as term counts, before they are numbered.
type FileChunks = (
NodeId,
PathBuf,
Vec<(HashMap<String, u16>, String, usize, usize)>,
);
type FileChunks = (NodeId, PathBuf, Vec<(HashMap<String, u16>, Label)>);

/// A chunk's definition name, first and last line, and whether it is a container.
type Label = (String, usize, usize, bool);

#[derive(Default)]
pub(crate) struct ChunkRankIndex {
Expand All @@ -62,7 +64,7 @@ pub(crate) struct ChunkRankIndex {
avg_len: f32,
post: HashMap<String, Vec<(u32, u16)>>,
/// Per chunk: definition name (`<head>` for the file head) and lines.
label: Vec<(String, usize, usize)>,
label: Vec<Label>,
}

impl ChunkRankIndex {
Expand All @@ -79,10 +81,15 @@ impl ChunkRankIndex {
};
let path_terms = terms_of(&stem_path, &st, true);
let lines: Vec<&str> = f.source.lines().collect();
let head = (1..HEAD_LINES.min(lines.len()), "<head>".to_string());
let head = (1..HEAD_LINES.min(lines.len()), ("<head>".to_string(), true));
let mut chunks = Vec::new();
let named = f.spans.iter().cloned().zip(f.names.iter().cloned());
for (span, name) in std::iter::once(head).chain(named) {
let named = f.spans.iter().cloned().zip(
f.names
.iter()
.cloned()
.zip(f.containers.iter().copied().chain(std::iter::repeat(false))),
);
for (span, (name, container)) in std::iter::once(head).chain(named) {
let start = span.start.max(1) - 1;
let end = span.end.min(lines.len()).min(start + MAX_CHUNK_LINES);
if start >= end {
Expand All @@ -97,7 +104,7 @@ impl ChunkRankIndex {
*c = c.saturating_add(1);
}
if !counts.is_empty() {
chunks.push((counts, name, start + 1, end));
chunks.push((counts, (name, start + 1, end, container)));
}
}
(f.id, f.path, chunks)
Expand All @@ -107,10 +114,10 @@ impl ChunkRankIndex {
for (id, path, chunks) in per_file {
let file = idx.files.len() as u32;
idx.files.push((id, path));
for (counts, name, start, end) in chunks {
for (counts, label) in chunks {
let chunk = idx.file_of.len() as u32;
idx.file_of.push(file);
idx.label.push((name, start, end));
idx.label.push(label);
idx.len.push(counts.values().map(|&c| c as u32).sum());
for (t, c) in counts {
idx.post.entry(t).or_default().push((chunk, c));
Expand All @@ -122,19 +129,21 @@ impl ChunkRankIndex {
idx
}

/// Definitions by score for `prompt`, best first (file heads left out).
/// Functions and methods by score for `prompt`, best first. File heads and
/// containers (a class) are left out: on SWE-bench dev a class body outranked
/// the method the patch edits (func Acc@5 0.268 -> 0.311 without them).
pub(crate) fn rank_definitions(&self, prompt: &str, limit: usize) -> Vec<RankedDefinition> {
let mut scored: Vec<(u32, f32)> = self
.scores(prompt)
.into_iter()
.filter(|(c, _)| self.label[*c as usize].0 != "<head>")
.filter(|(c, _)| !self.label[*c as usize].3)
.collect();
scored.sort_by(|a, b| b.1.total_cmp(&a.1).then_with(|| a.0.cmp(&b.0)));
scored
.into_iter()
.take(limit)
.map(|(c, score)| {
let (name, start, end) = &self.label[c as usize];
let (name, start, end, _) = &self.label[c as usize];
RankedDefinition {
path: self.files[self.file_of[c as usize] as usize]
.1
Expand Down Expand Up @@ -231,6 +240,7 @@ mod tests {
path: PathBuf::from(path),
source: source.to_string(),
names: spans.iter().map(|s| format!("def_{}", s.start)).collect(),
containers: vec![false; spans.len()],
spans,
}
}
Expand Down
15 changes: 13 additions & 2 deletions crates/neuromesh-graph/src/graph.rs
Original file line number Diff line number Diff line change
Expand Up @@ -825,23 +825,34 @@ impl NeuralProjectGraph {
}
}
}
// A definition is a container when another one in the file
// names it as its parent (a class with methods): it votes for
// its file but is left out of the function-level list.
let parents: HashSet<&str> = ids
.iter()
.filter_map(|id| data.mesh.node(id))
.filter_map(|n| n.parent.as_deref())
.collect();
let containers: Vec<bool> =
names.iter().map(|n| parents.contains(n.as_str())).collect();
if let Some(id) = file_id {
files.push((id, path.clone(), spans, names));
files.push((id, path.clone(), spans, names, containers));
}
}
(key, files)
};
let t0 = std::time::Instant::now();
let sources: Vec<crate::chunk_rank::ChunkSource> = files
.into_par_iter()
.filter_map(|(id, path, spans, names)| {
.filter_map(|(id, path, spans, names, containers)| {
let source = self.read_source(&path)?;
Some(crate::chunk_rank::ChunkSource {
id,
path,
source,
spans,
names,
containers,
})
})
.collect();
Expand Down
28 changes: 20 additions & 8 deletions crates/neuromesh-mcp/src/tools.rs
Original file line number Diff line number Diff line change
Expand Up @@ -514,15 +514,27 @@ impl McpToolHandler {
>= neuromesh_parser::text_normalize::REPORT_WORDS
{
if let Some(obj) = value.as_object_mut() {
obj.insert(
"where_to_look".into(),
json!(neuromesh_context::gold::localization_order(
&self.graph,
&task_desc,
&view,
10
)),
let files = neuromesh_context::gold::localization_order(
&self.graph,
&task_desc,
&view,
10,
);
// The functions inside those files to read first
// (`path:Class.method Lx-Ly`), five deep.
let functions: Vec<String> = neuromesh_context::gold::definition_order(
&self.graph,
&task_desc,
&files,
5,
)
.into_iter()
.map(|d| format!("{}:{} L{}-L{}", d.path, d.name, d.lines.0, d.lines.1))
.collect();
obj.insert("where_to_look".into(), json!(files));
if !functions.is_empty() {
obj.insert("functions_to_look".into(), json!(functions));
}
}
}
neuromesh_graph::timing("mcp: cache_and_build", t_s);
Expand Down
11 changes: 10 additions & 1 deletion docs/measured.md
Original file line number Diff line number Diff line change
Expand Up @@ -74,7 +74,16 @@ with Claude-3.5 0.726 / 0.792 / 0.796; LocAgent with Claude-3.5 0.777 / 0.920 /
remain ahead at Acc@1; the engine is an LLM-free first stage they can start from.

SWE-bench dev (tuning set, 225): plain BM25 0.160 / 0.347 / 0.427 / 0.538 → 0.249 / 0.484 / 0.600 /
0.680. Raw results: `swebench/results-final-test.jsonl` (outside the repository); harness
0.680.

**SWE-bench Verified** (500 human-validated issues, v1.2.0 engine, run once; 496 scored, 4 lost to
git checkout errors): plain BM25 0.216 / 0.391 / 0.490 / 0.641 → **0.405 / 0.669 / 0.732 / 0.804**
(Acc@1/3/5/10). On the 403 Verified issues that are not in Lite: BM25 0.194 / 0.372 / 0.476 / 0.620
→ 0.392 / 0.655 / 0.727 / 0.809.

**Which part does the work** (Lite strict holdout, one run per removed component, Acc@1/3/5):
definition-level ranking removed 0.388 / 0.569 / 0.652; report hygiene removed 0.467 / 0.707 /
0.743; packet order only 0.366 / 0.536 / 0.558; full 0.486 / 0.710 / 0.750. Raw results: `swebench/results-final-test.jsonl` (outside the repository); harness
`scripts/swebench_localize.py`, scoring `scripts/research/eval_loc.py`.

## Speed (same machine, same hour, release binary without embeddings)
Expand Down
34 changes: 34 additions & 0 deletions docs/research/contributions-log.md
Original file line number Diff line number Diff line change
Expand Up @@ -361,3 +361,37 @@ functions in the top k). dev-fast (48 issues with function gold): func Acc@1/5/1
0.062/0.229/0.292/0.396 (raw definition ranking, no file prior). Reference (LocAgent Table 4, Lite,
function level Acc@5/@10): BM25 0.318/0.369, CodeRankEmbed 0.518/0.588, Agentless+Claude 0.588,
LocAgent+Claude 0.734/0.774. File-prior orderings (`func_hier.py`) pending the full-dev run.

### 8.11 SWE-bench Verified (second standard benchmark) and the ablation table (2026-10-09)

**Verified, v1.2.0 engine, run once** (496 of 500; 4 lost to git checkout errors):

| | Acc@1 | Acc@3 | Acc@5 | Acc@10 |
|---|---|---|---|---|
| engine, 496 | **0.405** [0.36,0.45] | **0.669** [0.63,0.71] | **0.732** [0.69,0.77] | **0.804** |
| BM25, 496 | 0.216 | 0.391 | 0.490 | 0.641 |
| engine, 403 never seen before (not in Lite) | 0.392 | 0.655 | 0.727 | 0.809 |
| BM25, same 403 | 0.194 | 0.372 | 0.476 | 0.620 |

**Ablation on the Lite strict holdout (276)**, one run per removed component (`NM_ABLATE` switch
build, not shipped):

| removed | Acc@1 | Acc@3 | Acc@5 | Acc@10 |
|---|---|---|---|---|
| nothing (v1.2.0) | 0.486 | 0.710 | 0.750 | 0.804 |
| report hygiene + code tokens (C11) | 0.467 | 0.707 | 0.743 | 0.801 |
| definition-level ranking (C12) | **0.388** | **0.569** | **0.652** | 0.764 |
| localisation list — packet order only | 0.366 | 0.536 | 0.558 | – |

Reading: the definition-level ranking (LocAgent's entity-content BM25 idea, made LLM-free) carries
most of the gain over the packet (+0.10 @1, +0.14 @3). Hygiene is small on Lite (−0.019 @1 when
removed) — its dev gains came mostly from sqlfluff's template-heavy issues; reported as is.

### 8.12 Function level: classes out, file prior in (engine, dev 225)

Classes (a definition another one names as its parent) stay in the file ranking but leave the
function list; the list is fused (RRF) with each definition's file rank in `where_to_look`.
Engine on SWE-bench dev (189 issues with function gold), func Acc@1/5/10/20: 0.137/0.268/0.337/–
→ **0.175/0.323/0.402/0.460** (prototype predicted 0.174/0.321/0.400). Same run, file level with
everything since v1.2.0 (named files): 0.249/0.484/0.600/0.680 → **0.308/0.527/0.621/0.692**.
MCP: reports now also get `functions_to_look` (five, `path:Class.method Lx-Ly`).
10 changes: 6 additions & 4 deletions docs/research/paper-draft.md
Original file line number Diff line number Diff line change
Expand Up @@ -70,17 +70,19 @@ tokens, definition-level BM25 with title ×3, RRF with the packet order → `whe

† LocAgent Table 4, their 274-instance subset (head-to-head on the same subset: TODO).

### 5.2 SWE-bench dev (225) and Verified (500, TODO)
### 5.2 SWE-bench dev (225) and Verified (496 of 500)

dev: BM25 0.160/0.347/0.427/0.538 → engine 0.249/0.484/0.600/0.680 (Acc@1/3/5/10).
Verified (v1.2.0, run once): BM25 0.216/0.391/0.490/0.641 → engine 0.405/0.669/0.732/0.804; on the
403 Verified issues not in Lite: 0.194/0.372/0.476/0.620 → 0.392/0.655/0.727/0.809.

### 5.3 Ablation (Lite holdout) — TODO (runs in progress)
### 5.3 Ablation (Lite holdout, 276)

| removed | Acc@1 | Acc@3 | Acc@5 |
|---|---|---|---|
| nothing | 0.486 | 0.710 | 0.750 |
| report hygiene + code tokens | TODO | | |
| definition-level ranking | TODO | | |
| report hygiene + code tokens | 0.467 | 0.707 | 0.743 |
| definition-level ranking | 0.388 | 0.569 | 0.652 |
| localisation list (packet order only) | 0.366 | 0.536 | 0.558 |

### 5.4 Plain-language holdouts (R@3)
Expand Down
Loading