From 17a6be7e6d1f74f395525f3c7321caaf01c845cf Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 07:27:22 +0700 Subject: [PATCH 01/15] =?UTF-8?q?perf(bench):=20th=C3=AAm=20profiler=20RAM?= =?UTF-8?q?=20+=20workflow=20=C4=91o=20memory=20tr=C3=AAn=20CI?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `GraphIndex` là in-memory-first: `rebuild()` nạp toàn bộ symbols/chains/ call_names/edges vào HashMap. CodSpeed chỉ đo thời gian nên cái giá RAM này không ai thấy — thêm đường đo riêng trước khi tối ưu. - `codegraph-graph/src/meminfo.rs`: đọc RSS std-only (Linux `/proc/self/statm` + `VmHWM`, macOS fallback `ps -o rss=`), kèm `MemTracker` lấy mẫu theo phase. Không thêm dependency. - `codegraph-bench/src/bin/mem`: đo RSS theo extract → index → query trên repo thật, đồng thời dự đoán RAM theo cấu trúc (`count × size_of::()`) để quy kết quả về đúng HashMap nào. Chế độ `--synthetic N` sinh index deterministic (không phụ thuộc network) với quy mô đủ để `edges`/`call_names` chiếm RAM đáng kể. - `.github/workflows/memory.yml`: chạy profiler, in bảng vào job summary, upload artifact. Không chặn merge — GitHub runner là VM dùng chung nên RSS chỉ dùng so sánh tương đối, phần `predicted` mới là số ổn định. Còn lại: tối ưu `edges`/`call_names`/`symbols` ra khỏi RAM (storage-backed lazy). PR này mới phần đo, chưa đụng storage trait. --- .github/workflows/memory.yml | 73 ++++++ crates/codegraph-bench/src/bin/mem.rs | 328 ++++++++++++++++++++++++++ crates/codegraph-graph/src/lib.rs | 1 + crates/codegraph-graph/src/meminfo.rs | 199 ++++++++++++++++ 4 files changed, 601 insertions(+) create mode 100644 .github/workflows/memory.yml create mode 100644 crates/codegraph-bench/src/bin/mem.rs create mode 100644 crates/codegraph-graph/src/meminfo.rs diff --git a/.github/workflows/memory.yml b/.github/workflows/memory.yml new file mode 100644 index 0000000000..c8ddf769f8 --- /dev/null +++ b/.github/workflows/memory.yml @@ -0,0 +1,73 @@ +# Đo RAM của codegraph — chạy SONG SONG với CI, không chặn merge. +# +# Vì sao cần workflow riêng: CodSpeed đo *thời gian*, còn `GraphIndex` là +# in-memory-first — mở index là nạp toàn bộ vào RAM (`rebuild()` gọi +# `load_all_symbols` / `all_chains` / ...). Cái giá RAM đó không hiện ra trong +# benchmark thời gian nên phải đo riêng. +# +# Số liệu RSS trên GitHub-hosted runner là VM dùng chung → **chỉ dùng để so +# sánh tương đối** trước/sau trong cùng workflow, không dùng làm ngưỡng cứng. +# Đường `synthetic` là deterministic nên phần "dự đoán" (predicted symbols/edges) +# mới là con số ổn định thật sự. +name: Memory + +on: + push: + branches: [main] + pull_request: + workflow_dispatch: + +permissions: + contents: read + +env: + CARGO_TERM_COLOR: always + +jobs: + memory: + name: memory profile + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: dtolnay/rust-toolchain@stable + - uses: Swatinem/rust-cache@v2 + + - name: Fetch bench repos + run: bash .github/benches/fetch_repos.sh + + # Build sẵn để lỗi compile lộ ra ở step riêng, dễ đọc. + - name: Build memory profiler + run: cargo build --release -p codegraph-bench --bin mem + + # Quy mô synthetic: đủ lớn để `edges`/`call_names` chiếm RAM đáng kể, + # vẫn chạy nhanh trên runner. + - name: Synthetic 50k functions + run: | + mkdir -p mem-out + ./target/release/mem --synthetic 50000 --fanout 4 \ + --json | tee mem-out/synthetic-50k.json + + - name: Real repos + env: + CODEGRAPH_BENCH_REPOS_LIST: ${{ github.workspace }}/.github/benches/repos/list.txt + run: | + ./target/release/mem --json | tee mem-out/repos.json + ./target/release/mem | tee mem-out/repos.txt + + - name: Summary + run: | + echo "### Memory profile" >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" + cat mem-out/repos.txt >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" + echo '```json' >> "$GITHUB_STEP_SUMMARY" + cat mem-out/synthetic-50k.json >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" + + - name: Upload results + uses: actions/upload-artifact@v4 + with: + name: memory-report + path: mem-out/ + retention-days: 14 diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs new file mode 100644 index 0000000000..003f71ab9f --- /dev/null +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -0,0 +1,328 @@ +//! Profiler RAM: đo RSS theo từng phase của pipeline thật (extract → index → +//! query) trên danh sách repo, kèm ước lượng RAM của các HashMap trong +//! `GraphIndex` để **quy kết quả về đúng cấu trúc dữ liệu**. +//! +//! Vì sao cần: `GraphIndex` in-memory-first — mở index là nạp hết vào RAM +//! (`rebuild()` gọi `load_all_symbols` / `all_chains` / ...). Benchmark thời gian +//! của CodSpeed không thấy được cái giá này, nên phải đo riêng. +//! +//! Chạy: +//! ```bash +//! cargo run -p codegraph-bench --bin mem -- crates +//! CODEGRAPH_BENCH_REPOS_LIST=list.txt cargo run -p codegraph-bench --bin mem +//! ``` + +use camino::Utf8Path; +use clap::Parser; +use codegraph_bench::{BenchOptions, Repo, extract, index_at, orchestrator, run_queries}; +use codegraph_core::{EdgeMeta, Symbol}; +use codegraph_graph::meminfo::{MemTracker, fmt_bytes, rss_bytes}; + +#[derive(Parser)] +#[command(name = "codegraph-mem", about = "Đo RAM của codegraph theo từng phase")] +struct Cli { + /// Folder repo cần đo (nhiều được). + #[arg(value_name = "REPO")] + repos: Vec, + + /// File chứa danh sách repo (mỗi dòng 1 path, trống + `#` bị bỏ). + #[arg(short, long)] + file: Option, + + /// Giới hạn ngôn ngữ: `rust,go`… + #[arg(long)] + langs: Option, + + /// Số symbol lấy mẫu cho phase query. + #[arg(long, default_value_t = 200)] + queries: usize, + + /// In JSON thay cho bảng. + #[arg(long)] + json: bool, + + /// Dựng index synthetic với N function thay vì đọc repo thật — deterministic, + /// không phụ thuộc network. Mỗi function gọi `--fanout` function khác nên + /// `edges` + `call_names` (hai cấu trúc đang tối ưu) có quy mô đáng kể. + #[arg(long, value_name = "N")] + synthetic: Option, + + /// Số callee mỗi function trong chế độ `--synthetic`. + #[arg(long, default_value_t = 4, value_name = "K")] + fanout: usize, +} + +/// Kết quả 1 repo — RSS từng phase + phần RAM dự đoán theo cấu trúc. +#[derive(serde::Serialize)] +struct RepoMem { + repo: String, + symbols: u64, + chains: u64, + edges: u64, + files: u64, + /// RSS sau extract (bytes) — chưa có index. + rss_after_extract: u64, + /// RSS sau index (bytes) — có `GraphIndex` đầy đủ. + rss_after_index: u64, + /// RSS sau query (bytes). + rss_after_query: u64, + /// RSS đỉnh theo các mốc đã đánh dấu. + rss_peak: u64, + /// RSS tăng do `ingest` (sau_index − sau_extract). + rss_index_delta: u64, + /// `symbols.len() × size_of::()` — phần trong HashMap `symbols`. + predicted_symbols: u64, + /// `edges.len() × size_of::()` — phần trong HashMap `edges`. + predicted_edges: u64, +} + +fn load_repos(cli: &Cli) -> Vec { + let mut paths: Vec = cli.repos.clone(); + if paths.is_empty() { + let list_file = cli + .file + .clone() + .or_else(|| std::env::var("CODEGRAPH_BENCH_REPOS_LIST").ok()); + if let Some(list_file) = list_file + && let Ok(body) = std::fs::read_to_string(&list_file) + { + for line in body.lines() { + let line = line.trim(); + if !line.is_empty() && !line.starts_with('#') { + paths.push(line.to_string()); + } + } + } + } + if paths.is_empty() { + paths.push("crates".to_string()); + } + paths + .into_iter() + .map(|p| { + let name = std::path::Path::new(&p) + .file_name() + .and_then(|s| s.to_str()) + .unwrap_or(&p) + .to_string(); + Repo { + name, + root: p.into(), + } + }) + .collect() +} + +fn measure(repo: &Repo, opts: &BenchOptions) -> anyhow::Result { + let orch = orchestrator(opts); + let mut tracker = MemTracker::new(); + + tracker.mark("start"); + let (parsed, _stats) = extract(&orch, repo.root.as_path())?; + tracker.mark("extract"); + + // `None` = in-memory storage → đo đúng RAM của `GraphIndex`, không lẫn + // page cache của sqlite/lmdb. + let idx = index_at(&parsed, None)?; + tracker.mark("index"); + + let names = codegraph_bench::sample_query_names(&parsed, opts.queries); + let _ = run_queries(&idx, &names, opts.with_flow); + tracker.mark("query"); + + let st = idx.stats(); + let sample = |label: &str| { + tracker + .samples() + .iter() + .find(|(l, _)| l == label) + .map(|(_, v)| *v) + .unwrap_or(0) + }; + let after_extract = sample("extract"); + let after_index = sample("index"); + + Ok(RepoMem { + repo: repo.name.clone(), + symbols: st.symbols, + chains: st.chains, + edges: st.edges, + files: st.files, + rss_after_extract: after_extract, + rss_after_index: after_index, + rss_after_query: sample("query"), + rss_peak: tracker.peak(), + rss_index_delta: after_index.saturating_sub(after_extract), + predicted_symbols: st.symbols * size_of::() as u64, + predicted_edges: st.edges * size_of::() as u64, + }) +} + +/// Dựng `ParseResult` synthetic: `n` function, mỗi function gọi `fanout` +/// function khác (id local tính từ `SYMBOL_BASE`). +/// +/// Mục tiêu là **làm đầy `edges` + `call_names`** — hai `HashMap` đang tốn +/// nhiều RAM nhất trong `GraphIndex`. Call name cố tình trùng lặp (chỉ vài +/// tên lib giả) để `call_names` có nhiều key chứa nhiều site, đúng hình dạng +/// repo thật. +fn synthetic_parse_result(n: usize, fanout: usize) -> codegraph_graph::ParseResult { + // `n = 0` sẽ làm `% n` panic ở vòng sinh chain — chặn sớm, báo rõ. + assert!(n > 0, "--synthetic cần N > 0"); + let mut symbols = Vec::with_capacity(n); + let mut chains = std::collections::HashMap::with_capacity(n); + let mut calls = Vec::with_capacity(n * fanout); + + for i in 0..n { + symbols.push(Symbol { + id: codegraph_core::SYMBOL_BASE + i as u64, + name: format!("fn_{i:06}"), + kind: SymbolKind::Function, + scope: ScopeLevel::Global, + scope_id: 0, + type_ref: 0, + type_name: None, + file: format!("src/synthetic_{:04}.rs", i / 64), + line: 1, + end_line: 20, + signature: Some(format!("fn fn_{i:06}(x: i64) -> i64")), + doc: Some("Synthetic symbol để đo RAM — không phải code thật.".into()), + annotations: Vec::::new(), + language: "rust".into(), + }); + } + + let fanout = fanout.max(1); + for i in 0..n { + let caller = codegraph_core::SYMBOL_BASE + i as u64; + let mut chain = vec![caller]; + for k in 0..fanout { + // Callee = function k vòng sau (wrap-around) → id luôn hợp lệ, + // không tự gọi chính mình. + let callee = codegraph_core::SYMBOL_BASE + ((i + k + 1) % n) as u64; + chain.push(callee); + calls.push(CallRecord { + caller_id: caller, + call_name: format!("lib::helper_{}", k % 4), + position: chain.len() - 1, + arg_exprs: vec!["x".into()], + line: 5 + k as u32, + condition: (k % 3 == 0).then(|| "x > 0".to_string()), + is_loop_body: k % 5 == 0, + effect: EffectType::None, + effect_desc: None, + target_class: None, + target_method: None, + }); + } + chains.insert(caller, chain); + } + + codegraph_graph::ParseResult { + path: "synthetic.rs".into(), + language: "rust".into(), + bytes: n as u64 * 512, + lines: n as u32, + symbols, + chains, + calls, + } +} + +fn measure_synthetic(n: usize, fanout: usize) -> anyhow::Result { + let mut tracker = MemTracker::new(); + tracker.mark("start"); + + let parsed = vec![synthetic_parse_result(n, fanout)]; + tracker.mark("extract"); + + let idx = index_at(&parsed, None)?; + tracker.mark("index"); + + let st = idx.stats(); + let sample = |label: &str| { + tracker + .samples() + .iter() + .find(|(l, _)| l == label) + .map(|(_, v)| *v) + .unwrap_or(0) + }; + let after_extract = sample("extract"); + let after_index = sample("index"); + + Ok(RepoMem { + repo: format!("synthetic({n})"), + symbols: st.symbols, + chains: st.chains, + edges: st.edges, + files: st.files, + rss_after_extract: after_extract, + rss_after_index: after_index, + rss_after_query: 0, + rss_peak: tracker.peak(), + rss_index_delta: after_index.saturating_sub(after_extract), + predicted_symbols: st.symbols * size_of::() as u64, + predicted_edges: st.edges * size_of::() as u64, + }) +} + +fn main() -> anyhow::Result<()> { + let cli = Cli::parse(); + if rss_bytes().is_none() { + eprintln!("cảnh báo: không đọc được RSS trên nền tảng này — số liệu sẽ là 0"); + } + let opts = BenchOptions { + langs: cli + .langs + .as_ref() + .map(|s| s.split(',').map(|x| x.trim().to_string()).collect()), + queries: cli.queries, + with_flow: false, + }; + + let mut results = Vec::new(); + if let Some(n) = cli.synthetic { + results.push(measure_synthetic(n, cli.fanout)?); + } else { + for repo in load_repos(&cli) { + if Utf8Path::from_path(repo.root.as_std_path()).is_none() { + eprintln!("bỏ qua {}: path không phải UTF-8", repo.root); + continue; + } + match measure(&repo, &opts) { + Ok(r) => results.push(r), + Err(e) => eprintln!("lỗi {}: {e}", repo.root), + } + } + } + + if cli.json { + println!("{}", serde_json::to_string_pretty(&results)?); + return Ok(()); + } + + println!( + "{:<14} {:>8} {:>8} {:>10} {:>10} {:>10} {:>12} {:>12}", + "repo", "symbols", "edges", "rss extract", "rss index", "Δ index", "pred symbols", + "pred edges" + ); + for r in &results { + println!( + "{:<14} {:>8} {:>8} {:>10} {:>10} {:>10} {:>12} {:>12}", + r.repo, + r.symbols, + r.edges, + fmt_bytes(r.rss_after_extract), + fmt_bytes(r.rss_after_index), + fmt_bytes(r.rss_index_delta), + fmt_bytes(r.predicted_symbols), + fmt_bytes(r.predicted_edges), + ); + } + if let Some(peak) = codegraph_graph::meminfo::peak_rss_bytes() { + println!("\npeak RSS (VmHWM): {}", fmt_bytes(peak)); + } else { + println!("\npeak RSS: không đọc được trên nền tảng này"); + } + Ok(()) +} diff --git a/crates/codegraph-graph/src/lib.rs b/crates/codegraph-graph/src/lib.rs index c906cd2cb0..a111ee6350 100644 --- a/crates/codegraph-graph/src/lib.rs +++ b/crates/codegraph-graph/src/lib.rs @@ -77,6 +77,7 @@ mod bloom; pub mod diff; pub mod embeddings; mod lru; +pub mod meminfo; mod radix; mod search; mod shared; diff --git a/crates/codegraph-graph/src/meminfo.rs b/crates/codegraph-graph/src/meminfo.rs new file mode 100644 index 0000000000..070a5c2dc9 --- /dev/null +++ b/crates/codegraph-graph/src/meminfo.rs @@ -0,0 +1,199 @@ +//! Đo RAM của process — **std-only**, không thêm dependency. +//! +//! Mục tiêu: đo được memory thực của `GraphIndex` (in-memory-first nên RAM là +//! tài nguyên khan hiếm), phục vụ hai việc: benchmark tối ưu bộ nhớ, và hiển thị +//! trong `codegraph_status`. +//! +//! - **Linux/BSD**: đọc `/proc/self/statm` (RSS hiện tại, × page size) và +//! `/proc/self/status` `VmHWM` (RSS đỉnh). +//! - **macOS**: `/proc` không có → gọi `ps -o rss= -p ` (KB). Chỉ có RSS +//! hiện tại; đỉnh để `None` ( caller tự track bằng [`MemTracker`]). +//! - **Platform khác**: `None` — caller phải xử lý, KHÔNG panic. +//! +//! ```no_run +//! use codegraph_graph::meminfo::{MemTracker, rss_bytes}; +//! +//! let mut t = MemTracker::new(); +//! t.mark("baseline"); +//! // … mở index / ingest … +//! let peak = t.mark("sau ingest"); // max RSS thấy từ lần mark trước +//! println!("peak = {}", peak); +//! # let _ = rss_bytes(); +//! ``` + +/// Page size mặc định khi không đọc được từ hệ thống (Linux arm64/x86_64 đều +/// 4096). Sai số này chỉ ảnh hưởng con số báo cáo, không ảnh hưởng logic. +const FALLBACK_PAGE_SIZE: u64 = 4096; + +/// RSS hiện tại của process (bytes). `None` nếu platform không hỗ trợ. +pub fn rss_bytes() -> Option { + #[cfg(target_os = "linux")] + { + let statm = std::fs::read_to_string("/proc/self/statm").ok()?; + // statm: size resident shared text lib data dt — tính bằng **page**. + // resident là field thứ 2 (index 1). + let resident_pages: u64 = statm.split_whitespace().nth(1)?.parse().ok()?; + return Some(resident_pages.saturating_mul(page_size())); + } + #[cfg(not(target_os = "linux"))] + { + rss_via_ps() + } +} + +/// RSS đỉnh từ lúc process bắt đầu (bytes). +/// +/// Chỉ có trên Linux (`VmHWM`). macOS trả `None` — dùng [`MemTracker`] để tự +/// lấy mẫu theo phase. +pub fn peak_rss_bytes() -> Option { + #[cfg(target_os = "linux")] + { + let status = std::fs::read_to_string("/proc/self/status").ok()?; + for line in status.lines() { + if let Some(rest) = line.strip_prefix("VmHWM:") { + let kb: u64 = rest.split_whitespace().next()?.parse().ok()?; + return Some(kb.saturating_mul(1024)); + } + } + None + } + #[cfg(not(target_os = "linux"))] + { + None + } +} + +/// Định dạng byte cho log/bench: `12.3 MiB`. +pub fn fmt_bytes(bytes: u64) -> String { + const UNITS: [&str; 5] = ["B", "KiB", "MiB", "GiB", "TiB"]; + let mut value = bytes as f64; + let mut unit = 0; + while value >= 1024.0 && unit + 1 < UNITS.len() { + value /= 1024.0; + unit += 1; + } + if unit == 0 { + format!("{bytes} B") + } else { + format!("{value:.1} {}", UNITS[unit]) + } +} + +/// Theo dõi RSS theo từng phase — [`rss_bytes`] không có sẵn trên mọi nền tảng +/// (đỉnh), nên benchmark tự lấy mẫu tại các mốc rồi giữ max. +/// +/// ```no_run +/// # use codegraph_graph::meminfo::MemTracker; +/// let mut t = MemTracker::new(); +/// t.mark("open"); +/// t.mark("ingest"); +/// for (label, bytes) in t.samples() { println!("{label}: {bytes}"); } +/// ``` +#[derive(Debug, Clone)] +pub struct MemTracker { + samples: Vec<(String, u64)>, + peak: u64, +} + +impl Default for MemTracker { + fn default() -> Self { + Self::new() + } +} + +impl MemTracker { + pub fn new() -> Self { + Self { + samples: Vec::new(), + peak: 0, + } + } + + /// Lấy mẫu RSS ở mốc `label`. RSS không đọc được → ghi `0` và **không** cập + /// nhật peak, để không báo nhầm số 0 là "không tốn RAM". + pub fn mark(&mut self, label: impl Into) -> u64 { + let rss = rss_bytes().unwrap_or(0); + if rss > self.peak { + self.peak = rss; + } + self.samples.push((label.into(), rss)); + rss + } + + /// RSS đỉnh trong các mốc đã đánh dấu. + pub fn peak(&self) -> u64 { + self.peak + } + + /// Toàn bộ mẫu theo thứ tự thời gian. + pub fn samples(&self) -> &[(String, u64)] { + &self.samples + } +} + +// ── Platform helpers ── + +#[cfg(target_os = "linux")] +fn page_size() -> u64 { + // `getconf PAGESIZE` không portable qua std; đọc từ `/proc/self/smaps` quá + // nặng. Linux hầu hết là 4096 — chấp nhận sai số nhỏ thay vì thêm libc. + FALLBACK_PAGE_SIZE +} + +#[cfg(not(target_os = "linux"))] +fn rss_via_ps() -> Option { + // `ps -o rss= -p ` in ra RSS **tính bằng KiB**. Cần pid: đọc + // `/proc/self` không có trên macOS nên dùng `std::process::id()`. + let out = std::process::Command::new("ps") + .args(["-o", "rss=", "-p", &std::process::id().to_string()]) + .output() + .ok()?; + if !out.status.success() { + return None; + } + let kb: u64 = String::from_utf8_lossy(&out.stdout).trim().parse().ok()?; + Some(kb.saturating_mul(1024)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn rss_is_plausible_when_supported() { + // Chỉ assert khi platform đọc được — CI chạy Linux nên có, nhưng để + // chấp nhận `None` ở nền tảng khác. + if let Some(rss) = rss_bytes() { + assert!(rss > 0, "RSS phải > 0, got {rss}"); + // Một process Rust không thể dưới vài MB. + assert!(rss > 512 * 1024, "RSS nhỏ bất thường: {rss}"); + } + } + + #[test] + fn peak_ge_current_on_linux() { + if let Some(peak) = peak_rss_bytes() + && let Some(rss) = rss_bytes() + { + assert!(peak >= rss, "VmHWM ({peak}) phải >= RSS ({rss})"); + } + } + + #[test] + fn fmt_bytes_scales_units() { + assert_eq!(fmt_bytes(512), "512 B"); + assert_eq!(fmt_bytes(1024), "1.0 KiB"); + assert_eq!(fmt_bytes(1024 * 1024 * 3 / 2), "1.5 MiB"); + } + + #[test] + fn tracker_keeps_max() { + let mut t = MemTracker::new(); + t.mark("a"); + let first = t.samples()[0].1; + t.mark("b"); + assert_eq!(t.samples().len(), 2); + assert!(t.peak() >= first); + assert!(t.samples()[0].0 == "a"); + } +} From b18fd718ccfa3480eeeefc273437c776bd91c50c Mon Sep 17 00:00:00 2001 From: hungpham10 <136320753+hungpham10@users.noreply.github.com> Date: Mon, 5 Oct 2026 00:33:43 +0000 Subject: [PATCH 02/15] style: apply rustfmt --- crates/codegraph-bench/src/bin/mem.rs | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index 003f71ab9f..d10415b4ea 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -303,7 +303,13 @@ fn main() -> anyhow::Result<()> { println!( "{:<14} {:>8} {:>8} {:>10} {:>10} {:>10} {:>12} {:>12}", - "repo", "symbols", "edges", "rss extract", "rss index", "Δ index", "pred symbols", + "repo", + "symbols", + "edges", + "rss extract", + "rss index", + "Δ index", + "pred symbols", "pred edges" ); for r in &results { From dfb74dc471436d488d95d7c9998250345906b63f Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 07:40:42 +0700 Subject: [PATCH 03/15] =?UTF-8?q?fix(bench):=20s=E1=BB=ADa=203=20l?= =?UTF-8?q?=E1=BB=97i=20CI=20=E2=80=94=20needless=5Freturn,=20const=20ch?= =?UTF-8?q?=E1=BA=BFt,=20thi=E1=BA=BFu=20import?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Báo qua CI của PR #38: - `meminfo.rs`: tách `rss_linux()` riêng để `rss_bytes()` không có `return` (clippy `needless_return`, fail cả job clippy lẫn slim build). - `meminfo.rs`: `FALLBACK_PAGE_SIZE` chỉ dùng trong nhánh linux → gate `#[cfg(target_os = "linux")]` (dead_code trên Windows). - `bin/mem.rs`: bổ sung import `Annotation`, `CallRecord`, `EffectType`, `ScopeLevel`, `SymbolKind` — patch import đầu tiên không áp dụng nên các kiểu này chưa có tên trong scope. --- crates/codegraph-bench/src/bin/mem.rs | 4 +++- crates/codegraph-graph/src/meminfo.rs | 19 ++++++++++++++----- 2 files changed, 17 insertions(+), 6 deletions(-) diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index d10415b4ea..afb13d0c36 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -15,7 +15,9 @@ use camino::Utf8Path; use clap::Parser; use codegraph_bench::{BenchOptions, Repo, extract, index_at, orchestrator, run_queries}; -use codegraph_core::{EdgeMeta, Symbol}; +use codegraph_core::{ + Annotation, CallRecord, EffectType, EdgeMeta, ScopeLevel, Symbol, SymbolKind, +}; use codegraph_graph::meminfo::{MemTracker, fmt_bytes, rss_bytes}; #[derive(Parser)] diff --git a/crates/codegraph-graph/src/meminfo.rs b/crates/codegraph-graph/src/meminfo.rs index 070a5c2dc9..fda44cdd96 100644 --- a/crates/codegraph-graph/src/meminfo.rs +++ b/crates/codegraph-graph/src/meminfo.rs @@ -23,17 +23,14 @@ /// Page size mặc định khi không đọc được từ hệ thống (Linux arm64/x86_64 đều /// 4096). Sai số này chỉ ảnh hưởng con số báo cáo, không ảnh hưởng logic. +#[cfg(target_os = "linux")] const FALLBACK_PAGE_SIZE: u64 = 4096; /// RSS hiện tại của process (bytes). `None` nếu platform không hỗ trợ. pub fn rss_bytes() -> Option { #[cfg(target_os = "linux")] { - let statm = std::fs::read_to_string("/proc/self/statm").ok()?; - // statm: size resident shared text lib data dt — tính bằng **page**. - // resident là field thứ 2 (index 1). - let resident_pages: u64 = statm.split_whitespace().nth(1)?.parse().ok()?; - return Some(resident_pages.saturating_mul(page_size())); + rss_linux() } #[cfg(not(target_os = "linux"))] { @@ -41,6 +38,18 @@ pub fn rss_bytes() -> Option { } } +/// RSS qua `/proc/self/statm` — Linux. Tách riêng để `rss_bytes()` không cần +/// `return` (clippy `needless_return`) và để non-linux không phải giữ nhánh +/// chết. +#[cfg(target_os = "linux")] +fn rss_linux() -> Option { + let statm = std::fs::read_to_string("/proc/self/statm").ok()?; + // statm: size resident shared text lib data dt — tính bằng **page**. + // resident là field thứ 2 (index 1). + let resident_pages: u64 = statm.split_whitespace().nth(1)?.parse().ok()?; + Some(resident_pages.saturating_mul(page_size())) +} + /// RSS đỉnh từ lúc process bắt đầu (bytes). /// /// Chỉ có trên Linux (`VmHWM`). macOS trả `None` — dùng [`MemTracker`] để tự From 1112123c231b1c9f9be0d72358b7af018fe02a3f Mon Sep 17 00:00:00 2001 From: hungpham10 <136320753+hungpham10@users.noreply.github.com> Date: Mon, 5 Oct 2026 01:12:04 +0000 Subject: [PATCH 04/15] style: apply rustfmt --- crates/codegraph-bench/src/bin/mem.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index afb13d0c36..a9acfc6737 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -16,7 +16,7 @@ use camino::Utf8Path; use clap::Parser; use codegraph_bench::{BenchOptions, Repo, extract, index_at, orchestrator, run_queries}; use codegraph_core::{ - Annotation, CallRecord, EffectType, EdgeMeta, ScopeLevel, Symbol, SymbolKind, + Annotation, CallRecord, EdgeMeta, EffectType, ScopeLevel, Symbol, SymbolKind, }; use codegraph_graph::meminfo::{MemTracker, fmt_bytes, rss_bytes}; From e97afba79f3819a4e1c97f6f35a9cf39fddd562f Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 08:40:15 +0700 Subject: [PATCH 05/15] =?UTF-8?q?fix(bench):=20m=E1=BB=97i=20repo=20m?= =?UTF-8?q?=E1=BB=99t=20process=20ri=C3=AAng=20khi=20=C4=91o=20memory?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Chạy `mem` với nhiều repo trong cùng process làm RSS cộng dồn: repo sau đo trên nền của repo trước nên `rss_after_extract` không còn đọc được (synthetic thì không dính vì chỉ đo 1 index). Tách từng repo ra process riêng — chậm hơn vài giây nhưng số đo đúng. --- .github/workflows/memory.yml | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/.github/workflows/memory.yml b/.github/workflows/memory.yml index c8ddf769f8..ea222b988c 100644 --- a/.github/workflows/memory.yml +++ b/.github/workflows/memory.yml @@ -52,8 +52,15 @@ jobs: env: CODEGRAPH_BENCH_REPOS_LIST: ${{ github.workspace }}/.github/benches/repos/list.txt run: | - ./target/release/mem --json | tee mem-out/repos.json - ./target/release/mem | tee mem-out/repos.txt + # Mỗi repo một PROCESS riêng — chạy chung process thì RSS của repo + # sau cộng dồn vào repo trước, `rss_after_extract` không còn đọc được. + : > mem-out/repos.json + : > mem-out/repos.txt + while read -r repo; do + [ -z "$repo" ] && continue + ./target/release/mem "$repo" --json >> mem-out/repos.json + ./target/release/mem "$repo" | tee -a mem-out/repos.txt + done < "$CODEGRAPH_BENCH_REPOS_LIST" - name: Summary run: | From 98cbb5df90229e8b6ced36c824db30ff5e405df7 Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 08:48:37 +0700 Subject: [PATCH 06/15] =?UTF-8?q?perf(graph):=20quy=20RSS=20v=E1=BB=81=20?= =?UTF-8?q?=C4=91=C3=BAng=20c=E1=BA=A5u=20tr=C3=BAc=20d=E1=BB=AF=20li?= =?UTF-8?q?=E1=BB=87u=20(mem=5Fbreakdown)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Baseline đo được trên CI cho thấy 320 MiB `Δ index` với 50k symbol / 200k edge, nhưng `symbols + edges` chỉ quy được 29.7 MiB — 91% nằm ở nơi khác. Không có cách nào chọn đúng cấu trúc cần tối ưu khi chỉ có tổng RSS. - `memtrack.rs`: deep size từng cấu trúc in-memory của `GraphIndex` (`symbols`, `chains_map`, `call_names`, `edges`, `name_index`, `scope_index`, `name_keys`, `files`). Heap của `String`/`Vec` đo exact bằng `capacity()`; bucket `HashMap` ước lượng theo công thức hashbrown. Ghi rõ trong doc phần nào exact / ước lượng / không tính. - `LruCache::len` / `is_empty` (DashMap::len O(1)) + `Storage::cache_occupancy` (default `Vec::new()`, chỉ `CachedStorage` override) → biết LLR đang giữ bao nhiêu entry. - `GraphIndex::mem_breakdown()` gom tất cả lại, `ranked()` sort giảm dần để thứ tự tối ưu là thứ tự xếp hạng. - `bin/mem`: in bảng breakdown + JSON (`breakdown`, `accounted_total`, `caches`). Chưa đổi hành vi production: các method mới đều chỉ đọc, không vào hot path. --- crates/codegraph-bench/src/bin/mem.rs | 97 ++++++ crates/codegraph-graph/src/lib.rs | 25 ++ crates/codegraph-graph/src/lru.rs | 10 + crates/codegraph-graph/src/memtrack.rs | 310 +++++++++++++++++++ crates/codegraph-graph/src/storage.rs | 8 + crates/codegraph-graph/src/storage/cached.rs | 30 +- 6 files changed, 479 insertions(+), 1 deletion(-) create mode 100644 crates/codegraph-graph/src/memtrack.rs diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index a9acfc6737..33cd11dcd9 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -14,11 +14,90 @@ use camino::Utf8Path; use clap::Parser; +use std::sync::OnceLock; use codegraph_bench::{BenchOptions, Repo, extract, index_at, orchestrator, run_queries}; use codegraph_core::{ Annotation, CallRecord, EdgeMeta, EffectType, ScopeLevel, Symbol, SymbolKind, }; use codegraph_graph::meminfo::{MemTracker, fmt_bytes, rss_bytes}; +use codegraph_graph::memtrack::MemBreakdown; + +fn runtime() -> &'static tokio::runtime::Runtime { + static RT: OnceLock = OnceLock::new(); + RT.get_or_init(|| { + tokio::runtime::Builder::new_multi_thread() + .enable_all() + .build() + .expect("dựng tokio runtime") + }) +} + +/// Một dòng breakdown — JSON-friendly (thứ tự giữ nguyên như `ranked()`). +#[derive(serde::Serialize)] +struct BreakdownRow { + structure: String, + entries: u64, + fixed_bytes: u64, + heap_bytes: u64, + total_bytes: u64, +} + +fn breakdown_rows(b: &MemBreakdown) -> Vec { + b.ranked() + .into_iter() + .map(|(name, m)| BreakdownRow { + structure: name.to_string(), + entries: m.entries, + fixed_bytes: m.fixed_bytes, + heap_bytes: m.heap_bytes, + total_bytes: m.total_bytes(), + }) + .collect() +} + +/// In breakdown cấu trúc (đã sort giảm dần) — phần trả lời câu hỏi "RAM nằm ở +/// đâu", tách khỏi RSS tổng. +fn print_breakdown(rows: &[BreakdownRow], caches: &[(String, usize)], rss_index: u64) { + println!("\n {:<22} {:>9} {:>12} {:>12} {:>12}", "structure", "entries", "fixed", "heap", "total"); + let mut accounted = 0u64; + for row in rows { + if row.entries == 0 && row.total_bytes == 0 { + continue; + } + accounted += row.total_bytes; + println!( + " {:<22} {:>9} {:>12} {:>12} {:>12}", + row.structure, + row.entries, + fmt_bytes(row.fixed_bytes), + fmt_bytes(row.heap_bytes), + fmt_bytes(row.total_bytes) + ); + } + println!( + " {:<22} {:>9} {:>12} {:>12} {:>12}", + "SUM accounted", + "", + "", + "", + fmt_bytes(accounted) + ); + if rss_index > 0 { + let pct = accounted as f64 * 100.0 / rss_index as f64; + println!(" accounted / rss index = {pct:.1}% — phần còn lại: allocator + radix engine"); + } + let busy: Vec<&(String, usize)> = caches.iter().filter(|(_, n)| *n > 0).collect(); + if !busy.is_empty() { + print!(" LRU cache đang dùng: "); + println!( + "{}", + busy.iter() + .map(|(n, v)| format!("{n}={v}")) + .collect::>() + .join(" ") + ); + } +} #[derive(Parser)] #[command(name = "codegraph-mem", about = "Đo RAM của codegraph theo từng phase")] @@ -76,6 +155,12 @@ struct RepoMem { predicted_symbols: u64, /// `edges.len() × size_of::()` — phần trong HashMap `edges`. predicted_edges: u64, + /// Deep size từng cấu trúc, sort giảm dần. + breakdown: Vec, + /// Tổng bytes đã quy được về cấu trúc (chưa gồm allocator + radix engine). + accounted_total: u64, + /// LRU cache đang giữ entry (tên → số entry). + caches: Vec<(String, usize)>, } fn load_repos(cli: &Cli) -> Vec { @@ -133,6 +218,7 @@ fn measure(repo: &Repo, opts: &BenchOptions) -> anyhow::Result { tracker.mark("query"); let st = idx.stats(); + let breakdown = runtime().block_on(idx.mem_breakdown()); let sample = |label: &str| { tracker .samples() @@ -157,6 +243,9 @@ fn measure(repo: &Repo, opts: &BenchOptions) -> anyhow::Result { rss_index_delta: after_index.saturating_sub(after_extract), predicted_symbols: st.symbols * size_of::() as u64, predicted_edges: st.edges * size_of::() as u64, + accounted_total: breakdown.accounted_total(), + caches: breakdown.caches.clone(), + breakdown: breakdown_rows(&breakdown), }) } @@ -241,6 +330,7 @@ fn measure_synthetic(n: usize, fanout: usize) -> anyhow::Result { tracker.mark("index"); let st = idx.stats(); + let breakdown = runtime().block_on(idx.mem_breakdown()); let sample = |label: &str| { tracker .samples() @@ -265,6 +355,9 @@ fn measure_synthetic(n: usize, fanout: usize) -> anyhow::Result { rss_index_delta: after_index.saturating_sub(after_extract), predicted_symbols: st.symbols * size_of::() as u64, predicted_edges: st.edges * size_of::() as u64, + accounted_total: breakdown.accounted_total(), + caches: breakdown.caches.clone(), + breakdown: breakdown_rows(&breakdown), }) } @@ -327,6 +420,10 @@ fn main() -> anyhow::Result<()> { fmt_bytes(r.predicted_edges), ); } + for r in &results { + println!("\n=== breakdown: {} ===", r.repo); + print_breakdown(&r.breakdown, &r.caches, r.rss_after_index); + } if let Some(peak) = codegraph_graph::meminfo::peak_rss_bytes() { println!("\npeak RSS (VmHWM): {}", fmt_bytes(peak)); } else { diff --git a/crates/codegraph-graph/src/lib.rs b/crates/codegraph-graph/src/lib.rs index a111ee6350..fa0f536537 100644 --- a/crates/codegraph-graph/src/lib.rs +++ b/crates/codegraph-graph/src/lib.rs @@ -78,6 +78,7 @@ pub mod diff; pub mod embeddings; mod lru; pub mod meminfo; +pub mod memtrack; mod radix; mod search; mod shared; @@ -2653,6 +2654,30 @@ impl GraphIndex { }) } + /// Deep size từng cấu trúc in-memory + occupancy LRU cache. + /// + /// Dùng để **quy kết quả RSS về đúng cấu trúc**: `GraphIndex` in-memory-first + /// nên tổng bytes ở đây mới là thứ quyết định repo lớn tốn bao nhiêu RAM. + /// Xem [`crate::memtrack`] để biết phần nào exact và phần nào ước lượng. + /// + /// Chạy O(tổng số entry) — dùng cho profiler/diagnostics, không phải + /// hot path. + pub async fn mem_breakdown(&self) -> crate::memtrack::MemBreakdown { + use crate::memtrack as mt; + let caches = self.storage.read().await.cache_occupancy(); + mt::MemBreakdown { + symbols: mt::symbols_mem(&self.symbols), + chains_map: mt::chains_map_mem(&self.chains_map), + call_names: mt::call_names_mem(&self.call_names), + edges: mt::edges_mem(&self.edges), + name_index: mt::name_index_mem(&self.name_index), + scope_index: mt::scope_index_mem(&self.scope_index), + name_keys: mt::name_keys_mem(&self.name_records, &self.sorted_name_keys), + files: mt::files_mem(&self.files), + caches, + } + } + /// Số liệu tổng hợp. pub fn stats(&self) -> SemgraphStats { SemgraphStats { diff --git a/crates/codegraph-graph/src/lru.rs b/crates/codegraph-graph/src/lru.rs index 4a1447d67a..91a598c273 100644 --- a/crates/codegraph-graph/src/lru.rs +++ b/crates/codegraph-graph/src/lru.rs @@ -194,6 +194,16 @@ where self.caching[index].value.clone() } + /// Số entry đang cache — `DashMap::len()` O(1), dùng cho báo cáo memory. + pub fn len(&self) -> usize { + self.mapping.len() + } + + /// `true` khi chưa cache entry nào. + pub fn is_empty(&self) -> bool { + self.mapping.is_empty() + } + /// Xoá toàn bộ entry (dùng khi invalidate hàng loạt, VD sau transaction /// commit hoặc `clear_*` của storage). Reset cả arena lẫn linked-list. pub fn clear(&self) { diff --git a/crates/codegraph-graph/src/memtrack.rs b/crates/codegraph-graph/src/memtrack.rs new file mode 100644 index 0000000000..6c160e8aa2 --- /dev/null +++ b/crates/codegraph-graph/src/memtrack.rs @@ -0,0 +1,310 @@ +//! Quy kết quả RSS về **đúng cấu trúc dữ liệu nào** đang nuốt RAM. +//! +//! [`super::meminfo`] đo tổng RSS của process — nhưng con số đó không bảo được +//! `HashMap` nào cần sửa. Module này cộng **deep size** từng cấu trúc trong +//! `GraphIndex` để so `Σ accounted` với `RSS thực`. +//! +//! ## Độ chính xác +//! +//! - **Exact**: heap của `String`/`Vec` (dùng `capacity()` — đúng bytes đã cấp). +//! - **Ước lượng**: bucket array của `HashMap` dùng công thức hashbrown +//! `capacity() × (size_of::<(K, V)>() + 1)` (1 byte control mỗi slot). +//! - **Không tính**: `HashMap` bên trong `Annotation::args` (nhỏ, và capacity +//! không đọc được từ `&HashMap`), cùng overhead của allocator (jemalloc/malloc +//! thường ~10–20% bytes đã cấp). Vì vậy `Σ accounted` **luôn nhỏ hơn RSS +//! thực** — đó là bình thường, không phải bug. +//! +//! Dùng để **so tương đối** (cấu trúc nào phình theo quy mô), không dùng làm +//! ngưỡng cứng. + +use std::collections::HashMap; +use std::hash::Hash; + +use codegraph_core::{Annotation, CallSite, EdgeMeta, FileInfo, Symbol}; + +/// Một cấu trúc: số phần tử + bytes đã cấp (fixed + heap). +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct StructureMem { + /// Số phần tử (entry của map, phần tử của vec). + pub entries: u64, + /// Bytes cố định: bucket của `HashMap` hoặc `len × size_of::()` của `Vec`. + pub fixed_bytes: u64, + /// Bytes cấp động: `String`/`Vec` heap bên trong phần tử. + pub heap_bytes: u64, +} + +impl StructureMem { + pub fn total_bytes(&self) -> u64 { + self.fixed_bytes.saturating_add(self.heap_bytes) + } +} + +/// Bytes của `String` (đã cấp trên heap). +#[inline] +fn str_bytes(s: &str) -> u64 { + s.capacity() as u64 +} + +/// Bytes của `Option`. +#[inline] +fn opt_str_bytes(s: &Option) -> u64 { + s.as_ref().map_or(0, str_bytes) +} + +/// Bytes của `Vec`: buffer + phần tử. +#[inline] +fn u64_vec_bytes(v: &[u64]) -> u64 { + (v.capacity() * size_of::()) as u64 +} + +/// Bucket array của `HashMap` theo công thức hashbrown. +#[inline] +fn map_bucket_bytes(map: &HashMap) -> u64 { + // `(size_of::<(K, V)>() + 1)` — +1 là byte control/sentinel mỗi slot. + let slot = size_of::<(K, V)>() as u64 + 1; + (map.capacity() as u64).saturating_mul(slot) +} + +/// Bytes của một `Vec` (buffer đã cấp). +#[inline] +fn vec_bytes(v: &[T]) -> u64 { + (v.capacity() * size_of::()) as u64 +} + +// ── Deep size từng phần tử ── + +/// Deep size của `Symbol` (chưa tính slot trong HashMap). +pub fn symbol_heap(sym: &Symbol) -> u64 { + let annotations: u64 = sym + .annotations + .iter() + .map(|a| annotation_heap(a)) + .sum::() + + vec_bytes::(&sym.annotations); + str_bytes(&sym.name) + + str_bytes(&sym.file) + + opt_str_bytes(&sym.type_name) + + opt_str_bytes(&sym.signature) + + opt_str_bytes(&sym.doc) + + annotations +} + +/// Deep size của `Annotation` (chưa tính buffer `Vec` chứa nó). +fn annotation_heap(a: &Annotation) -> u64 { + // `args` là HashMap — không đọc được capacity từ `&HashMap`, bỏ qua. + str_bytes(&a.name) +} + +/// Deep size của `CallSite` (chưa tính slot trong HashMap). +pub fn call_site_heap(site: &CallSite) -> u64 { + str_bytes(&site.call_name) + + opt_str_bytes(&site.condition) + + vec_bytes::(&site.arg_exprs) + + site.arg_exprs.iter().map(|a| str_bytes(a)).sum::() +} + +/// Deep size của `EdgeMeta` (chưa tính slot trong HashMap). +pub fn edge_meta_heap(m: &EdgeMeta) -> u64 { + opt_str_bytes(&m.condition) + opt_str_bytes(&m.effect_desc) + u64_vec_bytes(&m.arg_ids) +} + +// ── Tổng hợp theo cấu trúc ── + +/// RAM của `symbols: HashMap`. +pub fn symbols_mem(m: &HashMap) -> StructureMem { + StructureMem { + entries: m.len() as u64, + fixed_bytes: map_bucket_bytes(m) + + (m.len() as u64).saturating_mul(size_of::() as u64), + heap_bytes: m.values().map(symbol_heap).sum(), + } +} + +/// RAM của `chains_map: HashMap>`. +pub fn chains_map_mem(m: &HashMap>) -> StructureMem { + StructureMem { + entries: m.len() as u64, + fixed_bytes: map_bucket_bytes(m), + heap_bytes: m.values().map(|v| u64_vec_bytes(v)).sum(), + } +} + +/// RAM của `call_names: HashMap>`. +pub fn call_names_mem(m: &HashMap>) -> StructureMem { + StructureMem { + entries: m.len() as u64, + fixed_bytes: map_bucket_bytes(m) + + (m.len() as u64).saturating_mul(size_of::>() as u64), + heap_bytes: m + .iter() + .map(|(k, v)| { + str_bytes(k) + + vec_bytes::(v) + + v.iter().map(call_site_heap).sum::() + }) + .sum(), + } +} + +/// RAM của `edges: HashMap<(u64, u64), EdgeMeta>`. +pub fn edges_mem(m: &HashMap<(u64, u64), EdgeMeta>) -> StructureMem { + StructureMem { + entries: m.len() as u64, + fixed_bytes: map_bucket_bytes(m), + heap_bytes: m.values().map(edge_meta_heap).sum(), + } +} + +/// RAM của `name_index: HashMap>` (key + buffer, không tính +/// phần tử `u64` vì đã nằm trong `u64_vec_bytes` của từng value). +pub fn name_index_mem(m: &HashMap>) -> StructureMem { + StructureMem { + entries: m.len() as u64, + fixed_bytes: map_bucket_bytes(m), + heap_bytes: m + .iter() + .map(|(k, v)| str_bytes(k) + u64_vec_bytes(v)) + .sum(), + } +} + +/// RAM của `scope_index: HashMap>`. +pub fn scope_index_mem(m: &HashMap>) -> StructureMem { + chains_map_mem(m) +} + +/// RAM của `name_records: Vec` + `sorted_name_keys: Vec`. +pub fn name_keys_mem(records: &[String], sorted: &[String]) -> StructureMem { + StructureMem { + entries: (records.len() + sorted.len()) as u64, + fixed_bytes: vec_bytes::(records) + vec_bytes::(sorted), + heap_bytes: records.iter().map(|s| str_bytes(s)).sum::() + + sorted.iter().map(|s| str_bytes(s)).sum::(), + } +} + +/// RAM của `files: Vec`. +pub fn files_mem(files: &[FileInfo]) -> StructureMem { + StructureMem { + entries: files.len() as u64, + fixed_bytes: vec_bytes::(files), + heap_bytes: files.iter().map(|f| str_bytes(&f.path)).sum(), + } +} + +/// Toàn bộ breakdown của một `GraphIndex`. +#[derive(Debug, Clone, Default)] +pub struct MemBreakdown { + pub symbols: StructureMem, + pub chains_map: StructureMem, + pub call_names: StructureMem, + pub edges: StructureMem, + pub name_index: StructureMem, + pub scope_index: StructureMem, + pub name_keys: StructureMem, + pub files: StructureMem, + /// Occupancy của LRU cache phía trên storage: `(tên, số entry)`. + pub caches: Vec<(String, usize)>, +} + +impl MemBreakdown { + /// Tổng bytes đã tính được (chưa gồm overhead allocator). + pub fn accounted_total(&self) -> u64 { + [ + self.symbols.total_bytes(), + self.chains_map.total_bytes(), + self.call_names.total_bytes(), + self.edges.total_bytes(), + self.name_index.total_bytes(), + self.scope_index.total_bytes(), + self.name_keys.total_bytes(), + self.files.total_bytes(), + ] + .into_iter() + .sum() + } + + /// Các cấu trúc xếp theo tổng bytes giảm dần — thứ tự nên tối ưu. + pub fn ranked(&self) -> Vec<(&'static str, StructureMem)> { + let mut v = vec![ + ("symbols", self.symbols), + ("chains_map", self.chains_map), + ("call_names", self.call_names), + ("edges", self.edges), + ("name_index", self.name_index), + ("scope_index", self.scope_index), + ("name_keys", self.name_keys), + ("files", self.files), + ]; + v.sort_by(|a, b| b.1.total_bytes().cmp(&a.1.total_bytes())); + v + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn symbol_heap_counts_strings() { + let sym = Symbol { + id: 100, + name: "foo".into(), + kind: codegraph_core::SymbolKind::Function, + scope: codegraph_core::ScopeLevel::Global, + scope_id: 0, + type_ref: 0, + type_name: None, + file: "a.rs".into(), + line: 1, + end_line: 1, + signature: None, + doc: None, + annotations: Vec::new(), + language: "rust".into(), + }; + let mem = symbols_mem(&HashMap::from([(100, sym)])); + assert_eq!(mem.entries, 1); + assert!(mem.heap_bytes >= 6, "phải tính name + file: {}", mem.heap_bytes); + } + + #[test] + fn call_names_heap_includes_args() { + let site = CallSite { + caller_id: 1, + call_name: "lib::helper_0".into(), + line: 3, + condition: Some("x > 0".into()), + is_loop_body: false, + arg_exprs: vec!["a".into(), "b".into()], + }; + let mem = call_names_mem(&HashMap::from([( + "lib::helper_0".to_string(), + vec![site], + )])); + assert_eq!(mem.entries, 1); + // key + call_name + condition + 2 args + Vec buffer. + assert!(mem.heap_bytes > 30, "heap quá nhỏ: {}", mem.heap_bytes); + } + + #[test] + fn ranked_is_sorted_desc() { + let b = MemBreakdown { + symbols: StructureMem { + entries: 1, + fixed_bytes: 0, + heap_bytes: 100, + }, + edges: StructureMem { + entries: 1, + fixed_bytes: 0, + heap_bytes: 10, + }, + ..Default::default() + }; + let r = b.ranked(); + assert_eq!(r[0].0, "symbols"); + assert_eq!(r[1].0, "edges"); + assert_eq!(b.accounted_total(), 110); + } +} diff --git a/crates/codegraph-graph/src/storage.rs b/crates/codegraph-graph/src/storage.rs index 68d13fb6ac..8e5fdfd21e 100644 --- a/crates/codegraph-graph/src/storage.rs +++ b/crates/codegraph-graph/src/storage.rs @@ -550,4 +550,12 @@ pub trait Storage: + Send + Sync { + /// Occupancy của LRU cache phía trên backend: `(tên cache, số entry)`. + /// + /// Mặc định `[]` — backend không cache thì không có gì để báo. Chỉ + /// [`CachedStorage`](crate::storage::cached::CachedStorage) override. + /// Dùng cho báo cáo memory (`GraphIndex::mem_breakdown`). + fn cache_occupancy(&self) -> Vec<(String, usize)> { + Vec::new() + } } diff --git a/crates/codegraph-graph/src/storage/cached.rs b/crates/codegraph-graph/src/storage/cached.rs index fffbc6967d..8adb41b202 100644 --- a/crates/codegraph-graph/src/storage/cached.rs +++ b/crates/codegraph-graph/src/storage/cached.rs @@ -54,6 +54,30 @@ struct CacheSet { } impl CacheSet { + /// Occupancy từng cache — `(tên, số entry đang cache)`. Dùng để báo cáo + /// memory: LRU pre-allocate cả arena `capacity` entry ngay khi `new`, nên + /// **bytes đã cấp là hằng số**; entry đang dùng mới là phần biến động. + fn occupancy(&self) -> Vec<(String, usize)> { + [ + ("nodes", self.nodes.len()), + ("children", self.children.len()), + ("chains", self.chains.len()), + ("metas", self.metas.len()), + ("key_lens", self.key_lens.len()), + ("edge_data", self.edge_data.len()), + ("node_meta", self.node_meta.len()), + ("roots", self.roots.len()), + ("shortcuts", self.shortcuts.len()), + ("symbols", self.symbols.len()), + ("embeddings", self.embeddings.len()), + ("call_records", self.call_records.len()), + ("call_name_index", self.call_name_index.len()), + ] + .into_iter() + .map(|(name, len)| (name.to_string(), len)) + .collect() + } + fn new(capacity: usize) -> Self { Self { nodes: LruCache::new(capacity), @@ -510,7 +534,11 @@ impl EntityStorage for CachedStorage { // Rust tự cộng method qua blanket bound — không cần viết gì thêm. #[async_trait] -impl Storage for CachedStorage {} +impl Storage for CachedStorage { + fn cache_occupancy(&self) -> Vec<(String, usize)> { + self.caches.occupancy() + } +} /// Tx bọc: delegate mọi mutation, khi `commit` xong thì `clear_radix()`. struct CachedTx { From 960222f6262eef6ff46f4467d17e6e099a5b6506 Mon Sep 17 00:00:00 2001 From: hungpham10 <136320753+hungpham10@users.noreply.github.com> Date: Mon, 5 Oct 2026 01:49:50 +0000 Subject: [PATCH 07/15] style: apply rustfmt --- crates/codegraph-bench/src/bin/mem.rs | 7 +++++-- crates/codegraph-graph/src/memtrack.rs | 20 ++++++++------------ 2 files changed, 13 insertions(+), 14 deletions(-) diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index 33cd11dcd9..17c11dc6aa 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -14,13 +14,13 @@ use camino::Utf8Path; use clap::Parser; -use std::sync::OnceLock; use codegraph_bench::{BenchOptions, Repo, extract, index_at, orchestrator, run_queries}; use codegraph_core::{ Annotation, CallRecord, EdgeMeta, EffectType, ScopeLevel, Symbol, SymbolKind, }; use codegraph_graph::meminfo::{MemTracker, fmt_bytes, rss_bytes}; use codegraph_graph::memtrack::MemBreakdown; +use std::sync::OnceLock; fn runtime() -> &'static tokio::runtime::Runtime { static RT: OnceLock = OnceLock::new(); @@ -58,7 +58,10 @@ fn breakdown_rows(b: &MemBreakdown) -> Vec { /// In breakdown cấu trúc (đã sort giảm dần) — phần trả lời câu hỏi "RAM nằm ở /// đâu", tách khỏi RSS tổng. fn print_breakdown(rows: &[BreakdownRow], caches: &[(String, usize)], rss_index: u64) { - println!("\n {:<22} {:>9} {:>12} {:>12} {:>12}", "structure", "entries", "fixed", "heap", "total"); + println!( + "\n {:<22} {:>9} {:>12} {:>12} {:>12}", + "structure", "entries", "fixed", "heap", "total" + ); let mut accounted = 0u64; for row in rows { if row.entries == 0 && row.total_bytes == 0 { diff --git a/crates/codegraph-graph/src/memtrack.rs b/crates/codegraph-graph/src/memtrack.rs index 6c160e8aa2..ddbbfcfd85 100644 --- a/crates/codegraph-graph/src/memtrack.rs +++ b/crates/codegraph-graph/src/memtrack.rs @@ -138,9 +138,7 @@ pub fn call_names_mem(m: &HashMap>) -> StructureMem { heap_bytes: m .iter() .map(|(k, v)| { - str_bytes(k) - + vec_bytes::(v) - + v.iter().map(call_site_heap).sum::() + str_bytes(k) + vec_bytes::(v) + v.iter().map(call_site_heap).sum::() }) .sum(), } @@ -161,10 +159,7 @@ pub fn name_index_mem(m: &HashMap>) -> StructureMem { StructureMem { entries: m.len() as u64, fixed_bytes: map_bucket_bytes(m), - heap_bytes: m - .iter() - .map(|(k, v)| str_bytes(k) + u64_vec_bytes(v)) - .sum(), + heap_bytes: m.iter().map(|(k, v)| str_bytes(k) + u64_vec_bytes(v)).sum(), } } @@ -265,7 +260,11 @@ mod tests { }; let mem = symbols_mem(&HashMap::from([(100, sym)])); assert_eq!(mem.entries, 1); - assert!(mem.heap_bytes >= 6, "phải tính name + file: {}", mem.heap_bytes); + assert!( + mem.heap_bytes >= 6, + "phải tính name + file: {}", + mem.heap_bytes + ); } #[test] @@ -278,10 +277,7 @@ mod tests { is_loop_body: false, arg_exprs: vec!["a".into(), "b".into()], }; - let mem = call_names_mem(&HashMap::from([( - "lib::helper_0".to_string(), - vec![site], - )])); + let mem = call_names_mem(&HashMap::from([("lib::helper_0".to_string(), vec![site])])); assert_eq!(mem.entries, 1); // key + call_name + condition + 2 args + Vec buffer. assert!(mem.heap_bytes > 30, "heap quá nhỏ: {}", mem.heap_bytes); From a0ed8415d9cbe0d8bc479c74d008385ee252652c Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 09:31:36 +0700 Subject: [PATCH 08/15] =?UTF-8?q?fix(graph):=20memtrack=20=E2=80=94=20`&st?= =?UTF-8?q?r`/`&[T]`=20kh=C3=B4ng=20c=C3=B3=20`.capacity()`?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CI báo 4 lỗi E0599 + 1 E0631: helper deep-size nhận `&str`/`&[T]` nhưng gọi `.capacity()` — method đó chỉ có trên `String`/`Vec`. Đổi helper sang `&String`/`&Vec` và thêm `#[allow(clippy::ptr_arg)]` kèm lý do: cần bytes **đã cấp** (`capacity`), slice không đọc được — dùng `len` sẽ lệch tới ~2× do `Vec` over-allocate. --- crates/codegraph-graph/src/memtrack.rs | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/crates/codegraph-graph/src/memtrack.rs b/crates/codegraph-graph/src/memtrack.rs index ddbbfcfd85..7dd4446b7d 100644 --- a/crates/codegraph-graph/src/memtrack.rs +++ b/crates/codegraph-graph/src/memtrack.rs @@ -40,20 +40,24 @@ impl StructureMem { } /// Bytes của `String` (đã cấp trên heap). +/// +/// Nhận `&String` chứ không phải `&str`: `str` không có `capacity()`. +#[allow(clippy::ptr_arg)] // xem `vec_bytes` — cần `capacity()` của `String`. #[inline] -fn str_bytes(s: &str) -> u64 { +fn str_bytes(s: &String) -> u64 { s.capacity() as u64 } /// Bytes của `Option`. #[inline] fn opt_str_bytes(s: &Option) -> u64 { - s.as_ref().map_or(0, str_bytes) + s.as_ref().map_or(0, |x| x.capacity() as u64) } /// Bytes của `Vec`: buffer + phần tử. +#[allow(clippy::ptr_arg)] // xem `vec_bytes` — cần `capacity()` của `Vec`. #[inline] -fn u64_vec_bytes(v: &[u64]) -> u64 { +fn u64_vec_bytes(v: &Vec) -> u64 { (v.capacity() * size_of::()) as u64 } @@ -66,8 +70,11 @@ fn map_bucket_bytes(map: &HashMap) -> u64 { } /// Bytes của một `Vec` (buffer đã cấp). +// `&Vec` thay vì `&[T]`: `capacity()` là method riêng của `Vec`, cần biết +// bytes **đã cấp** chứ không phải `len` — mà slice không đọc được. +#[allow(clippy::ptr_arg)] #[inline] -fn vec_bytes(v: &[T]) -> u64 { +fn vec_bytes(v: &Vec) -> u64 { (v.capacity() * size_of::()) as u64 } @@ -169,7 +176,8 @@ pub fn scope_index_mem(m: &HashMap>) -> StructureMem { } /// RAM của `name_records: Vec` + `sorted_name_keys: Vec`. -pub fn name_keys_mem(records: &[String], sorted: &[String]) -> StructureMem { +#[allow(clippy::ptr_arg)] // xem `vec_bytes` — cần `capacity()` của `Vec`. +pub fn name_keys_mem(records: &Vec, sorted: &Vec) -> StructureMem { StructureMem { entries: (records.len() + sorted.len()) as u64, fixed_bytes: vec_bytes::(records) + vec_bytes::(sorted), @@ -179,7 +187,8 @@ pub fn name_keys_mem(records: &[String], sorted: &[String]) -> StructureMem { } /// RAM của `files: Vec`. -pub fn files_mem(files: &[FileInfo]) -> StructureMem { +#[allow(clippy::ptr_arg)] // xem `vec_bytes` — cần `capacity()` của `Vec`. +pub fn files_mem(files: &Vec) -> StructureMem { StructureMem { entries: files.len() as u64, fixed_bytes: vec_bytes::(files), From 0cadb586ac70bde8228f17b5e21b474e4296a8b6 Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 10:45:53 +0700 Subject: [PATCH 09/15] =?UTF-8?q?fix(graph):=20s=E1=BB=ADa=20clippy=20?= =?UTF-8?q?=E2=80=94=20dead=5Fcode=20`is=5Fempty`,=204=20redundant=5Fclosu?= =?UTF-8?q?re,=20sort=5Fby?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - `LruCache::is_empty` không nơi nào gọi → bỏ (chỉ cần `len` cho occupancy). - `.map(|x| f(x))` → `.map(f)` ở 4 chỗ trong `memtrack`. - `ranked()` dùng `sort_unstable_by` + tiebreaker theo tên: vừa tránh `stable_sort_primitive`, vừa giữ thứ tự **deterministic** khi hai cấu trúc trùng bytes (sort thuần tuỳ thứ tự ban đầu → báo cáo khác nhau giữa 2 lần chạy). --- crates/codegraph-graph/src/lru.rs | 5 ----- crates/codegraph-graph/src/memtrack.rs | 14 ++++++++------ 2 files changed, 8 insertions(+), 11 deletions(-) diff --git a/crates/codegraph-graph/src/lru.rs b/crates/codegraph-graph/src/lru.rs index 91a598c273..f2a2581b32 100644 --- a/crates/codegraph-graph/src/lru.rs +++ b/crates/codegraph-graph/src/lru.rs @@ -199,11 +199,6 @@ where self.mapping.len() } - /// `true` khi chưa cache entry nào. - pub fn is_empty(&self) -> bool { - self.mapping.is_empty() - } - /// Xoá toàn bộ entry (dùng khi invalidate hàng loạt, VD sau transaction /// commit hoặc `clear_*` của storage). Reset cả arena lẫn linked-list. pub fn clear(&self) { diff --git a/crates/codegraph-graph/src/memtrack.rs b/crates/codegraph-graph/src/memtrack.rs index 7dd4446b7d..c1390f2ecf 100644 --- a/crates/codegraph-graph/src/memtrack.rs +++ b/crates/codegraph-graph/src/memtrack.rs @@ -85,7 +85,7 @@ pub fn symbol_heap(sym: &Symbol) -> u64 { let annotations: u64 = sym .annotations .iter() - .map(|a| annotation_heap(a)) + .map(annotation_heap) .sum::() + vec_bytes::(&sym.annotations); str_bytes(&sym.name) @@ -107,7 +107,7 @@ pub fn call_site_heap(site: &CallSite) -> u64 { str_bytes(&site.call_name) + opt_str_bytes(&site.condition) + vec_bytes::(&site.arg_exprs) - + site.arg_exprs.iter().map(|a| str_bytes(a)).sum::() + + site.arg_exprs.iter().map(str_bytes).sum::() } /// Deep size của `EdgeMeta` (chưa tính slot trong HashMap). @@ -132,7 +132,7 @@ pub fn chains_map_mem(m: &HashMap>) -> StructureMem { StructureMem { entries: m.len() as u64, fixed_bytes: map_bucket_bytes(m), - heap_bytes: m.values().map(|v| u64_vec_bytes(v)).sum(), + heap_bytes: m.values().map(u64_vec_bytes).sum(), } } @@ -181,8 +181,8 @@ pub fn name_keys_mem(records: &Vec, sorted: &Vec) -> StructureMe StructureMem { entries: (records.len() + sorted.len()) as u64, fixed_bytes: vec_bytes::(records) + vec_bytes::(sorted), - heap_bytes: records.iter().map(|s| str_bytes(s)).sum::() - + sorted.iter().map(|s| str_bytes(s)).sum::(), + heap_bytes: records.iter().map(str_bytes).sum::() + + sorted.iter().map(str_bytes).sum::(), } } @@ -240,7 +240,9 @@ impl MemBreakdown { ("name_keys", self.name_keys), ("files", self.files), ]; - v.sort_by(|a, b| b.1.total_bytes().cmp(&a.1.total_bytes())); + // `sort_unstable_by` + tiebreaker theo tên → thứ tự **hoàn toàn xác định** + // (không phụ thuộc thứ tự ban đầu khi hai cấu trúc bằng bytes). + v.sort_unstable_by(|a, b| b.1.total_bytes().cmp(&a.1.total_bytes()).then(a.0.cmp(b.0))); v } } From 24abb5601ba0aef17baa49f7f690f9ec735557a4 Mon Sep 17 00:00:00 2001 From: hungpham10 <136320753+hungpham10@users.noreply.github.com> Date: Mon, 5 Oct 2026 04:05:48 +0000 Subject: [PATCH 10/15] style: apply rustfmt --- crates/codegraph-graph/src/memtrack.rs | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/crates/codegraph-graph/src/memtrack.rs b/crates/codegraph-graph/src/memtrack.rs index c1390f2ecf..61b68bb603 100644 --- a/crates/codegraph-graph/src/memtrack.rs +++ b/crates/codegraph-graph/src/memtrack.rs @@ -82,11 +82,7 @@ fn vec_bytes(v: &Vec) -> u64 { /// Deep size của `Symbol` (chưa tính slot trong HashMap). pub fn symbol_heap(sym: &Symbol) -> u64 { - let annotations: u64 = sym - .annotations - .iter() - .map(annotation_heap) - .sum::() + let annotations: u64 = sym.annotations.iter().map(annotation_heap).sum::() + vec_bytes::(&sym.annotations); str_bytes(&sym.name) + str_bytes(&sym.file) From 27f16e72fc0518b81b3038e3a848f2669bf2d26d Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 11:59:00 +0700 Subject: [PATCH 11/15] =?UTF-8?q?perf(bench):=20=C4=91o=20vi=20sai=20radix?= =?UTF-8?q?=20engine=20b=E1=BA=B1ng=203=20shape=20synthetic?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Breakdown hiện tại chỉ tính HashMap của `GraphIndex` (97.5 MiB / 320 MiB Δ). ~223 MiB còn lại nằm ở hai radix engine — sống trong `InMemoryStorage` chứ không phải HashMap nên `mem_breakdown` không thấy. Cách đo: dựng cùng 50k symbol nhưng bỏ dần từng thành phần, lấy hiệu RSS. - `symbols` — chỉ symbol → `symbols` + `name_index` + **name engine** - `chains` — + chain, không call record → **chain engine** (trie + rt_shortcuts) - `full` — + call record → `call_names` + `edges` Không đụng `Storage` trait nên không có rủi ro 6 backend; đổi hoàn toàn ở tầng benchmark. Kết quả quyết định có nên tối ưu `call_names` (30 MiB) hay không. --- .github/workflows/memory.yml | 19 +++++++ crates/codegraph-bench/src/bin/mem.rs | 82 ++++++++++++++++++++------- 2 files changed, 79 insertions(+), 22 deletions(-) diff --git a/.github/workflows/memory.yml b/.github/workflows/memory.yml index ea222b988c..880bd6270b 100644 --- a/.github/workflows/memory.yml +++ b/.github/workflows/memory.yml @@ -48,6 +48,21 @@ jobs: ./target/release/mem --synthetic 50000 --fanout 4 \ --json | tee mem-out/synthetic-50k.json + # Đo VI SAI: cùng 50k symbol nhưng bỏ dần từng thành phần. Hiệu giữa các + # shape = chi phí của phần vừa bị bỏ, **gồm cả radix engine** — phần mà + # `mem_breakdown` (chỉ tính HashMap của GraphIndex) không quy được. + # symbols → chains: + chain engine (trie + rt_shortcuts) + # chains → full: + call_names + edges + - name: Synthetic differential shapes + run: | + : > mem-out/shapes.txt + for shape in symbols chains full; do + ./target/release/mem --synthetic 50000 --fanout 4 --shape "$shape" \ + | tee -a mem-out/shapes.txt + ./target/release/mem --synthetic 50000 --fanout 4 --shape "$shape" --json \ + >> mem-out/shapes.json + done + - name: Real repos env: CODEGRAPH_BENCH_REPOS_LIST: ${{ github.workspace }}/.github/benches/repos/list.txt @@ -68,6 +83,10 @@ jobs: echo '```' >> "$GITHUB_STEP_SUMMARY" cat mem-out/repos.txt >> "$GITHUB_STEP_SUMMARY" echo '```' >> "$GITHUB_STEP_SUMMARY" + echo '### Differential shapes (50k)' >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" + cat mem-out/shapes.txt >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" echo '```json' >> "$GITHUB_STEP_SUMMARY" cat mem-out/synthetic-50k.json >> "$GITHUB_STEP_SUMMARY" echo '```' >> "$GITHUB_STEP_SUMMARY" diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index 17c11dc6aa..710950cfbe 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -132,8 +132,36 @@ struct Cli { synthetic: Option, /// Số callee mỗi function trong chế độ `--synthetic`. - #[arg(long, default_value_t = 4, value_name = "K")] - fanout: usize, + #[arg(long, default_value_t = 4, value_name = "K")] + fanout: usize, + + /// Thành phần nào của index synthetic cần dựng — dùng để **đo vi sai**: + /// mỗi shape thiếu một phần, hiệu RSS cho ra chi phí của phần đó (gồm cả + /// radix engine mà `mem_breakdown` chưa tính). + /// + /// - `symbols` — chỉ symbol → `symbols` + `name_index` + **name engine** + /// - `chains` — symbol + chain, không call record → thêm **chain engine** + /// - `full` — kèm call record → thêm `call_names` + `edges` + #[arg(long, value_enum, default_value_t = Shape::Full)] + shape: Shape, +} + +#[derive(Clone, Copy, PartialEq, Eq, clap::ValueEnum)] +enum Shape { + Symbols, + Chains, + Full, +} + +impl std::fmt::Display for Shape { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + let s = match self { + Self::Symbols => "symbols", + Self::Chains => "chains", + Self::Full => "full", + }; + f.write_str(s) + } } /// Kết quả 1 repo — RSS từng phase + phần RAM dự đoán theo cấu trúc. @@ -259,7 +287,11 @@ fn measure(repo: &Repo, opts: &BenchOptions) -> anyhow::Result { /// nhiều RAM nhất trong `GraphIndex`. Call name cố tình trùng lặp (chỉ vài /// tên lib giả) để `call_names` có nhiều key chứa nhiều site, đúng hình dạng /// repo thật. -fn synthetic_parse_result(n: usize, fanout: usize) -> codegraph_graph::ParseResult { +fn synthetic_parse_result( + n: usize, + fanout: usize, + shape: Shape, +) -> codegraph_graph::ParseResult { // `n = 0` sẽ làm `% n` panic ở vòng sinh chain — chặn sớm, báo rõ. assert!(n > 0, "--synthetic cần N > 0"); let mut symbols = Vec::with_capacity(n); @@ -285,7 +317,7 @@ fn synthetic_parse_result(n: usize, fanout: usize) -> codegraph_graph::ParseResu }); } - let fanout = fanout.max(1); + let fanout = if shape == Shape::Symbols { 0 } else { fanout.max(1) }; for i in 0..n { let caller = codegraph_core::SYMBOL_BASE + i as u64; let mut chain = vec![caller]; @@ -294,21 +326,27 @@ fn synthetic_parse_result(n: usize, fanout: usize) -> codegraph_graph::ParseResu // không tự gọi chính mình. let callee = codegraph_core::SYMBOL_BASE + ((i + k + 1) % n) as u64; chain.push(callee); - calls.push(CallRecord { - caller_id: caller, - call_name: format!("lib::helper_{}", k % 4), - position: chain.len() - 1, - arg_exprs: vec!["x".into()], - line: 5 + k as u32, - condition: (k % 3 == 0).then(|| "x > 0".to_string()), - is_loop_body: k % 5 == 0, - effect: EffectType::None, - effect_desc: None, - target_class: None, - target_method: None, - }); + // `shape = chains`: có chain nhưng KHÔNG call record → không dựng + // `call_names`/`edges`, chỉ bật chain engine. + if shape == Shape::Full { + calls.push(CallRecord { + caller_id: caller, + call_name: format!("lib::helper_{}", k % 4), + position: chain.len() - 1, + arg_exprs: vec!["x".into()], + line: 5 + k as u32, + condition: (k % 3 == 0).then(|| "x > 0".to_string()), + is_loop_body: k % 5 == 0, + effect: EffectType::None, + effect_desc: None, + target_class: None, + target_method: None, + }); + } + } + if shape != Shape::Symbols { + chains.insert(caller, chain); } - chains.insert(caller, chain); } codegraph_graph::ParseResult { @@ -322,11 +360,11 @@ fn synthetic_parse_result(n: usize, fanout: usize) -> codegraph_graph::ParseResu } } -fn measure_synthetic(n: usize, fanout: usize) -> anyhow::Result { +fn measure_synthetic(n: usize, fanout: usize, shape: Shape) -> anyhow::Result { let mut tracker = MemTracker::new(); tracker.mark("start"); - let parsed = vec![synthetic_parse_result(n, fanout)]; + let parsed = vec![synthetic_parse_result(n, fanout, shape)]; tracker.mark("extract"); let idx = index_at(&parsed, None)?; @@ -346,7 +384,7 @@ fn measure_synthetic(n: usize, fanout: usize) -> anyhow::Result { let after_index = sample("index"); Ok(RepoMem { - repo: format!("synthetic({n})"), + repo: format!("synthetic({n},{shape})"), symbols: st.symbols, chains: st.chains, edges: st.edges, @@ -380,7 +418,7 @@ fn main() -> anyhow::Result<()> { let mut results = Vec::new(); if let Some(n) = cli.synthetic { - results.push(measure_synthetic(n, cli.fanout)?); + results.push(measure_synthetic(n, cli.fanout, cli.shape)?); } else { for repo in load_repos(&cli) { if Utf8Path::from_path(repo.root.as_std_path()).is_none() { From 7667f6a20c21c32cc2f50336693eb692b6eb7381 Mon Sep 17 00:00:00 2001 From: hungpham10 <136320753+hungpham10@users.noreply.github.com> Date: Mon, 5 Oct 2026 04:59:53 +0000 Subject: [PATCH 12/15] style: apply rustfmt --- crates/codegraph-bench/src/bin/mem.rs | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index 710950cfbe..bae17d66c6 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -132,8 +132,8 @@ struct Cli { synthetic: Option, /// Số callee mỗi function trong chế độ `--synthetic`. - #[arg(long, default_value_t = 4, value_name = "K")] - fanout: usize, + #[arg(long, default_value_t = 4, value_name = "K")] + fanout: usize, /// Thành phần nào của index synthetic cần dựng — dùng để **đo vi sai**: /// mỗi shape thiếu một phần, hiệu RSS cho ra chi phí của phần đó (gồm cả @@ -287,11 +287,7 @@ fn measure(repo: &Repo, opts: &BenchOptions) -> anyhow::Result { /// nhiều RAM nhất trong `GraphIndex`. Call name cố tình trùng lặp (chỉ vài /// tên lib giả) để `call_names` có nhiều key chứa nhiều site, đúng hình dạng /// repo thật. -fn synthetic_parse_result( - n: usize, - fanout: usize, - shape: Shape, -) -> codegraph_graph::ParseResult { +fn synthetic_parse_result(n: usize, fanout: usize, shape: Shape) -> codegraph_graph::ParseResult { // `n = 0` sẽ làm `% n` panic ở vòng sinh chain — chặn sớm, báo rõ. assert!(n > 0, "--synthetic cần N > 0"); let mut symbols = Vec::with_capacity(n); @@ -317,7 +313,11 @@ fn synthetic_parse_result( }); } - let fanout = if shape == Shape::Symbols { 0 } else { fanout.max(1) }; + let fanout = if shape == Shape::Symbols { + 0 + } else { + fanout.max(1) + }; for i in 0..n { let caller = codegraph_core::SYMBOL_BASE + i as u64; let mut chain = vec![caller]; From 4943913f9c721bb8c12bc16d4c2cc8c8edd72005 Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 13:24:30 +0700 Subject: [PATCH 13/15] =?UTF-8?q?perf(bench):=20th=C3=AAm=20`--mode=20open?= =?UTF-8?q?`=20=C4=91=E1=BB=83=20=C4=91o=20chi=20ph=C3=AD=20th=C6=B0?= =?UTF-8?q?=E1=BB=9Dng=20tr=E1=BB=B1c,=20kh=C3=B4ng=20ph=E1=BA=A3i=20peak?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sửa lỗi phương pháp: `rss_after_index` đo **high-water của allocator**, không phải live memory. Trong `ingest` tồn tại 3 bản sao call song song — `results` (borrow, `lib.rs:983`) → `all_calls` (clone, `:1054`) → `recs_by_caller` (clone, `:1389`) — rồi cả hai bản sao cuối bị drop nhưng allocator không trả arena về OS. Nên "137 MiB do call records" ở lượt trước là đỉnh lúc ingest, không phải chi phí giữ lại sau restart. `--mode open`: ingest vào sqlite → `drop(index)` → **`drop(parsed)`** → `open` lại. `open` chỉ chạy `rebuild()` (nạp blob + dựng HashMap) nên không có bản sao tạm nào; `rss_after_open − rss_before_open` là chi phí thường trực thật. Chạy cả `--mode ingest` và `--mode open` cho 3 shape trong CI để so trực tiếp: hiệu giữa hai mode = phần transient, còn mode open = phần phải giữ. --- .github/workflows/memory.yml | 17 +++++ crates/codegraph-bench/src/bin/mem.rs | 98 ++++++++++++++++++++++++++- 2 files changed, 114 insertions(+), 1 deletion(-) diff --git a/.github/workflows/memory.yml b/.github/workflows/memory.yml index 880bd6270b..5c26dc38db 100644 --- a/.github/workflows/memory.yml +++ b/.github/workflows/memory.yml @@ -63,6 +63,19 @@ jobs: >> mem-out/shapes.json done + # Reopen = chi phí THƯỜNG TRỰC. Ingest giữ 3 bản sao call song song và + # allocator không trả arena về OS → RSS sau ingest là peak, không phải live. + # `open` chạy `rebuild()` từ sqlite nên đo đúng phần giữ lại sau restart. + - name: Reopen vs ingest (persistent cost) + run: | + : > mem-out/reopen.txt + for shape in symbols chains full; do + ./target/release/mem --synthetic 50000 --fanout 4 --shape "$shape" --mode open \ + | tee -a mem-out/reopen.txt + ./target/release/mem --synthetic 50000 --fanout 4 --shape "$shape" --mode open --json \ + >> mem-out/reopen.json + done + - name: Real repos env: CODEGRAPH_BENCH_REPOS_LIST: ${{ github.workspace }}/.github/benches/repos/list.txt @@ -87,6 +100,10 @@ jobs: echo '```' >> "$GITHUB_STEP_SUMMARY" cat mem-out/shapes.txt >> "$GITHUB_STEP_SUMMARY" echo '```' >> "$GITHUB_STEP_SUMMARY" + echo '### Reopen = persistent cost (50k)' >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" + cat mem-out/reopen.txt >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" echo '```json' >> "$GITHUB_STEP_SUMMARY" cat mem-out/synthetic-50k.json >> "$GITHUB_STEP_SUMMARY" echo '```' >> "$GITHUB_STEP_SUMMARY" diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index bae17d66c6..9456b60677 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -144,6 +144,34 @@ struct Cli { /// - `full` — kèm call record → thêm `call_names` + `edges` #[arg(long, value_enum, default_value_t = Shape::Full)] shape: Shape, + + /// Cách dựng index để đo. + /// + /// - `ingest` — parse + `ingest` thẳng vào in-memory. **RSS đo được ở đây là + /// high-water của allocator**: trong lúc ingest tồn tại 3 bản sao call + /// song song (`results`, `all_calls`, `recs_by_caller`) và drop xong + /// allocator không trả arena về OS → con số này KHÔNG phải chi phí thường trực. + /// - `open` — ingest vào sqlite, **drop hết**, rồi `open` lại. Lúc này chỉ + /// còn `rebuild()` nạp blob từ storage và dựng HashMap, không có bản sao + /// tạm nào → đây mới là chi phí thường trực thật. + #[arg(long, value_enum, default_value_t = Mode::Ingest)] + mode: Mode, +} + +#[derive(Clone, Copy, PartialEq, Eq, clap::ValueEnum)] +enum Mode { + Ingest, + Open, +} + +impl std::fmt::Display for Mode { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + let s = match self { + Self::Ingest => "ingest", + Self::Open => "open", + }; + f.write_str(s) + } } #[derive(Clone, Copy, PartialEq, Eq, clap::ValueEnum)] @@ -402,6 +430,71 @@ fn measure_synthetic(n: usize, fanout: usize, shape: Shape) -> anyhow::Result anyhow::Result { + let mut tracker = MemTracker::new(); + tracker.mark("start"); + + let parsed = vec![synthetic_parse_result(n, fanout, shape)]; + + // Thư mục riêng cho mỗi lần chạy — không dùng `tempfile` vì `src/bin/` không + // có dev-dependency. + let dir = std::env::temp_dir().join(format!("codegraph-mem-{}-{}", std::process::id(), n)); + let _ = std::fs::remove_dir_all(&dir); + std::fs::create_dir_all(&dir).map_err(|e| anyhow::anyhow!("tạo {}: {e}", dir.display()))?; + let dsn = format!("sqlite://{}/db.sqlite", dir.display()); + + let ingest_idx = index_at(&parsed, Some(&dsn))?; + let st = ingest_idx.stats(); + drop(ingest_idx); + // **Quan trọng**: nhả `ParseResult` (chứa toàn bộ `CallRecord`) trước khi đo, + // không thì nó chiếm RSS suốt và mọi delta đều sai. + drop(parsed); + tracker.mark("before_open"); + + let idx = runtime().block_on(codegraph_graph::GraphIndex::open(&dsn))?; + tracker.mark("open"); + + let breakdown = runtime().block_on(idx.mem_breakdown()); + let sample = |label: &str| { + tracker + .samples() + .iter() + .find(|(l, _)| l == label) + .map(|(_, v)| *v) + .unwrap_or(0) + }; + let before_open = sample("before_open"); + let after_open = sample("open"); + + let _ = std::fs::remove_dir_all(&dir); + + Ok(RepoMem { + repo: format!("reopen({n},{shape})"), + symbols: st.symbols, + chains: st.chains, + edges: st.edges, + files: st.files, + rss_after_extract: before_open, + rss_after_index: after_open, + rss_after_query: 0, + rss_peak: tracker.peak(), + rss_index_delta: after_open.saturating_sub(before_open), + predicted_symbols: st.symbols * size_of::() as u64, + predicted_edges: st.edges * size_of::() as u64, + accounted_total: breakdown.accounted_total(), + caches: breakdown.caches.clone(), + breakdown: breakdown_rows(&breakdown), + }) +} + fn main() -> anyhow::Result<()> { let cli = Cli::parse(); if rss_bytes().is_none() { @@ -418,7 +511,10 @@ fn main() -> anyhow::Result<()> { let mut results = Vec::new(); if let Some(n) = cli.synthetic { - results.push(measure_synthetic(n, cli.fanout, cli.shape)?); + results.push(match cli.mode { + Mode::Ingest => measure_synthetic(n, cli.fanout, cli.shape)?, + Mode::Open => measure_reopen(n, cli.fanout, cli.shape)?, + }); } else { for repo in load_repos(&cli) { if Utf8Path::from_path(repo.root.as_std_path()).is_none() { From dd7b88eb08381f4b49f0d85212bd7cc81b6a3e25 Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 19:45:52 +0700 Subject: [PATCH 14/15] =?UTF-8?q?perf(graph):=20b=E1=BB=8F=202=20b?= =?UTF-8?q?=E1=BA=A3n=20clone=20`CallRecord`=20khi=20ingest?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `ingest` chỉ cần sửa đúng một field của `CallRecord` là `caller_id` (sang id global) nhưng lại clone toàn bộ struct (176 B + heap) hai lần: - `all_calls: Vec` → `Vec` (16 B, mượn từ `ParseResult` của caller vốn sống suốt `ingest` + `caller_id` đã remap). - `recs_by_caller: HashMap>` → `Vec` (index vào `calls`), tra lại qua `calls[i].rec`. 200k call record: 2 × ~43.6 MiB → ~4 MiB, tiết kiệm ~83 MiB / peak 319.7 MiB (~26%) theo số đo `--mode ingest` shape `full`. Tiền lệ có sẵn: `resolve_calls` đã dùng `Vec<&CallRecord>` để group. Ghi chú: `caller_id` trong blob persist giờ mang id **local** thay vì global. Vô hại — `flow`, `rebuild_edges`, `bingraph` đều lấy caller id từ key của blob, không đọc field đó; đã ghi chú tại chỗ ghi. Test mới `call_records_grouped_by_global_caller_id`: hai file cùng dùng `SYMBOL_BASE` làm caller local, chặn bug gom nhầm theo id local. Kèm: `--mode open` trong CI hạ 50k → 5k (rebuild() còn nạp `all_call_records()` rồi deserialize lần hai, 50k mất ~1h40m/shape ≈ 5h/job), và sửa 3 comment mô tả "3 bản sao call" đã lỗi thời. --- .github/workflows/memory.yml | 18 +++-- crates/codegraph-bench/src/bin/mem.rs | 12 +-- crates/codegraph-graph/src/lib.rs | 102 +++++++++++++++++++++----- 3 files changed, 100 insertions(+), 32 deletions(-) diff --git a/.github/workflows/memory.yml b/.github/workflows/memory.yml index 5c26dc38db..5c8fef9b29 100644 --- a/.github/workflows/memory.yml +++ b/.github/workflows/memory.yml @@ -63,16 +63,20 @@ jobs: >> mem-out/shapes.json done - # Reopen = chi phí THƯỜNG TRỰC. Ingest giữ 3 bản sao call song song và - # allocator không trả arena về OS → RSS sau ingest là peak, không phải live. - # `open` chạy `rebuild()` từ sqlite nên đo đúng phần giữ lại sau restart. - - name: Reopen vs ingest (persistent cost) + # Reopen = chi phí THƯỜNG TRỰC. `ParseResult` giữ call record xuyên suốt + # `ingest` và allocator không trả arena về OS → RSS sau ingest là peak, không + # phải live. `open` chạy `rebuild()` từ sqlite nên đo phần giữ lại sau restart. + # Quy mô 5k (không phải 50k): `rebuild()` hiện còn nạp `all_call_records()` + # + deserialize lần hai nên **rất chậm** — 50k mất ~1h40m/shape, cả job + # ~5h. Số liệu RSS của `open` vẫn chỉ để so sánh tương đối (xem ghi chú + # đầu file); ở 5k thì đủ thấy xu hướng mà job còn chạy được. + - name: Reopen vs ingest (persistent cost, 5k) run: | : > mem-out/reopen.txt for shape in symbols chains full; do - ./target/release/mem --synthetic 50000 --fanout 4 --shape "$shape" --mode open \ + ./target/release/mem --synthetic 5000 --fanout 4 --shape "$shape" --mode open \ | tee -a mem-out/reopen.txt - ./target/release/mem --synthetic 50000 --fanout 4 --shape "$shape" --mode open --json \ + ./target/release/mem --synthetic 5000 --fanout 4 --shape "$shape" --mode open --json \ >> mem-out/reopen.json done @@ -100,7 +104,7 @@ jobs: echo '```' >> "$GITHUB_STEP_SUMMARY" cat mem-out/shapes.txt >> "$GITHUB_STEP_SUMMARY" echo '```' >> "$GITHUB_STEP_SUMMARY" - echo '### Reopen = persistent cost (50k)' >> "$GITHUB_STEP_SUMMARY" + echo '### Reopen = persistent cost (5k)' >> "$GITHUB_STEP_SUMMARY" echo '```' >> "$GITHUB_STEP_SUMMARY" cat mem-out/reopen.txt >> "$GITHUB_STEP_SUMMARY" echo '```' >> "$GITHUB_STEP_SUMMARY" diff --git a/crates/codegraph-bench/src/bin/mem.rs b/crates/codegraph-bench/src/bin/mem.rs index 9456b60677..493c677f43 100644 --- a/crates/codegraph-bench/src/bin/mem.rs +++ b/crates/codegraph-bench/src/bin/mem.rs @@ -148,8 +148,8 @@ struct Cli { /// Cách dựng index để đo. /// /// - `ingest` — parse + `ingest` thẳng vào in-memory. **RSS đo được ở đây là - /// high-water của allocator**: trong lúc ingest tồn tại 3 bản sao call - /// song song (`results`, `all_calls`, `recs_by_caller`) và drop xong + /// high-water của allocator**: `ParseResult` giữ `CallRecord` xuyên suốt + /// `ingest` (`all_calls`/`recs_by_caller` chỉ giữ borrow/index) và drop xong /// allocator không trả arena về OS → con số này KHÔNG phải chi phí thường trực. /// - `open` — ingest vào sqlite, **drop hết**, rồi `open` lại. Lúc này chỉ /// còn `rebuild()` nạp blob từ storage và dựng HashMap, không có bản sao @@ -433,10 +433,10 @@ fn measure_synthetic(n: usize, fanout: usize, shape: Shape) -> anyhow::Result anyhow::Result { let mut tracker = MemTracker::new(); diff --git a/crates/codegraph-graph/src/lib.rs b/crates/codegraph-graph/src/lib.rs index fa0f536537..f31ff38e69 100644 --- a/crates/codegraph-graph/src/lib.rs +++ b/crates/codegraph-graph/src/lib.rs @@ -230,6 +230,19 @@ pub struct ParseResult { pub calls: Vec, } +/// Một call record của `ParseResult` **mượn** (không clone) + `caller_id` đã +/// remap sang id global. +/// +/// `ingest` chỉ cần sửa đúng một field của `CallRecord` là `caller_id` (mọi +/// field khác giữ nguyên), nên clone toàn bộ `CallRecord` (176 B + heap) vào +/// `all_calls` là phí vô ích — struct này giữ 16 B thay vì 176 B, trỏ về +/// `ParseResult` của caller (vốn sống suốt `ingest`). +struct CallRef<'a> { + rec: &'a CallRecord, + /// `caller_id` sau remap (giữ `0` nếu local id không có trong `id_map`). + caller: u64, +} + // ==================== GraphIndex ==================== /// Index chính (semgraph-style): registry + 2 engine + inverted indexes. @@ -1019,7 +1032,7 @@ impl GraphIndex { if let Some(p) = p { p.phase("register symbols", total_symbols); } - let mut all_calls: Vec = Vec::new(); + let mut all_calls: Vec> = Vec::new(); for result in results { let mut id_map: HashMap = HashMap::new(); for sym in &result.symbols { @@ -1047,11 +1060,8 @@ impl GraphIndex { } // Calls — remap caller_id (position giữ nguyên — đã trỏ đúng chain). for c in &result.calls { - let mut c2 = c.clone(); - if let Some(&nid) = id_map.get(&c.caller_id) { - c2.caller_id = nid; - } - all_calls.push(c2); + let caller = id_map.get(&c.caller_id).copied().unwrap_or(c.caller_id); + all_calls.push(CallRef { rec: c, caller }); } if let Some(p) = p { p.advance(result.symbols.len()); @@ -1164,11 +1174,11 @@ impl GraphIndex { } /// Thay placeholder `0` trong chain bằng id thật (resolve per-caller). - fn resolve_calls(&mut self, calls: &[CallRecord]) { + fn resolve_calls(&mut self, calls: &[CallRef<'_>]) { let mut caller_calls: HashMap> = HashMap::new(); for c in calls { - if c.caller_id != 0 { - caller_calls.entry(c.caller_id).or_default().push(c); + if c.caller != 0 { + caller_calls.entry(c.caller).or_default().push(c.rec); } } for (caller_id, ccs) in caller_calls { @@ -1380,13 +1390,15 @@ impl GraphIndex { /// position. Chain dựng thẳng (không qua placeholder) vẫn sinh edge đủ. async fn build_edges_from_calls( &mut self, - calls: &[CallRecord], + calls: &[CallRef<'_>], progress: Option<&dyn IngestProgress>, ) -> Result<()> { - let mut recs_by_caller: HashMap> = HashMap::new(); - for c in calls { - let caller = c.caller_id; - recs_by_caller.entry(caller).or_default().push(c.clone()); + // Chỉ lưu **index** vào `calls` (4 B) thay vì clone `CallRecord` + // (176 B + heap) — `calls` sống tới hết hàm nên vẫn tra được. + let mut recs_by_caller: HashMap> = HashMap::new(); + for (i, entry) in calls.iter().enumerate() { + let (c, caller) = (entry.rec, entry.caller); + recs_by_caller.entry(caller).or_default().push(i as u32); // Call-site index: key theo tên thô + alias type-qualified (nếu có). let site = CallSite { @@ -1411,10 +1423,13 @@ impl GraphIndex { // Edges từ mọi chain — rec lookup theo position cho metadata. for (&caller, chain) in &self.chains_map { - let rec_by_pos: HashMap = recs_by_caller - .get(&caller) - .map(|rs| rs.iter().map(|c| (c.position, c)).collect()) - .unwrap_or_default(); + let mut rec_by_pos: HashMap = HashMap::new(); + if let Some(idxs) = recs_by_caller.get(&caller) { + for &i in idxs { + let rec = calls[i as usize].rec; + rec_by_pos.insert(rec.position, rec); + } + } for (i, &e) in chain.iter().enumerate() { // Vị trí 0 = chính func id (owner) — không phải call. Recursion // thật xuất hiện ở vị trí > 0 (vẫn giữ là edge is_recursive). @@ -1448,7 +1463,12 @@ impl GraphIndex { { p.phase("save call records", recs_by_caller.len()); } - for (caller, recs) in recs_by_caller { + for (caller, idxs) in recs_by_caller { + let recs: Vec<&CallRecord> = idxs.iter().map(|&i| calls[i as usize].rec).collect(); + // `rec.caller_id` ghi ra đây là id **local** của file, không phải + // id global — vô hại: mọi reader (`flow`, `rebuild_edges`, + // `bingraph`) lấy caller id từ **key** của blob, không đọc field + // này. Nếu sau này cần id global thì đừng đọc field — dùng key. let bytes = serde_json::to_vec(&recs).map_err(|e| Error::Search(e.to_string()))?; self.storage .write() @@ -3193,6 +3213,50 @@ mod tests { assert_eq!(hits[0].call_sites[0].line, 3); } + /// Call records phải gom theo caller id **đã remap** (global), không theo + /// `caller_id` local của file — hai file đều dùng `SYMBOL_BASE` làm caller + /// local. Gom nhầm theo id local thì blob của file 1 nuốt luôn record của + /// file 2 và `flow` của caller thứ hai mất sạch call. + #[tokio::test] + async fn call_records_grouped_by_global_caller_id() { + let call = |name: &str| CallRecord { + caller_id: SYMBOL_BASE, + call_name: name.to_string(), + position: 1, + arg_exprs: vec![], + line: 7, + condition: None, + is_loop_body: false, + effect: EffectType::None, + effect_desc: None, + target_class: None, + target_method: None, + }; + let mut idx = GraphIndex::in_memory(); + let first = result( + "first.ts", + vec![sym("first.ts", "first_fn", SYMBOL_BASE)], + HashMap::from([(SYMBOL_BASE, vec![SYMBOL_BASE, 0])]), + vec![call("alpha.only_a")], + ); + let second = result( + "second.ts", + vec![sym("second.ts", "second_fn", SYMBOL_BASE)], + HashMap::from([(SYMBOL_BASE, vec![SYMBOL_BASE, 0])]), + vec![call("beta.only_b")], + ); + idx.ingest(&[first, second]).await.unwrap(); + + let a = SYMBOL_BASE; + let b = SYMBOL_BASE + 1; + let fa = idx.flow(a).await.unwrap(); + let fb = idx.flow(b).await.unwrap(); + assert_eq!(fa.calls.len(), 1); + assert_eq!(fa.calls[0].to_name, "alpha.only_a"); + assert_eq!(fb.calls.len(), 1); + assert_eq!(fb.calls[0].to_name, "beta.only_b"); + } + #[tokio::test] async fn loop_marker_golden_chain() { let mut idx = GraphIndex::in_memory(); From 25a4bdb85d339e2a31f67cae2fe0cfbbe9a0f44e Mon Sep 17 00:00:00 2001 From: Hung Pham Date: Mon, 5 Oct 2026 21:35:17 +0700 Subject: [PATCH 15/15] Bump version to v2.2.5 --- Cargo.toml | 2 +- packaging/aur/codegraph-rs-bin/PKGBUILD | 2 +- packaging/choco/codegraph.nuspec | 2 +- packaging/winget/codegraph.yaml | 4 ++-- scripts/install.ps1 | 4 ++-- 5 files changed, 7 insertions(+), 7 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index e21162172f..2b462f3081 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -17,7 +17,7 @@ members = [ ] [workspace.package] -version = "2.2.4" +version = "2.2.5" edition = "2021" rust-version = "1.80" license = "MIT" diff --git a/packaging/aur/codegraph-rs-bin/PKGBUILD b/packaging/aur/codegraph-rs-bin/PKGBUILD index 92611c2478..e8ef80e9b5 100644 --- a/packaging/aur/codegraph-rs-bin/PKGBUILD +++ b/packaging/aur/codegraph-rs-bin/PKGBUILD @@ -1,6 +1,6 @@ # Maintainer: Hung Pham pkgname=codegraph-rs-bin -pkgver=2.2.4 +pkgver=2.2.5 pkgrel=1 pkgdesc="Local-first code intelligence: tree-sitter knowledge graph + MCP server (prebuilt binary)" arch=('x86_64' 'aarch64') diff --git a/packaging/choco/codegraph.nuspec b/packaging/choco/codegraph.nuspec index 5196ceeb62..8cb9fa796d 100644 --- a/packaging/choco/codegraph.nuspec +++ b/packaging/choco/codegraph.nuspec @@ -2,7 +2,7 @@ codegraph - 2.2.4 + 2.2.5 codegraph Hung Pham https://github.com/hungpham10/codegraph-rs diff --git a/packaging/winget/codegraph.yaml b/packaging/winget/codegraph.yaml index aa2248cda1..bbd92795ee 100644 --- a/packaging/winget/codegraph.yaml +++ b/packaging/winget/codegraph.yaml @@ -6,7 +6,7 @@ # release time (or automate it in the release pipeline before submitting to # microsoft/winget-pkgs). PackageIdentifier: hungpham10.codegraph -PackageVersion: 2.2.4 +PackageVersion: 2.2.5 PackageName: codegraph Publisher: Hung Pham PublisherUrl: https://github.com/hungpham10/codegraph-rs @@ -18,7 +18,7 @@ PackageUrl: https://github.com/hungpham10/codegraph-rs InstallerType: zip Installers: - Architecture: x64 - InstallerUrl: https://github.com/hungpham10/codegraph-rs/releases/download/v2.2.4/codegraph-x86_64-pc-windows-msvc.zip + InstallerUrl: https://github.com/hungpham10/codegraph-rs/releases/download/v2.2.5/codegraph-x86_64-pc-windows-msvc.zip InstallerSha256: 0000000000000000000000000000000000000000000000000000000000000000 InstallerType: zip ManifestType: singleton diff --git a/scripts/install.ps1 b/scripts/install.ps1 index e91564c560..3dc5134aab 100644 --- a/scripts/install.ps1 +++ b/scripts/install.ps1 @@ -5,11 +5,11 @@ # # Usage (pin a version / download the script first): # irm https://raw.githubusercontent.com/hungpham10/codegraph-rs/main/scripts/install.ps1 -OutFile install.ps1 -# .\install.ps1 -Version 2.2.4 +# .\install.ps1 -Version 2.2.5 [CmdletBinding()] param( - # Pin a specific version, e.g. "2.2.4". Empty = latest release. + # Pin a specific version, e.g. "2.2.5". Empty = latest release. [string]$Version )