New features

This commit is contained in:
Ingo Schnabel
2026-08-19 09:52:24 +03:00
parent 907286dabb
commit 29269b83c3
12 changed files with 571 additions and 43 deletions

View File

@@ -40,6 +40,16 @@ final class RefreshCommand extends AbstractProjectCommand {
@Option(names = "--neighborhood", description = "Module refresh only: also deep-ingest the module's transitive callers (not just callees/data areas)")
boolean neighborhood;
@Option(names = "--paths", split = ",",
description = "Whole-project refresh only: re-ingest just these relative source paths (repeatable or comma-separated) "
+ "instead of the whole root. Paths matching no file come back under 'unresolved'")
List<String> paths = List.of();
@Option(names = "--changed-only",
description = "Whole-project refresh only: re-parse only files whose content differs from the graph's stored hash. "
+ "Enrichment still runs in full; a changed Natural copycode re-parses everything (its text is inlined at parse time)")
boolean changedOnly;
@Override
public Integer call() throws Exception {
if (name != null && !name.isBlank()) {
@@ -60,6 +70,12 @@ final class RefreshCommand extends AbstractProjectCommand {
return printResponse(apiClient().post(path));
}
String path = projectPath() + "/refresh" + (deep ? "?deep=true" : "");
if (!paths.isEmpty()) {
path = appendQuery(path, "paths", String.join(",", paths));
}
if (changedOnly) {
path = appendQuery(path, "changedOnly", "true");
}
return printResponse(apiClient().post(path));
}
}

View File

@@ -4,4 +4,4 @@
server.url=http://localhost:8787
# Stamped by manage-ac.sh (stamp_cli_version) from ac-code-server's agenticcode.version
# at build time. "dev" means this jar wasn't built via manage-ac.sh.
version=222
version=226

View File

@@ -363,9 +363,28 @@ public class AnalysisResource {
@APIResponse(responseCode = "200", content = @Content(schema = @Schema(implementation = IngestSummary.class)))
@APIResponse(responseCode = "404", description = "Project not found.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
public Response refresh(@PathParam("project") String project,
@QueryParam("deep") @Nullable Boolean deep) {
@QueryParam("deep") @Nullable Boolean deep,
@Parameter(description = "Item 129: comma-separated relative source paths to re-ingest instead of the whole root — "
+ "the fast loop after editing a few files. Always deep for the named files and their dependencies. "
+ "Requested paths that match no file under the root come back in 'unresolved' rather than being dropped. "
+ "Does not run the deleted-file sweep and does not move the project's ingestedAt, both of which need a whole-root walk.")
@QueryParam("paths") @Nullable String paths,
@Parameter(description = "Item 129: re-parse only files whose content differs from the graph's stored hash. "
+ "Opt-in: enrichment still runs in full (so this cuts parse time only), duplicate detection sees just the "
+ "changed files, user-exit LoC is re-stamped only on re-parsed files, and a changed Natural copycode "
+ "disables the skipping for that run because copycode text is inlined at parse time.")
@QueryParam("changedOnly") @Nullable Boolean changedOnly) {
boolean deepIngest = deep != null && deep;
return withResolvedRoot(project, info -> projectIngestService.refreshProject(info, deepIngest));
if (paths != null && !paths.isBlank()) {
List<String> requested = Arrays.stream(paths.split(",")).map(String::trim).filter(p -> !p.isEmpty()).toList();
if (requested.isEmpty()) {
return ProjectResource.error(Response.Status.BAD_REQUEST, "MISSING_PATHS",
"Query parameter 'paths' was given but contained no path");
}
return withResolvedRoot(project, info -> projectIngestService.refreshPaths(info, requested));
}
return withResolvedRoot(project, info ->
projectIngestService.refreshProject(info, deepIngest, Boolean.TRUE.equals(changedOnly)));
}
/**

View File

@@ -165,6 +165,22 @@ public class AstIngestService {
return graphRepository.recordProjectIngest(project, ingest);
}
/**
* Item 129: marks a whole-root ingest as in flight, so one that never finishes leaves the graph
* visibly half-updated rather than looking clean.
*/
public Uni<Void> markProjectIngestStarted(String project, String mode, String startedAt) {
return graphRepository.markProjectIngestStarted(project, mode, startedAt);
}
/**
* Item 129: every ingested file's stored content hash, the input to a {@code changedOnly} refresh's
* skip decision.
*/
public Uni<java.util.Map<String, String>> sourceHashes(String project) {
return graphRepository.sourceHashes(project);
}
/**
* Item 43: deletes every node of {@code project} belonging to one of {@code sourceFiles} — the
* deleted-file orphan sweep run after a whole-project refresh.

View File

@@ -289,6 +289,18 @@ public class ProjectIngestService {
return new IngestSummary.Truncation(reason, depthLimit, nodeLimit, hint);
}
/**
* @return {@code file}'s content hash, or {@code ""} when it cannot be read — which never equals a
* stored hash, so an unreadable file is re-parsed (and fails loudly there) rather than skipped.
*/
private static String hashOf(Path file) {
try {
return SourceHash.of(SourceFiles.read(file));
} catch (IOException e) {
return "";
}
}
/**
* Walks the entire project root and ingests every ingestible file. {@code deep} selects the
* enrichment level: {@code deep=false} (default) runs the fast call-graph pass (placeholder
@@ -299,7 +311,7 @@ public class ProjectIngestService {
* {@link IngestDepth#FULL}).
*/
public IngestSummary ingestAll(ProjectInfo project, boolean deep) throws IOException {
return ingestRoot(project, deep ? EnrichmentLevel.FULL : EnrichmentLevel.CALL_GRAPH, false);
return ingestRoot(project, deep ? EnrichmentLevel.FULL : EnrichmentLevel.CALL_GRAPH, false, false);
}
/**
@@ -312,7 +324,7 @@ public class ProjectIngestService {
* resolve; field-level dataflow is deferred to the on-demand deep ingest.
*/
public IngestSummary scanTier1(ProjectInfo project) throws IOException {
return ingestRoot(project, EnrichmentLevel.CALL_GRAPH, true);
return ingestRoot(project, EnrichmentLevel.CALL_GRAPH, true, false);
}
/**
@@ -322,7 +334,7 @@ public class ProjectIngestService {
* caller to deep-ingest a program first. Intended as the fast first pass after project creation.
*/
public IngestSummary ingestCallGraph(ProjectInfo project) throws IOException {
return ingestRoot(project, EnrichmentLevel.CALL_GRAPH, false);
return ingestRoot(project, EnrichmentLevel.CALL_GRAPH, false, false);
}
/**
@@ -335,8 +347,32 @@ public class ProjectIngestService {
* wipe.)
*/
public IngestSummary refreshProject(ProjectInfo project, boolean deep) throws IOException {
return refreshProject(project, deep, false);
}
/**
* Item 129: {@code changedOnly} re-parses only files whose content hash differs from the graph's
* (plus files with no stored hash). <b>Opt-in, and deliberately not the default</b> — three
* whole-walk behaviours are reduced in this mode:
*
* <ul>
* <li><b>Natural copycodes are inlined at parse time</b>, so a module whose {@code .cpy} changed
* parses differently while its own hash is unchanged. A changed copycode therefore
* <b>disables skipping for the whole run</b> (logged) rather than silently keeping stale
* expansions — the one case where "incremental" would corrupt the graph outright.</li>
* <li><b>Duplicate identity detection</b> groups the files it parsed; with a subset it can only
* confirm duplicates among changed files. Existing markers are never cleared (the query only
* MERGEs), so this loses discovery, not recorded facts.</li>
* <li><b>User-exit LoC annotation</b> (item 47) is re-stamped only on files that were re-parsed:
* a changed user-exit twin does not refresh an unchanged generated module's metrics.</li>
* </ul>
*
* <p>Enrichment is project-wide and still runs in full, so this cuts parse+persist time only.
*/
public IngestSummary refreshProject(ProjectInfo project, boolean deep, boolean changedOnly) throws IOException {
long startedAt = System.nanoTime();
IngestSummary summary = deep ? ingestAll(project, true) : ingestCallGraph(project);
IngestSummary summary = ingestRoot(project, deep ? EnrichmentLevel.FULL : EnrichmentLevel.CALL_GRAPH,
false, changedOnly);
sweepDeletedFileOrphans(project);
LOG.infof("Refresh finished: project='%s', mode=%s, files=%d, modules=%d, failed=%d, %d s",
project.name(), deep ? "deep" : "call-graph", summary.examinedFiles().size(),
@@ -380,17 +416,64 @@ public class ProjectIngestService {
return ingestModule(project, moduleName, maxDepth, maxNodes);
}
/**
* Item 129: re-ingests <b>only the named files</b> (relative paths), for the common case of a code
* change touching a handful of files where a whole-root refresh re-examines thousands.
*
* <p>Deep by construction: it runs the same BFS deep ingest the fan-out warm uses, so the named
* files and their dependencies land {@code FULL}. Two things it deliberately does <b>not</b> do,
* because they are only meaningful for a whole-root walk: the deleted-file sweep (item 43 —
* nothing here says which files disappeared) and the project-shell ingest stamp (item 126 — this
* walks a fraction of the tree, and moving {@code ingestedAt} would advertise the project as
* freshly walked).
*
* <p>Paths that match no file under the root are returned in the summary's {@code unresolved} list
* rather than dropped: "I ingested 2 of your 3 files" must be visible, or a typo'd path reads as a
* successful refresh.
*/
public IngestSummary refreshPaths(ProjectInfo project, Collection<String> sourceFiles) throws IOException {
Path root = Path.of(project.root()).toAbsolutePath().normalize();
List<String> requested = sourceFiles.stream()
.map(String::trim)
.filter(sf -> !sf.isEmpty())
.distinct()
.toList();
// Same guard as the source endpoints: these paths are client-supplied, so a '..' or an
// absolute path must not reach outside the project root.
List<String> unresolved = requested.stream()
.filter(sf -> !root.resolve(sf).normalize().startsWith(root)
|| !Files.isRegularFile(root.resolve(sf).normalize()))
.toList();
List<String> ingestable = requested.stream().filter(sf -> !unresolved.contains(sf)).toList();
if (ingestable.isEmpty()) {
return new IngestSummary(0, unresolved, List.of(), List.of(), List.of(), null);
}
IngestSummary summary = ingestFiles(project, ingestable, null);
LOG.infof("Targeted refresh of '%s': %d file(s) requested, %d ingested, %d unresolved",
project.name(), requested.size(), summary.ingested(), unresolved.size());
return new IngestSummary(summary.ingested(), unresolved, summary.duplicates(), summary.failed(),
ingestable, summary.truncation());
}
/**
* Shared whole-root ingest. {@code level} selects how far enrichment goes, and the
* {@link IngestDepth} the ingested modules are tagged with ({@code FULL} only when field-level
* resolution ran, else {@code CALL_GRAPH}).
*/
private IngestSummary ingestRoot(ProjectInfo project, EnrichmentLevel level, boolean coarse) throws IOException {
private IngestSummary ingestRoot(ProjectInfo project, EnrichmentLevel level, boolean coarse,
boolean changedOnly) throws IOException {
long startedAt = System.nanoTime();
// Item 129: mark the pass in flight before touching anything. If it never reaches the record
// call at the end — crash, container stop, an aborted deep refresh — the marker stays set and
// every later answer can say the graph is half-updated instead of looking clean.
markIngestStarted(project, coarse ? "tier1" : level.name().toLowerCase(Locale.ROOT));
Path root = Path.of(project.root());
List<String> excludeDirs = ingestExcludeDirs(project);
List<Candidate> candidates = walk(root, excludeDirs);
List<Candidate> allCandidates = walk(root, excludeDirs);
CopycodeLibrary copycodes = CopycodeLibrary.scan(root, excludeDirs);
// Item 129: opt-in skip of files whose content is byte-identical to what the graph holds.
List<Candidate> candidates = changedOnly ? changedCandidates(project, root, excludeDirs, allCandidates)
: allCandidates;
Map<String, LocMetrics> userExit = UserExitMetrics.scan(root, project.userExitDir(), project.excludeDirs(), astIngestService);
List<String> examinedFiles = candidates.stream()
@@ -497,10 +580,81 @@ public class ProjectIngestService {
// three whole-root passes (Tier-1 scan, call-graph refresh, deep refresh). By-name and fan-out
// ingests deliberately do not come through here — they walk a fraction of the tree, and moving
// ingestedAt for them would report the project as freshly walked when one module was deepened.
// Item 129: filesExamined counts what this pass actually walked. In changedOnly mode that is
// the changed subset, and the log line above says so — a smaller number here is the point, not
// a sign of a short walk.
recordIngest(project, mode, examinedFiles.size(), ingested, failed, durationSeconds);
return new IngestSummary(ingested, List.of(), duplicates, failed, examinedFiles, null);
}
/**
* Item 129: the subset of {@code candidates} whose on-disk content differs from the hash stored in
* the graph (or that has no stored hash at all — a new file, or one ingested before hashes existed).
*
* <p>Returns <b>every</b> candidate — i.e. skips nothing — when a Natural copycode has changed.
* Copycode text is inlined into the including module at parse time, so those modules parse
* differently while their own hashes are unchanged; skipping them would leave stale expansions in
* the graph with nothing to indicate it. Reading the copycodes' own hashes is not enough to know
* <em>which</em> modules include them at this point in the walk, and guessing wrong is silent
* corruption, so the whole optimisation stands down for that run.
*/
private List<Candidate> changedCandidates(ProjectInfo project, Path root, List<String> excludeDirs,
List<Candidate> candidates) {
Map<String, String> stored = astIngestService.sourceHashes(project.name()).await().indefinitely();
if (copycodeChanged(root, excludeDirs, stored)) {
LOG.infof("changedOnly refresh of '%s': a copycode changed, so every file is re-parsed "
+ "(copycode text is inlined at parse time; skipping would keep stale expansions)",
project.name());
return candidates;
}
List<Candidate> changed = new ArrayList<>();
for (Candidate candidate : candidates) {
String relative = relativeSourceFile(root, candidate.file());
@Nullable String known = stored.get(relative);
if (known == null || !known.equals(hashOf(candidate.file()))) {
changed.add(candidate);
}
}
LOG.infof("changedOnly refresh of '%s': %d of %d files changed since the last ingest",
project.name(), changed.size(), candidates.size());
return changed;
}
/**
* @return whether any {@code .cpy} member's content differs from the hash the graph holds for it.
* A copycode with no stored hash counts as changed: it may never have been ingested, and assuming
* otherwise is the unsafe direction.
*/
private boolean copycodeChanged(Path root, List<String> excludeDirs, Map<String, String> stored) {
try (Stream<Path> stream = Files.walk(root)) {
return stream.filter(Files::isRegularFile)
.filter(f -> f.getFileName().toString().toLowerCase(Locale.ROOT).endsWith(".cpy"))
.filter(f -> !SourceFiles.isExcluded(root, f, excludeDirs))
.anyMatch(f -> {
@Nullable String known = stored.get(relativeSourceFile(root, f));
return known == null || !known.equals(hashOf(f));
});
} catch (IOException e) {
LOG.warnf("Could not scan copycodes under '%s' for changes (%s); re-parsing everything",
root, e.toString());
return true;
}
}
/**
* Item 129: stamps the in-flight marker. Never fatal — a project whose bookkeeping cannot be written
* must still be ingestable; the cost of failing here is only that an interruption would go unmarked.
*/
private void markIngestStarted(ProjectInfo project, String mode) {
try {
astIngestService.markProjectIngestStarted(project.name(), mode,
Instant.now().truncatedTo(ChronoUnit.SECONDS).toString()).await().indefinitely();
} catch (RuntimeException e) {
LOG.warnf(e, "Could not mark ingest start for project '%s'; an interrupted run will not be flagged",
project.name());
}
}
/**
* Item 126: writes the whole-root ingest's outcome onto the {@code (:Project)} shell. Failure
* <em>paths</em> are capped at {@link ProjectIngestInfo#MAX_FAILURES} while the count stays exact,
@@ -519,7 +673,9 @@ public class ProjectIngestService {
ProjectIngestInfo ingest = new ProjectIngestInfo(
Instant.now().truncatedTo(ChronoUnit.SECONDS).toString(), mode,
filesExamined, filesPersisted, failed.size(), failurePaths,
failed.size() > failurePaths.size(), durationSeconds, versionInfo.version());
failed.size() > failurePaths.size(), durationSeconds, versionInfo.version(),
// Reaching here means the pass completed; the write clears the in-flight marker.
false, null);
try {
astIngestService.recordProjectIngest(project.name(), ingest).await().indefinitely();
} catch (RuntimeException e) {

View File

@@ -3,7 +3,7 @@ quarkus.http.port=8787
# AgenticCode's own release counter (not the Maven project version) — bump this by hand for each
# release. Single source of truth for the startup log line, GET /api/version, and the OpenAPI
# info version (referenced below via property expression, not duplicated).
agenticcode.version=222
agenticcode.version=226
# OpenAPI / Swagger UI (item 48) — the generated spec is the contract the web-UI TS client
# is generated against. Served at /q/openapi (yaml/json); Swagger UI at /q/swagger-ui in dev.
mp.openapi.extensions.smallrye.info.title=AgenticCode API

View File

@@ -0,0 +1,183 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.*;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 129: a refresh must be able to touch only what changed, and an interrupted one must not look
* like a clean graph.
*
* <p>Covers the three pieces separately because they are independent: {@code ?paths=} (targeted
* re-ingest), the {@code incomplete} marker (in-flight / aborted), and opt-in {@code ?changedOnly=}
* skipping. The assertions about what {@code changedOnly} deliberately does <em>not</em> do matter as
* much as the skipping itself — an incremental refresh that quietly kept stale copycode expansions
* would be worse than no incremental refresh at all.
*/
@QuarkusTest
@TestMethodOrder(MethodOrderer.OrderAnnotation.class)
class TargetedRefreshIT {
private static final String PROJECT = "targeted-refresh-project";
@TempDir
static Path root;
@BeforeAll
static void createProject() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
writeSource("MAIN.nat", """
DEFINE DATA
LOCAL
1 #A (A8)
END-DEFINE
*
CALLNAT 'HELPER'
END
""");
writeSource("HELPER.nat", """
DEFINE DATA
LOCAL
1 #B (A8)
END-DEFINE
*
END
""");
given()
.contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
.when().post("/api/projects/" + PROJECT)
.then()
.statusCode(201);
}
private static void writeSource(String fileName, String content) {
try {
Files.writeString(root.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
@Test
@Order(1)
void pathsReIngestsOnlyTheNamedFile() {
given()
.when().post("/api/projects/" + PROJECT + "/refresh?paths=MAIN.nat")
.then()
.statusCode(200)
.body("examinedFiles", contains("MAIN.nat"))
.body("unresolved", empty());
}
@Test
@Order(2)
void aPathThatMatchesNothingIsReportedRatherThanDropped() {
given()
.when().post("/api/projects/" + PROJECT + "/refresh?paths=MAIN.nat,NOPE.nat")
.then()
.statusCode(200)
.body("examinedFiles", contains("MAIN.nat"))
.body("unresolved", contains("NOPE.nat"));
}
@Test
@Order(3)
void aPathEscapingTheRootIsRefusedAsUnresolved() {
given()
.when().post("/api/projects/" + PROJECT + "/refresh?paths=../outside.nat")
.then()
.statusCode(200)
.body("ingested", equalTo(0))
.body("unresolved", contains("../outside.nat"));
}
@Test
@Order(4)
void aTargetedRefreshDoesNotMoveTheProjectsIngestedAt() {
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
String before = given().when().get("/api/projects/" + PROJECT).jsonPath().getString("ingest.ingestedAt");
given().when().post("/api/projects/" + PROJECT + "/refresh?paths=MAIN.nat").then().statusCode(200);
given().when().get("/api/projects/" + PROJECT)
.then().body("ingest.ingestedAt", equalTo(before));
}
@Test
@Order(5)
void aCompletedRefreshLeavesTheProjectNotIncomplete() {
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
given().when().get("/api/projects/" + PROJECT)
.then()
.statusCode(200)
.body("ingest.incomplete", equalTo(false))
.body("ingest.ingestedAt", notNullValue());
}
@Test
@Order(6)
void changedOnlyExaminesJustTheChangedFile() {
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
writeSource("HELPER.nat", """
DEFINE DATA
LOCAL
1 #B (A8)
1 #C (N4)
END-DEFINE
*
END
""");
given()
.when().post("/api/projects/" + PROJECT + "/refresh?changedOnly=true")
.then()
.statusCode(200)
.body("examinedFiles", contains("HELPER.nat"));
}
@Test
@Order(7)
void changedOnlyWithNoChangesExaminesNothing() {
given()
.when().post("/api/projects/" + PROJECT + "/refresh?changedOnly=true")
.then()
.statusCode(200)
.body("examinedFiles", empty())
.body("ingested", equalTo(0));
}
/**
* The correctness case that keeps {@code changedOnly} opt-in: copycode text is inlined into the
* including module at parse time, so a module whose {@code .cpy} changed parses differently while
* its own hash is unchanged. Skipping it would leave a stale expansion in the graph with nothing
* to indicate it — so a changed copycode must re-parse everything, not just itself.
*/
@Test
@Order(8)
void aChangedCopycodeDisablesSkippingForTheWholeRun() {
writeSource("SHARED.cpy", "* shared\n");
writeSource("USER.nat", """
DEFINE DATA
LOCAL
1 #D (A8)
END-DEFINE
*
INCLUDE SHARED
END
""");
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
writeSource("SHARED.cpy", "* shared, now different\n");
given()
.when().post("/api/projects/" + PROJECT + "/refresh?changedOnly=true")
.then()
.statusCode(200)
.body("examinedFiles", hasItems("MAIN.nat", "HELPER.nat", "USER.nat"));
}
}

View File

@@ -332,6 +332,18 @@ public final class CypherQueries {
* slice. Returns {@code null} when no hash was stored (e.g. a legacy project), so the read falls
* back to serving the snippet unchecked.
*/
/**
* Item 129: every ingested file's stored content hash in one round trip, so a {@code changedOnly}
* refresh can decide what to re-parse without a query per file. One row per {@code sourceFile};
* files whose shell carries no hash (never coarse-scanned) are absent, and therefore treated as
* changed — the safe direction.
*/
public static final String SOURCE_HASHES = """
MATCH (n:AstNode {project: $project})
WHERE n.sourceHash IS NOT NULL AND n.sourceFile <> ""
RETURN DISTINCT n.sourceFile AS sourceFile, n.sourceHash AS hash
""";
public static final String SOURCE_HASH = """
MATCH (n:AstNode {project: $project, sourceFile: $sourceFile})
WHERE n.sourceHash IS NOT NULL
@@ -2622,7 +2634,10 @@ public final class CypherQueries {
p.ingestFilesExamined AS ingestFilesExamined, p.ingestFilesPersisted AS ingestFilesPersisted,
p.ingestFilesFailed AS ingestFilesFailed, p.ingestFailures AS ingestFailures,
p.ingestFailuresTruncated AS ingestFailuresTruncated,
p.ingestDurationSeconds AS ingestDurationSeconds, p.ingestServerVersion AS ingestServerVersion
p.ingestDurationSeconds AS ingestDurationSeconds, p.ingestServerVersion AS ingestServerVersion,
// Item 129: true while a whole-root pass is running, and left true by one that never
// finished — the graph is then half-updated and every answer is drawn from that state.
p.ingestIncomplete AS ingestIncomplete, p.ingestStartedAt AS ingestStartedAt
""";
/**
@@ -2654,7 +2669,10 @@ public final class CypherQueries {
p.ingestFilesExamined AS ingestFilesExamined, p.ingestFilesPersisted AS ingestFilesPersisted,
p.ingestFilesFailed AS ingestFilesFailed, p.ingestFailures AS ingestFailures,
p.ingestFailuresTruncated AS ingestFailuresTruncated,
p.ingestDurationSeconds AS ingestDurationSeconds, p.ingestServerVersion AS ingestServerVersion
p.ingestDurationSeconds AS ingestDurationSeconds, p.ingestServerVersion AS ingestServerVersion,
// Item 129: true while a whole-root pass is running, and left true by one that never
// finished — the graph is then half-updated and every answer is drawn from that state.
p.ingestIncomplete AS ingestIncomplete, p.ingestStartedAt AS ingestStartedAt
ORDER BY p.name
""";
@@ -2667,13 +2685,30 @@ public final class CypherQueries {
* <p>Separate from {@link #UPDATE_PROJECT} on purpose: that one {@code COALESCE}s user config, and
* these are server observations. Neither can clobber the other.
*/
/**
* Item 129: marks a whole-root ingest as <b>in flight</b> before it starts. {@link #RECORD_PROJECT_INGEST}
* clears it on success, so a pass that never finished — a crash, a container stop, an aborted deep
* refresh — leaves {@code ingestIncomplete = true} behind and every later answer can say so.
*
* <p>Before this, an interrupted deep refresh was indistinguishable from a clean graph: the earlier
* enrichment steps are already committed, so queries keep answering, just from a half-updated graph.
* The marker cannot self-heal (a killed process clears nothing), and that is the correct direction to
* fail — a false "incomplete" costs one refresh, a false "clean" costs trust in every answer.
*/
public static final String MARK_PROJECT_INGEST_STARTED = """
MATCH (p:Project {name: $name})
SET p.ingestIncomplete = true, p.ingestStartedAt = $startedAt, p.ingestMode = $mode
""";
public static final String RECORD_PROJECT_INGEST = """
MATCH (p:Project {name: $name})
SET p.ingestedAt = $ingestedAt, p.ingestMode = $mode,
p.ingestFilesExamined = $filesExamined, p.ingestFilesPersisted = $filesPersisted,
p.ingestFilesFailed = $filesFailed, p.ingestFailures = $failures,
p.ingestFailuresTruncated = $failuresTruncated,
p.ingestDurationSeconds = $durationSeconds, p.ingestServerVersion = $serverVersion
p.ingestDurationSeconds = $durationSeconds, p.ingestServerVersion = $serverVersion,
// Item 129: reaching here means the pass completed, so the in-flight marker is cleared.
p.ingestIncomplete = false
""";
/**

View File

@@ -1080,6 +1080,43 @@ public class GraphRepository {
.map(hashes -> hashes.isEmpty() ? null : hashes.get(0));
}
/**
* Item 126: the last whole-root ingest's facts, or {@code null} when the project shell carries no
* {@code ingestedAt} — i.e. it was last ingested before this was recorded. Null rather than a
* zero-filled record: "never measured" and "measured as none" are different answers, and
* conflating them is the failure mode the item is about.
*/
private static @Nullable ProjectIngestInfo toProjectIngestInfo(Record record) {
@Nullable String ingestedAt = nullableString(record, "ingestedAt");
boolean incomplete = record.containsKey("ingestIncomplete") && !record.get("ingestIncomplete").isNull()
&& record.get("ingestIncomplete").asBoolean();
// Item 129: a project whose very first whole-root pass is still running (or died) has no
// ingestedAt yet, but the in-flight marker is the most important thing to report about it —
// so it is not folded into the "never recorded" null case below.
if (ingestedAt == null && !incomplete) {
return null;
}
if (ingestedAt == null) {
return new ProjectIngestInfo("", nullableString(record, "ingestMode") != null
? record.get("ingestMode").asString() : "", 0, 0, 0, List.of(), false, 0L, "",
true, nullableString(record, "ingestStartedAt"));
}
List<String> failures = record.containsKey("ingestFailures") && !record.get("ingestFailures").isNull()
? record.get("ingestFailures").asList(value -> value.asString())
: List.of();
return new ProjectIngestInfo(ingestedAt,
nullableString(record, "ingestMode") != null ? record.get("ingestMode").asString() : "",
intOrZero(record, "ingestFilesExamined"), intOrZero(record, "ingestFilesPersisted"),
intOrZero(record, "ingestFilesFailed"), failures,
record.containsKey("ingestFailuresTruncated") && !record.get("ingestFailuresTruncated").isNull()
&& record.get("ingestFailuresTruncated").asBoolean(),
record.containsKey("ingestDurationSeconds") && !record.get("ingestDurationSeconds").isNull()
? record.get("ingestDurationSeconds").asLong() : 0L,
nullableString(record, "ingestServerVersion") != null
? record.get("ingestServerVersion").asString() : "",
incomplete, nullableString(record, "ingestStartedAt"));
}
/**
* Paginated variant of {@link #searchIdentifier(String, String, String, String, String, String, boolean, int)}
* for the endpoint.
@@ -1343,29 +1380,15 @@ public class GraphRepository {
}
/**
* Item 126: the last whole-root ingest's facts, or {@code null} when the project shell carries no
* {@code ingestedAt} — i.e. it was last ingested before this was recorded. Null rather than a
* zero-filled record: "never measured" and "measured as none" are different answers, and
* conflating them is the failure mode the item is about.
* Item 129: {@code sourceFile -> sourceHash} for the whole project, the input to a
* {@code changedOnly} refresh's skip decision. A file missing from this map has no stored hash and
* counts as changed.
*/
private static @Nullable ProjectIngestInfo toProjectIngestInfo(Record record) {
@Nullable String ingestedAt = nullableString(record, "ingestedAt");
if (ingestedAt == null) {
return null;
}
List<String> failures = record.containsKey("ingestFailures") && !record.get("ingestFailures").isNull()
? record.get("ingestFailures").asList(value -> value.asString())
: List.of();
return new ProjectIngestInfo(ingestedAt,
nullableString(record, "ingestMode") != null ? record.get("ingestMode").asString() : "",
intOrZero(record, "ingestFilesExamined"), intOrZero(record, "ingestFilesPersisted"),
intOrZero(record, "ingestFilesFailed"), failures,
record.containsKey("ingestFailuresTruncated") && !record.get("ingestFailuresTruncated").isNull()
&& record.get("ingestFailuresTruncated").asBoolean(),
record.containsKey("ingestDurationSeconds") && !record.get("ingestDurationSeconds").isNull()
? record.get("ingestDurationSeconds").asLong() : 0L,
nullableString(record, "ingestServerVersion") != null
? record.get("ingestServerVersion").asString() : "");
public Uni<Map<String, String>> sourceHashes(String project) {
return read(CypherQueries.SOURCE_HASHES, Map.of("project", project),
record -> Map.entry(record.get("sourceFile").asString(), record.get("hash").asString()))
.map(entries -> entries.stream().collect(java.util.stream.Collectors.toMap(
Map.Entry::getKey, Map.Entry::getValue, (a, b) -> a)));
}
private static int intOrZero(Record record, String key) {
@@ -1377,6 +1400,24 @@ public class GraphRepository {
* list is capped at {@link ProjectIngestInfo#MAX_FAILURES} with an explicit truncation flag — the
* count stays exact, so a short list never reads as the whole story.
*/
/**
* Item 129: marks a whole-root pass as in flight. Cleared by {@link #recordProjectIngest}, so an
* interrupted run leaves the marker set and stops looking like a clean graph.
*/
public Uni<Void> markProjectIngestStarted(String project, String mode, String startedAt) {
Map<String, @Nullable Object> params = new HashMap<>();
params.put("name", project);
params.put("mode", mode);
params.put("startedAt", startedAt);
return Uni.createFrom().item(() -> {
try (Session session = driver.session()) {
session.executeWriteWithoutResult(tx -> tx.run(CypherQueries.MARK_PROJECT_INGEST_STARTED, params));
}
return project;
}).replaceWithVoid();
}
public Uni<Void> recordProjectIngest(String project, ProjectIngestInfo ingest) {
Map<String, @Nullable Object> params = new HashMap<>();
params.put("name", project);

View File

@@ -1,5 +1,7 @@
package com.agenticcode.neo4jstore.graph;
import org.jspecify.annotations.Nullable;
import java.util.List;
/**
@@ -40,7 +42,8 @@ import java.util.List;
*/
public record ProjectIngestInfo(String ingestedAt, String mode, int filesExamined, int filesPersisted,
int filesFailed, List<String> failures, boolean failuresTruncated,
long durationSeconds, String serverVersion) {
long durationSeconds, String serverVersion, boolean incomplete,
@Nullable String startedAt) {
/**
* Cap on the persisted {@code failures} list: a misconfigured root can fail thousands of files and

View File

@@ -69,6 +69,37 @@ numbers — re-ingest (refresh) the project to update the graph. Line ranges you
read directly off disk are of course always current; this only guards the API's
own slicing. Copycode/INCLUDE slices are raw pre-expansion file text.
## Refreshing only what changed (item 129)
`POST /api/projects/{p}/refresh` (CLI `ac refresh`) has two ways to avoid re-walking a whole root:
| Form | What it does |
|---------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| `?paths=a/B.java,c/D.java` (`ac refresh --paths a/B.java,c/D.java`) | Re-ingests exactly those relative paths, deep, plus their dependencies. Paths matching no file under the root come back in `unresolved` — a typo'd path is never silently dropped. Does **not** run the deleted-file sweep and does **not** move `ingestedAt`: both need a whole-root walk. |
| `?changedOnly=true` (`ac refresh --changed-only`) | Whole-root walk, but re-parses only files whose content hash differs from the graph's (files with no stored hash count as changed). |
**`changedOnly` is opt-in on purpose.** Three whole-walk behaviours are reduced, and one of them would
be outright corruption if it were hidden:
* **A changed Natural copycode disables skipping for that entire run** (logged). Copycode text is
inlined into the including module *at parse time*, so a module whose `.cpy` changed parses
differently while its own hash is unchanged — skipping it would leave a stale expansion behind with
nothing to indicate it.
* **Duplicate-identity detection** only sees the changed files, so it can confirm duplicates among
them but not discover new ones elsewhere. Existing markers are never cleared.
* **User-exit LoC annotation** (item 47) is re-stamped only on re-parsed files.
Enrichment is project-wide and still runs in full, so this cuts **parse+persist** time only — not the
finalize pass. On small projects the whole deep refresh is already ~30 s, so measure before assuming
a win.
**`ingest.incomplete`** (on `/projects` and `/projects/{p}`) is `true` while a whole-root pass runs and
**stays true if one never finished** — a crash, a container stop, an aborted deep refresh. Before
this, an interrupted deep refresh was indistinguishable from a clean graph: the enrichment steps that
already ran are committed, so queries keep answering, just from a half-updated graph. It cannot
self-heal (a killed process clears nothing) and does not distinguish "running right now" from "died an
hour ago" — both mean the same thing to a caller. A completed refresh clears it.
## Is this project's graph any good? (item 126)
`GET /api/projects` and `GET /api/projects/{p}` (CLI `ac project list` / `ac project show <p>`)

View File

@@ -43,7 +43,7 @@ cost. All probes below were run against a freshly refreshed graph and are reprod
Items 125 and 127 were the two said to make the API return a *wrong* answer rather than a missing
one, which is why they led the list. **125 is fixed (2026-08-18); 127 was retracted the same day —
its probe was not ambiguous, so the answer had been correct all along (see the retraction below).**
**126 is fixed (2026-08-18).** That leaves **128, 129, 130** open.
**126 and 129 are fixed (2026-08-18).** That leaves **128 and 130** open.
- [x] **125. `search/identifier` did not match a type declaration's short name — and silently ignored `contains`** (
fixed 2026-08-18)
@@ -149,7 +149,7 @@ its probe was not ambiguous, so the answer had been correct all along (see the r
"a utility for this already exists, reuse it" — today that answer rests on an index that does
not see type declarations at all (item 125).
- [ ] **129. Refresh is all-or-nothing**
- [x] **129. Refresh was all-or-nothing** (fixed 2026-08-18)
**Symptom.** Twelve changed files; `POST /pur/refresh` examined 2 981. The feedback loop after a
code change is minutes, which discourages the verify-after-change step the workflow depends on.
@@ -157,10 +157,38 @@ its probe was not ambiguous, so the answer had been correct all along (see the r
query silently answers from that state — a hazard the consuming project documents in its own
instructions because the API does not surface it.
**Fix.** `POST /refresh?paths=…` for a targeted re-ingest, or mtime/hash-based incremental
selection by default (item 43's hash-based invalidation may already carry most of this). Second,
record refresh state so an interrupted run is reported as `INCOMPLETE` on the next query rather
than being indistinguishable from a clean graph.
**Delivered**, in three independent pieces (CLI in step: `ac refresh --paths` / `--changed-only`):
1. `POST /refresh?paths=a/B.java,c/D.java` — targeted deep re-ingest of the named files plus their
dependencies, on the existing `ingestFiles` BFS. Paths matching no file come back in
`unresolved` (a typo'd path must not read as a successful refresh), and paths escaping the root
are refused by the same guard the `source` endpoints use. It deliberately skips the deleted-file
sweep and does not move `ingestedAt` — both are only meaningful for a whole-root walk.
2. `ingest.incomplete` on `/projects` — set before a whole-root pass, cleared on success, so a run
that never finished stays flagged. It cannot self-heal (a killed process clears nothing), which
is the correct direction to fail: a false "incomplete" costs one refresh, a false "clean" costs
trust in every answer.
3. `?changedOnly=true` — hash-based skipping, **opt-in rather than the default the item asked for**.
**Why `changedOnly` is not the default.** Implementing it that way would have silently corrupted the
graph in three ways, all found while validating rather than after shipping:
* **Natural copycodes are inlined at parse time.** A module whose `.cpy` changed parses differently
while its own hash is unchanged — so it would be skipped and keep a stale expansion, with nothing
in the graph to indicate it. A changed copycode now disables skipping for the whole run (tested).
This alone rules out "incremental by default" for `upms`, where 137 modules share one copycode.
* **Duplicate detection groups the files it parsed** — a subset can only confirm duplicates among
changed files. Markers are never cleared (the query only MERGEs), so this loses discovery, not
recorded facts.
* **User-exit LoC annotation** is re-stamped only on re-parsed files.
**Honest about the win:** enrichment is project-wide and still runs in full, so this cuts
parse+persist only. The premise in the symptom above has also moved — a full deep `pur` refresh
measured **29 s** on 2026-08-18 (`upms`: 20 m 30 s, the one project where this really pays).
Covered by `TargetedRefreshIT` (8 tests).
**Not delivered:** surfacing `incomplete` on *every* query response, as opposed to on `/projects`.
That is the same envelope change item 126 deferred, and it belongs with item 130's scope echo.
- [ ] **130. Two smaller ones: REST-path routing, and project scope invisible in the answer**