Compare commits

...

19 Commits

Author SHA1 Message Date
Ingo Schnabel
c3eaa0f55e Cleanup 2026-09-04 19:02:31 +02:00
Ingo Schnabel
039ff1e176 Roadmap 2026-08-27 18:24:28 +02:00
Ingo Schnabel
2732d69bd1 No DB-ACCESS without JPA Entities 2026-08-27 17:30:57 +02:00
Ingo Schnabel
8bfef47e1e CONTAINS Cycles 2026-08-23 11:22:53 +03:00
Ingo Schnabel
bd86a01e92 New feature 2026-08-21 23:55:50 +03:00
Ingo Schnabel
7adf7e54e4 Bug fixes 2026-08-20 17:11:16 +03:00
Ingo Schnabel
4b8a441566 New features 2026-08-20 10:32:15 +03:00
Ingo Schnabel
5091cba820 New features 2026-08-19 22:39:10 +03:00
Ingo Schnabel
ec924971a9 New features 2026-08-19 13:18:24 +03:00
Ingo Schnabel
56e20aee3b New features 2026-08-19 10:55:03 +03:00
Ingo Schnabel
29269b83c3 New features 2026-08-19 09:52:24 +03:00
Ingo Schnabel
907286dabb New features 2026-08-18 17:22:31 +03:00
Ingo Schnabel
f5ca2584f3 New features 2026-08-18 13:17:13 +03:00
Ingo Schnabel
0e54cc1589 New features 2026-08-18 12:54:39 +03:00
Ingo Schnabel
3309afe0e1 New features 2026-08-18 12:02:09 +03:00
Ingo Schnabel
cf5d5ea821 Reap feature 2026-08-14 10:25:39 +03:00
Ingo Schnabel
b6007b2096 Reap feature 2026-08-10 13:36:44 +03:00
Ingo Schnabel
97081718fd Java improvements 2026-08-09 13:07:52 +03:00
Ingo Schnabel
93589c2eb9 Java improvements 2026-08-09 10:30:26 +03:00
74 changed files with 9613 additions and 2296 deletions

View File

@@ -1,5 +1,7 @@
package com.agenticcode.cli;
import org.jspecify.annotations.Nullable;
import picocli.CommandLine.Option;
import java.net.URLEncoder;
@@ -56,6 +58,21 @@ abstract class AbstractApiCommand implements Callable<Integer> {
return path + (path.contains("?") ? "&" : "?") + key + "=" + encode(value);
}
/**
* Item 131: says so when the server cut the answer. Written to <b>stderr</b> so piping the body
* into {@code jq} stays clean, and printed at all because the alternative — a capped list that
* looks like the whole set — is what made a real audit report 17 missing annotations that were
* never missing.
*/
private static void warnIfTruncated(ApiClient.ApiResponse response) {
if (!"true".equalsIgnoreCase(String.valueOf(response.header("X-AC-Truncated")))) {
return;
}
@Nullable String total = response.header("X-AC-Total-Count");
System.err.println("note: this answer is truncated" + (total == null ? "" : " (" + total + " rows match)")
+ " — re-run with a larger --limit, page with --offset, or use --count-only");
}
/**
* Prints the response body and returns an exit code derived from the HTTP status.
*/
@@ -64,6 +81,7 @@ abstract class AbstractApiCommand implements Callable<Integer> {
if (!pretty.isBlank()) {
System.out.println(pretty);
}
warnIfTruncated(response);
if (!response.isSuccess()) {
System.err.println("HTTP " + response.statusCode());
return 1;

View File

@@ -43,10 +43,13 @@ import java.util.concurrent.Callable;
DbTableColumnsCommand.class,
EntityColumnsCommand.class,
SearchIdentifierCommand.class,
SearchReferencesCommand.class,
RestEndpointsCommand.class,
ModulesCommand.class,
LocCommand.class,
ModuleDataStructuresCommand.class,
PayloadCommand.class,
CommentsCommand.class,
DispatchTableCommand.class,
DynamicCallsCommand.class,
DigestCommand.class,

View File

@@ -78,13 +78,26 @@ public final class ApiClient {
private ApiResponse send(HttpRequest request) throws IOException, InterruptedException {
HttpResponse<String> response = httpClient.send(request, HttpResponse.BodyHandlers.ofString());
return new ApiResponse(response.statusCode(), response.body());
return new ApiResponse(response.statusCode(), response.body(), response.headers());
}
/**
* Result of an API call.
*/
public record ApiResponse(int statusCode, String body) {
/**
* Item 131: the response now carries its headers, not just the body. Without them a truncated
* answer looked complete on the command line — the same silent cut the item is about, moved one
* layer out.
*/
public record ApiResponse(int statusCode, String body, java.net.http.HttpHeaders headers) {
/**
* @return a header's value, or {@code null} when the server did not send it
*/
public @org.jspecify.annotations.Nullable String header(String name) {
return headers.firstValue(name).orElse(null);
}
public boolean isSuccess() {
return statusCode >= 200 && statusCode < 300;

View File

@@ -12,17 +12,18 @@ import java.util.Objects;
import java.util.concurrent.Callable;
/**
* Groups project management subcommands: create, update, delete, recreate, list.
* Groups project management subcommands: create, update, delete, recreate, show, list.
*/
@Command(
name = "project",
mixinStandardHelpOptions = true,
description = "Create, update, delete, recreate or list projects",
description = "Create, update, delete, recreate, show or list projects",
subcommands = {
ProjectCommand.CreateCommand.class,
ProjectCommand.UpdateCommand.class,
ProjectCommand.DeleteCommand.class,
ProjectCommand.RecreateCommand.class,
ProjectCommand.ShowCommand.class,
ProjectCommand.ListCommand.class
}
)
@@ -193,6 +194,20 @@ final class ProjectCommand implements Callable<Integer> {
}
}
@Command(name = "show", mixinStandardHelpOptions = true,
description = "Show one project's config and its last whole-root ingest (ingestedAt, mode, file counts)")
static final class ShowCommand extends AbstractApiCommand {
@SuppressWarnings("NullAway.Init")
@Parameters(index = "0", description = "Project name")
String name;
@Override
public Integer call() throws Exception {
return printResponse(apiClient().get("/api/projects/" + encode(name)));
}
}
@Command(name = "list", mixinStandardHelpOptions = true, description = "List all projects")
static final class ListCommand extends AbstractApiCommand {

View File

@@ -40,6 +40,16 @@ final class RefreshCommand extends AbstractProjectCommand {
@Option(names = "--neighborhood", description = "Module refresh only: also deep-ingest the module's transitive callers (not just callees/data areas)")
boolean neighborhood;
@Option(names = "--paths", split = ",",
description = "Whole-project refresh only: re-ingest just these relative source paths (repeatable or comma-separated) "
+ "instead of the whole root. Paths matching no file come back under 'unresolved'")
List<String> paths = List.of();
@Option(names = "--changed-only",
description = "Whole-project refresh only: re-parse only files whose content differs from the graph's stored hash. "
+ "Enrichment still runs in full; a changed Natural copycode re-parses everything (its text is inlined at parse time)")
boolean changedOnly;
@Override
public Integer call() throws Exception {
if (name != null && !name.isBlank()) {
@@ -60,6 +70,12 @@ final class RefreshCommand extends AbstractProjectCommand {
return printResponse(apiClient().post(path));
}
String path = projectPath() + "/refresh" + (deep ? "?deep=true" : "");
if (!paths.isEmpty()) {
path = appendQuery(path, "paths", String.join(",", paths));
}
if (changedOnly) {
path = appendQuery(path, "changedOnly", "true");
}
return printResponse(apiClient().post(path));
}
}

View File

@@ -0,0 +1,42 @@
package com.agenticcode.cli;
import org.jspecify.annotations.Nullable;
import picocli.CommandLine.Command;
import picocli.CommandLine.Option;
/**
* Item 130: lists a project's REST endpoints — composed path, HTTP verb, declaring class and handler
* method. Answers "which code runs for this URL" directly, instead of composing the class-level and
* method-level {@code @Path} by hand from two annotation searches.
*/
@Command(name = "rest-endpoints", mixinStandardHelpOptions = true,
description = "List the project's REST endpoints (path, HTTP method, declaring class, handler)")
final class RestEndpointsCommand extends AbstractProjectCommand {
@Option(names = "--module", description = "Only the endpoints declared by this class (identity or short name)")
@Nullable String module;
@Option(names = "--count-only", description = "Print only how many rows match, instead of the rows themselves")
boolean countOnly;
@Option(names = "--limit", description = "Max items to return (default: all)")
int limit = -1;
@Option(names = "--offset", description = "Items to skip")
int offset = -1;
@Override
public Integer call() throws Exception {
try {
String path = appendQuery(projectPath() + "/rest-endpoints", "module", module);
if (countOnly) {
path = appendQuery(path, "countOnly", "true");
}
path = appendQuery(appendQuery(path, "limit", limit), "offset", offset);
return printResponse(apiClient().get(path));
} catch (IllegalStateException e) {
System.err.println(e.getMessage());
return 1;
}
}
}

View File

@@ -19,6 +19,9 @@ final class SearchAnnotationCommand extends AbstractProjectCommand {
@Option(names = "--type", description = "Filter by node type (MODULE, FUNCTION, VARIABLE, DATA_STRUCTURE, DB_TABLE)")
@Nullable String type;
@Option(names = "--count-only", description = "Print only how many rows match, instead of the rows themselves")
boolean countOnly;
@Option(names = "--limit", description = "Max items to return")
int limit = -1;
@@ -32,6 +35,9 @@ final class SearchAnnotationCommand extends AbstractProjectCommand {
if (type != null) {
path += "&type=" + encode(type);
}
if (countOnly) {
path = appendQuery(path, "countOnly", "true");
}
path = appendQuery(appendQuery(path, "limit", limit), "offset", offset);
return printResponse(apiClient().get(path));
} catch (IllegalStateException e) {

View File

@@ -12,9 +12,12 @@ import picocli.CommandLine.Parameters;
@Command(name = "search-identifier", mixinStandardHelpOptions = true, description = "Search for an identifier across all modules, or list all identifiers")
final class SearchIdentifierCommand extends AbstractProjectCommand {
@Parameters(index = "0", arity = "0..1", description = "Identifier name; a leading Natural sigil (# & +) is ignored (omit to list all identifiers)")
@Parameters(index = "0", arity = "0..1", description = "Identifier name — a Java type's short or fully-qualified form; a leading Natural sigil (# & +) is ignored (omit to list all identifiers)")
@Nullable String name;
@Option(names = "--contains", description = "Match names that contain the given name (case-insensitive) instead of equalling it; requires a name")
boolean contains;
@Option(names = "--type", description = "Node type filter (MODULE, FUNCTION, VARIABLE, DATA_STRUCTURE, DB_TABLE)")
@Nullable String type;
@@ -27,6 +30,9 @@ final class SearchIdentifierCommand extends AbstractProjectCommand {
@Option(names = "--source-file", description = "Scope to this exact source file (relative path)")
@Nullable String sourceFile;
@Option(names = "--count-only", description = "Print only how many rows match, instead of the rows themselves")
boolean countOnly;
@Option(names = "--limit", description = "Max items to return")
int limit = -1;
@@ -42,6 +48,12 @@ final class SearchIdentifierCommand extends AbstractProjectCommand {
path = appendQuery(path, "module", module);
path = appendQuery(path, "priorityModule", priorityModule);
path = appendQuery(path, "sourceFile", sourceFile);
if (contains) {
path = appendQuery(path, "contains", "true");
}
if (countOnly) {
path = appendQuery(path, "countOnly", "true");
}
path = appendQuery(appendQuery(path, "limit", limit), "offset", offset);
return printResponse(apiClient().get(path));
} catch (IllegalStateException e) {

View File

@@ -0,0 +1,49 @@
package com.agenticcode.cli;
import org.jspecify.annotations.Nullable;
import picocli.CommandLine.Command;
import picocli.CommandLine.Option;
import picocli.CommandLine.Parameters;
/**
* Item 128: every place a type is mentioned — imports, declared type positions, annotation usages,
* calls, inheritance and wiring — not just its callers. This is what scopes a rename honestly:
* {@code callers} sees calls alone, so a file that only imports or declares the type was invisible.
*/
@Command(name = "references", mixinStandardHelpOptions = true,
description = "Find every reference site of a type (imports, type positions, annotations, calls, inheritance)")
final class SearchReferencesCommand extends AbstractProjectCommand {
@SuppressWarnings("NullAway.Init")
@Parameters(index = "0", description = "Type identity (FQN) or short name")
String name;
@Option(names = "--kind",
description = "Narrow to one kind: CALL, IMPORT, TYPE, ANNOTATION, EXTENDS, IMPLEMENTS, INJECTS, CLASS_LITERAL, INCLUDE")
@Nullable String kind;
@Option(names = "--count-only", description = "Print only how many rows match, instead of the rows themselves")
boolean countOnly;
@Option(names = "--limit", description = "Max items to return")
int limit = -1;
@Option(names = "--offset", description = "Items to skip")
int offset = -1;
@Override
public Integer call() throws Exception {
try {
String path = appendQuery(projectPath() + "/search/references", "name", name);
path = appendQuery(path, "kind", kind);
if (countOnly) {
path = appendQuery(path, "countOnly", "true");
}
path = appendQuery(appendQuery(path, "limit", limit), "offset", offset);
return printResponse(apiClient().get(path));
} catch (IllegalStateException e) {
System.err.println(e.getMessage());
return 1;
}
}
}

View File

@@ -17,6 +17,12 @@ final class SearchValueCommand extends AbstractProjectCommand {
@Option(names = "--contains", description = "Match values that contain the given string, not just exact matches")
boolean contains;
@Option(names = "--include-comments", description = "Also search comment blocks (item 141); their hits come back as kind=COMMENT")
boolean includeComments;
@Option(names = "--count-only", description = "Print only how many rows match, instead of the rows themselves")
boolean countOnly;
@Option(names = "--limit", description = "Max items to return")
int limit = -1;
@@ -30,6 +36,12 @@ final class SearchValueCommand extends AbstractProjectCommand {
if (contains) {
path += "&contains=true";
}
if (includeComments) {
path = appendQuery(path, "includeComments", "true");
}
if (countOnly) {
path = appendQuery(path, "countOnly", "true");
}
path = appendQuery(appendQuery(path, "limit", limit), "offset", offset);
return printResponse(apiClient().get(path));
} catch (IllegalStateException e) {

View File

@@ -4,4 +4,4 @@
server.url=http://localhost:8787
# Stamped by manage-ac.sh (stamp_cli_version) from ac-code-server's agenticcode.version
# at build time. "dev" means this jar wasn't built via manage-ac.sh.
version=198
version=265

View File

@@ -162,6 +162,57 @@ public class AnalysisResource {
IngestSummary run(ProjectInfo project) throws IOException;
}
/**
* Item 131: how many rows match in total, and whether this page leaves some out. Headers rather
* than body fields for the same reason as item 130's scope echo: these endpoints answer with a
* bare JSON array, and turning that into an object would break the web UI's generated client, the
* CLI printers and every agent that indexes {@code [0]}.
*
* <p>Silence here is what did the damage: {@code search/annotation?name=Immutable} returned 50 of
* 95 rows with no total, no flag and no {@code Link}/{@code X-Total-Count} header, and a real
* audit read that page as the whole set — concluding 17 entities had lost the annotation when
* none had.
*/
static final String TOTAL_COUNT = "X-AC-Total-Count";
static final String TRUNCATED = "X-AC-Truncated";
/**
* Item 128: the reference kinds {@code /search/references} can report. Rejecting anything else with
* {@code 400} rather than answering {@code []} — an empty list for a misspelt kind reads as "this
* name is referenced nowhere", which is the failure this endpoint exists to remove.
*/
private static final Set<String> REFERENCE_KINDS = Set.of("CALL", "IMPORT", "TYPE", "ANNOTATION",
"EXTENDS", "IMPLEMENTS", "INJECTS", "CLASS_LITERAL", "INCLUDE");
/**
* Item 131: {@code ?countOnly=true} — a completeness question ("how many classes carry
* {@code @Immutable}?") is a counting question, and answering it with rows costs ~68 tokens each
* for information the caller then throws away.
*/
private static Response countOnly(Page<?> page) {
return Response.ok(Map.of("count", page.total()))
.header(TOTAL_COUNT, page.total())
.header(TRUNCATED, false)
.build();
}
private static boolean isCountOnly(@Nullable Boolean countOnly) {
return Boolean.TRUE.equals(countOnly);
}
/**
* A count-only request must not be capped to a page, or the total it reports is the page size.
*/
private static int countAwareLimit(@Nullable Boolean countOnly, @Nullable Integer limit) {
return isCountOnly(countOnly) ? 1 : effectiveLimit(limit);
}
private Response paged(Page<?> page) {
return Response.ok(page.rows())
.header(TOTAL_COUNT, page.total())
.header(TRUNCATED, page.truncated())
.build();
}
/**
* Default page size for paginated list endpoints, per the documented API design principles.
*/
@@ -363,9 +414,28 @@ public class AnalysisResource {
@APIResponse(responseCode = "200", content = @Content(schema = @Schema(implementation = IngestSummary.class)))
@APIResponse(responseCode = "404", description = "Project not found.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
public Response refresh(@PathParam("project") String project,
@QueryParam("deep") @Nullable Boolean deep) {
@QueryParam("deep") @Nullable Boolean deep,
@Parameter(description = "Item 129: comma-separated relative source paths to re-ingest instead of the whole root — "
+ "the fast loop after editing a few files. Always deep for the named files and their dependencies. "
+ "Requested paths that match no file under the root come back in 'unresolved' rather than being dropped. "
+ "Does not run the deleted-file sweep and does not move the project's ingestedAt, both of which need a whole-root walk.")
@QueryParam("paths") @Nullable String paths,
@Parameter(description = "Item 129: re-parse only files whose content differs from the graph's stored hash. "
+ "Opt-in: enrichment still runs in full (so this cuts parse time only), duplicate detection sees just the "
+ "changed files, user-exit LoC is re-stamped only on re-parsed files, and a changed Natural copycode "
+ "disables the skipping for that run because copycode text is inlined at parse time.")
@QueryParam("changedOnly") @Nullable Boolean changedOnly) {
boolean deepIngest = deep != null && deep;
return withResolvedRoot(project, info -> projectIngestService.refreshProject(info, deepIngest));
if (paths != null && !paths.isBlank()) {
List<String> requested = Arrays.stream(paths.split(",")).map(String::trim).filter(p -> !p.isEmpty()).toList();
if (requested.isEmpty()) {
return ProjectResource.error(Response.Status.BAD_REQUEST, "MISSING_PATHS",
"Query parameter 'paths' was given but contained no path");
}
return withResolvedRoot(project, info -> projectIngestService.refreshPaths(info, requested));
}
return withResolvedRoot(project, info ->
projectIngestService.refreshProject(info, deepIngest, Boolean.TRUE.equals(changedOnly)));
}
/**
@@ -688,6 +758,33 @@ public class AnalysisResource {
: Response.ok(resp).build()));
}
/**
* Item 141: the module's comment blocks, each with the declaration it documents.
*
* <p>Deep-gated on purpose. Comments are produced by the full parse, not by the Tier-1 coarse
* scan, so a {@code CALL_GRAPH}-depth module has none in the graph - and answering {@code []}
* there would be exactly the confident false negative this item exists to remove ("no comments"
* is indistinguishable from "not analysed"). The module is deep-ingested on demand first, and if
* it still is not {@code FULL} the caller gets {@code 409 NOT_DEEPLY_INGESTED} rather than a
* misleading empty list.
*/
@GET
@Path("/modules/{name}/comments")
@APIResponse(responseCode = "200", content = @Content(schema = @Schema(type = SchemaType.ARRAY, implementation = CommentBlock.class)))
@APIResponse(responseCode = "404", description = "Project or module not found.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
@APIResponse(responseCode = "409", description = "Ambiguous module name (AMBIGUOUS_NAME), an unresolved placeholder (NOT_INGESTED), or a module that is not deeply ingested (NOT_DEEPLY_INGESTED).", content = @Content(schema = @Schema(implementation = DeepIngestRequired.class)))
public Uni<Response> moduleComments(@PathParam("project") String project, @PathParam("name") String name,
@Parameter(description = "One comment kind: LINE|BLOCK|JAVADOC (Java), NATURAL_BANNER|NATURAL_INLINE|SAG (Natural). Omitted, every kind but the machine-written SAG directives is returned.")
@QueryParam("kind") @Nullable String kind,
@QueryParam("limit") @Nullable Integer limit, @QueryParam("offset") @Nullable Integer offset,
@QueryParam("sourceFile") @Nullable String sourceFile) {
return withDeeplyIngestedModule(project, name, sourceFile, resolvedName ->
graphRepository.moduleComments(project, resolvedName, anySource(sourceFile),
kind == null || kind.isBlank() ? null : kind.toUpperCase(Locale.ROOT),
uncappedLimit(limit), effectiveOffset(offset))
.map(this::ok));
}
@GET
@Path("/modules/{name}/db-accesses")
@APIResponse(responseCode = "200", content = @Content(schema = @Schema(type = SchemaType.ARRAY, implementation = DbAccess.class)))
@@ -707,19 +804,72 @@ public class AnalysisResource {
});
}
@GET
@Path("/rest-endpoints")
@APIResponse(responseCode = "200", content = @Content(schema = @Schema(type = SchemaType.ARRAY, implementation = RestEndpoint.class)))
@APIResponse(responseCode = "404", description = "Project not found.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
public Uni<Response> restEndpoints(@PathParam("project") String project,
@Parameter(description = "Narrow to the endpoints declared by one class (identity or short name).")
@QueryParam("module") @Nullable String module,
@Parameter(description = "Item 135: report only how many endpoints match, without the rows.")
@QueryParam("countOnly") @Nullable Boolean countOnly,
@QueryParam("limit") @Nullable Integer limit,
@QueryParam("offset") @Nullable Integer offset) {
int effLimit = isCountOnly(countOnly) ? 1 : uncappedLimit(limit);
return withProject(project, () -> graphRepository.restEndpointsPage(project, module,
effLimit, effectiveOffset(offset))
.map(page -> isCountOnly(countOnly) ? countOnly(page) : paged(page)));
}
@GET
@Path("/search/references")
@APIResponse(responseCode = "200", content = @Content(schema = @Schema(type = SchemaType.ARRAY, implementation = ReferenceSite.class)))
@APIResponse(responseCode = "400", description = "Missing 'name', or unknown 'kind'.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
@APIResponse(responseCode = "404", description = "Project not found.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
public Uni<Response> searchReferences(@PathParam("project") String project,
@Parameter(description = "Type identity or short name to find references to.")
@QueryParam("name") @Nullable String name,
@Parameter(description = "Narrow to one kind: CALL, IMPORT, TYPE, ANNOTATION, EXTENDS, IMPLEMENTS, INJECTS, CLASS_LITERAL, INCLUDE.")
@QueryParam("kind") @Nullable String kind,
@Parameter(description = "Item 135: report only how many reference sites match, without the rows.")
@QueryParam("countOnly") @Nullable Boolean countOnly,
@QueryParam("limit") @Nullable Integer limit,
@QueryParam("offset") @Nullable Integer offset) {
if (name == null || name.isBlank()) {
return Uni.createFrom().item(ProjectResource.error(Response.Status.BAD_REQUEST, "MISSING_NAME",
"Query parameter 'name' is required"));
}
@Nullable String upperKind = kind == null ? null : kind.toUpperCase(Locale.ROOT);
if (upperKind != null && !REFERENCE_KINDS.contains(upperKind)) {
return Uni.createFrom().item(ProjectResource.error(Response.Status.BAD_REQUEST, "INVALID_KIND",
"Unknown reference kind '" + kind + "'; expected one of " + REFERENCE_KINDS));
}
return withFanoutWarm(project,
() -> graphRepository.searchReferencesPage(project, name, upperKind,
countAwareLimit(countOnly, limit), effectiveOffset(offset)),
page -> page.rows().stream().map(ReferenceSite::sourceFile)
.filter(sf -> !sf.isEmpty()).distinct().toList(),
page -> isCountOnly(countOnly) ? countOnly(page) : paged(page));
}
@GET
@Path("/search/value")
@APIResponse(responseCode = "200", content = @Content(schema = @Schema(type = SchemaType.ARRAY, implementation = ValueMatch.class)))
@APIResponse(responseCode = "404", description = "Project not found.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
public Uni<Response> searchValue(@PathParam("project") String project, @QueryParam("value") @Nullable String value,
@QueryParam("contains") @Nullable Boolean contains,
@Parameter(description = "Item 141: also search comment blocks. Their hits come back as kind=COMMENT. Off by default - a comment is not the same evidence as a literal in code, and folding it in silently would move every existing completeness count.")
@QueryParam("includeComments") @Nullable Boolean includeComments,
@Parameter(description = "Return only {\"count\": n} instead of the rows (item 131).")
@QueryParam("countOnly") @Nullable Boolean countOnly,
@QueryParam("limit") @Nullable Integer limit, @QueryParam("offset") @Nullable Integer offset) {
if (value == null || value.isBlank()) {
return Uni.createFrom().item(ProjectResource.error(Response.Status.BAD_REQUEST, "MISSING_VALUE",
"Query parameter 'value' is required"));
}
return withProject(project, () -> graphRepository.searchByValue(project, value, Boolean.TRUE.equals(contains),
effectiveLimit(limit), effectiveOffset(offset)).map(this::ok));
return withProject(project, () -> graphRepository.searchByValuePage(project, value, Boolean.TRUE.equals(contains),
Boolean.TRUE.equals(includeComments), countAwareLimit(countOnly, limit), effectiveOffset(offset))
.map(page -> isCountOnly(countOnly) ? countOnly(page) : paged(page)));
}
@GET
@@ -728,6 +878,8 @@ public class AnalysisResource {
@APIResponse(responseCode = "404", description = "Project not found.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
public Uni<Response> searchAnnotation(@PathParam("project") String project, @QueryParam("name") @Nullable String name,
@QueryParam("type") @Nullable String type,
@Parameter(description = "Return only {\"count\": n} instead of the rows (item 131).")
@QueryParam("countOnly") @Nullable Boolean countOnly,
@QueryParam("limit") @Nullable Integer limit, @QueryParam("offset") @Nullable Integer offset) {
if (name == null || name.isBlank()) {
return Uni.createFrom().item(ProjectResource.error(Response.Status.BAD_REQUEST, "MISSING_NAME",
@@ -741,8 +893,9 @@ public class AnalysisResource {
"Unknown node type '" + type + "'"));
}
}
return withProject(project, () -> graphRepository.searchAnnotation(project, name,
type != null ? type.toUpperCase() : null, effectiveLimit(limit), effectiveOffset(offset)).map(this::ok));
return withProject(project, () -> graphRepository.searchAnnotationPage(project, name,
type != null ? type.toUpperCase() : null, countAwareLimit(countOnly, limit), effectiveOffset(offset))
.map(page -> isCountOnly(countOnly) ? countOnly(page) : paged(page)));
}
@GET
@@ -771,6 +924,7 @@ public class AnalysisResource {
@GET
@Path("/search/identifier")
@APIResponse(responseCode = "200", content = @Content(schema = @Schema(type = SchemaType.ARRAY, implementation = IdentifierMatch.class)))
@APIResponse(responseCode = "400", description = "Unknown node 'type', or 'contains' without a 'name'.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
@APIResponse(responseCode = "404", description = "Project not found.", content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
public Uni<Response> searchIdentifier(@PathParam("project") String project, @QueryParam("name") @Nullable String name,
@QueryParam("type") @Nullable String type,
@@ -780,8 +934,19 @@ public class AnalysisResource {
@QueryParam("module") @Nullable String module,
@Parameter(description = "Do not filter, but pin this module's matches to the front so its local declaration survives the limit when a name recurs across many modules.")
@QueryParam("priorityModule") @Nullable String priorityModule,
@Parameter(description = "Match names that contain 'name' (case-insensitive) instead of equalling it, as on /search/value. Matches a Java type's fully-qualified name as well as its short form, so a package fragment also hits. Requires a non-blank 'name'.")
@QueryParam("contains") @Nullable Boolean contains,
@Parameter(description = "Return only {\"count\": n} instead of the rows (item 131).")
@QueryParam("countOnly") @Nullable Boolean countOnly,
@QueryParam("limit") @Nullable Integer limit, @QueryParam("offset") @Nullable Integer offset,
@QueryParam("fields") @Nullable String fields) {
// Item 125: a substring search for nothing is a full node dump, and was previously the
// silent outcome — the parameter did not exist, so JAX-RS dropped it and the caller read the
// resulting [] as "no such identifier".
if (Boolean.TRUE.equals(contains) && (name == null || name.isBlank())) {
return Uni.createFrom().item(ProjectResource.error(Response.Status.BAD_REQUEST, "MISSING_NAME",
"Query parameter 'name' is required when 'contains' is true"));
}
if (type != null) {
try {
NodeType.valueOf(type.toUpperCase());
@@ -791,10 +956,14 @@ public class AnalysisResource {
}
}
return withFanoutWarm(project,
() -> graphRepository.searchIdentifier(project, name,
type != null ? type.toUpperCase() : null, sourceFile, module, priorityModule, effectiveLimit(limit), effectiveOffset(offset)),
matches -> matches.stream().map(IdentifierMatch::sourceFile).filter(sf -> !sf.isEmpty()).distinct().toList(),
matches -> namesOnly(fields) ? ok(identifierNames(matches)) : ok(matches));
() -> graphRepository.searchIdentifierPage(project, name,
type != null ? type.toUpperCase() : null, sourceFile, module, priorityModule,
Boolean.TRUE.equals(contains), countAwareLimit(countOnly, limit), effectiveOffset(offset)),
page -> page.rows().stream().map(IdentifierMatch::sourceFile).filter(sf -> !sf.isEmpty()).distinct().toList(),
page -> isCountOnly(countOnly) ? countOnly(page)
: namesOnly(fields) ? Response.ok(identifierNames(page.rows()))
.header(TOTAL_COUNT, page.total()).header(TRUNCATED, page.truncated()).build()
: paged(page));
}
@GET
@@ -1106,6 +1275,22 @@ public class AnalysisResource {
}));
}
/**
* Item 141: {@link #withIngestedModule} plus the deep tier - for endpoints whose data only exists
* after a full parse. Deep-ingests the module on demand (as the fan-out warm does), then refuses
* with the {@code 409 DeepIngestRequired} hint if it still is not {@code FULL}, so an empty answer
* never masquerades as "analysed, nothing found".
*/
private Uni<Response> withDeeplyIngestedModule(String project, String name, @Nullable String sourceFile,
Function<String, Uni<Response>> action) {
return withIngestedModule(project, name, sourceFile, resolvedName ->
deepIngestCoordinator.ensureDeep(project, resolvedName)
.flatMap(ignored -> graphRepository.moduleIngestState(project, resolvedName, anySource(sourceFile))
.flatMap(state -> state.isFull()
? action.apply(resolvedName)
: Uni.createFrom().item(deepIngestRequired(project, resolvedName, state)))));
}
/**
* Looks up {@code sourceFile}'s stored content hash (item 41), then reads its {@code [startLine,
* endLine]} slice with a stale check (see {@link #withSnippet}).

View File

@@ -0,0 +1,53 @@
package com.agenticcode.codeserver.api;
import jakarta.ws.rs.WebApplicationException;
import jakarta.ws.rs.core.MediaType;
import jakarta.ws.rs.core.Response;
import jakarta.ws.rs.ext.ExceptionMapper;
import jakarta.ws.rs.ext.Provider;
import org.jboss.logging.Logger;
import java.util.Map;
/**
* Item 136: makes an <em>unexpected</em> failure look like every other error this API returns —
* {@code { "error": ..., "code": "INTERNAL_ERROR", "details": {} }} — instead of Quarkus's default
* plain-text error page.
*
* <p>Why it matters: the API's whole contract is "errors are structured JSON with a {@code code} to
* branch on". A {@code Uncoercible: Cannot coerce NULL to Java int} escaping the identifier search
* broke that promise at exactly the moment a client most needs a machine-readable answer — it got an
* HTML-ish body with no {@code code} at all, and no way to tell a server fault from a bad request.
*
* <p><b>Pass-through is load-bearing.</b> JAX-RS picks the most specific mapper for an exception, and
* {@code Throwable} is the least specific one there is: without the {@link WebApplicationException}
* branch below, this mapper would also swallow every {@code 404}/{@code 405}/{@code 415} the runtime
* raises and the deliberate statuses built by {@link ProjectResource#error}, turning correct answers
* into {@code 500}s across the board.
*
* <p>The error id in the message is the correlation handle: the full stack trace goes to the server
* log under the same id, and never into the response — a client has no use for it and a stack trace
* is not something to hand out.
*/
@Provider
public class ApiExceptionMapper implements ExceptionMapper<Throwable> {
private static final Logger LOG = Logger.getLogger(ApiExceptionMapper.class);
@Override
public Response toResponse(Throwable exception) {
// A deliberate status (404 PROJECT_NOT_FOUND, 400 MISSING_NAME, 409 STALE_SOURCE, and the
// runtime's own routing failures) already carries its response. Hand it back untouched.
if (exception instanceof WebApplicationException webApplicationException) {
return webApplicationException.getResponse();
}
String errorId = java.util.UUID.randomUUID().toString();
LOG.errorf(exception, "Unhandled failure, error id %s", errorId);
return Response.status(Response.Status.INTERNAL_SERVER_ERROR)
.type(MediaType.APPLICATION_JSON)
.entity(ErrorResponse.of("INTERNAL_ERROR",
"Unexpected server error (error id " + errorId + "); see the server log for details",
Map.of("errorId", errorId)))
.build();
}
}

View File

@@ -101,6 +101,23 @@ public class ProjectResource {
return graphRepository.listProjects();
}
@GET
@Path("/{project}")
@Operation(summary = "One project's config and last whole-root ingest",
description = "Item 126: the project's configuration plus what its last whole-root ingest did "
+ "(ingestedAt, mode, filesExamined/Persisted/Failed, serverVersion). 'ingest' is null when "
+ "no whole-root ingest has been recorded, which is not the same as one that found nothing.")
@APIResponse(responseCode = "200", description = "The project.")
@APIResponse(responseCode = "404", description = "Project not found.",
content = @Content(schema = @Schema(implementation = ErrorResponse.class)))
public Uni<Response> get(@PathParam("project") String project) {
return graphRepository.getProject(project)
.map(info -> info == null
? error(Response.Status.NOT_FOUND, "PROJECT_NOT_FOUND",
"Project '" + project + "' does not exist")
: Response.ok(info).build());
}
@POST
@Path("/{project}")
@Consumes(MediaType.APPLICATION_JSON)

View File

@@ -0,0 +1,63 @@
package com.agenticcode.codeserver.api;
import com.agenticcode.codeserver.service.ProjectMetadataCache;
import com.agenticcode.neo4jstore.graph.ProjectInfo;
import jakarta.inject.Inject;
import jakarta.ws.rs.container.ContainerRequestContext;
import jakarta.ws.rs.container.ContainerResponseContext;
import jakarta.ws.rs.container.ContainerResponseFilter;
import jakarta.ws.rs.ext.Provider;
import org.jspecify.annotations.Nullable;
import java.util.List;
/**
* Item 130: stamps every project-scoped response with the facts that decide how to read an
* <em>empty</em> one.
*
* <ul>
* <li>{@code X-AC-Exclude-Dirs} — the directories the ingest skipped. "No callers" means "none
* outside tests" in a project excluding {@code test} and "none at all" in one that does not,
* and nothing in the body said which.</li>
* <li>{@code X-AC-Ingested-At} — when the graph was last walked (item 126), so a stale answer is
* visible at the point of use rather than only on the project listing.</li>
* <li>{@code X-AC-Ingest-Incomplete} — a whole-root pass is running or never finished (item 129).</li>
* </ul>
*
* <p><b>Headers, not body fields, and deliberately so.</b> Most endpoints answer with a bare JSON
* array ({@code db-accesses}, {@code functions}, {@code search/identifier}, …); adding a field there
* means restructuring array → object, which breaks the web UI's generated client, the CLI printers
* and every agent that indexes {@code [0]}. A header costs no shape change and covers every endpoint
* at once. The trade-off is real: an agent reading only the JSON body will not see these.
*/
@Provider
public class ProjectScopeHeaderFilter implements ContainerResponseFilter {
static final String EXCLUDE_DIRS = "X-AC-Exclude-Dirs";
static final String INGESTED_AT = "X-AC-Ingested-At";
static final String INCOMPLETE = "X-AC-Ingest-Incomplete";
@Inject
ProjectMetadataCache projects;
@Override
public void filter(ContainerRequestContext request, ContainerResponseContext response) {
@Nullable String project = request.getUriInfo().getPathParameters().getFirst("project");
if (project == null || project.isBlank()) {
return;
}
@Nullable ProjectInfo info = projects.get(project);
if (info == null) {
return;
}
List<String> excludeDirs = info.excludeDirs();
response.getHeaders().putSingle(EXCLUDE_DIRS, excludeDirs.isEmpty() ? "(none)" : String.join(",", excludeDirs));
if (info.ingest() != null) {
response.getHeaders().putSingle(INGESTED_AT, info.ingest().ingestedAt());
response.getHeaders().putSingle(INCOMPLETE, Boolean.toString(info.ingest().incomplete()));
} else {
// Item 126's "never recorded" state, carried through honestly rather than as a false "false".
response.getHeaders().putSingle(INCOMPLETE, "unknown");
}
}
}

View File

@@ -1,9 +1,6 @@
package com.agenticcode.codeserver.service;
import com.agenticcode.neo4jstore.graph.EnrichmentLevel;
import com.agenticcode.neo4jstore.graph.GraphRepository;
import com.agenticcode.neo4jstore.graph.IngestDepth;
import com.agenticcode.neo4jstore.graph.ModuleIngestState;
import com.agenticcode.neo4jstore.graph.*;
import com.agenticcode.parsercore.ast.model.LocMetrics;
import com.agenticcode.parsercore.ast.model.NodeType;
import com.agenticcode.parsercore.ast.spi.CoarseScanner;
@@ -160,6 +157,30 @@ public class AstIngestService {
return graphRepository.distinctSourceFiles(project);
}
/**
* Item 126: records what the last <b>whole-root</b> ingest of {@code project} did, so an agent can
* date an answer and see how complete the graph is without crawling the file system.
*/
public Uni<Void> recordProjectIngest(String project, ProjectIngestInfo ingest) {
return graphRepository.recordProjectIngest(project, ingest);
}
/**
* Item 129: marks a whole-root ingest as in flight, so one that never finishes leaves the graph
* visibly half-updated rather than looking clean.
*/
public Uni<Void> markProjectIngestStarted(String project, String mode, String startedAt) {
return graphRepository.markProjectIngestStarted(project, mode, startedAt);
}
/**
* Item 129: every ingested file's stored content hash, the input to a {@code changedOnly} refresh's
* skip decision.
*/
public Uni<java.util.Map<String, String>> sourceHashes(String project) {
return graphRepository.sourceHashes(project);
}
/**
* Item 43: deletes every node of {@code project} belonging to one of {@code sourceFiles} — the
* deleted-file orphan sweep run after a whole-project refresh.

View File

@@ -1,52 +1,15 @@
package com.agenticcode.codeserver.service;
import java.util.Set;
import com.agenticcode.parsercore.ast.model.ExternalTypeNames;
/**
* Allowlist-by-exclusion of JDK/stdlib and common-framework type names (item J6). A by-name ingest
* follows every referenced type; JDK/framework types (e.g. {@code List}, {@code String},
* {@code Optional}, {@code EntityManager}) never resolve to a file in the project, so without this
* filter they dominate the {@code unresolved} list and needlessly inflate the dependency fan-out.
*
* <p>Matched by uppercased simple name (dependency refs are uppercased). This is a deliberate
* heuristic: a project class deliberately named like a JDK type would also be skipped — acceptable
* and vanishingly rare, especially for Natural modules (8-char codes).
* Item J6 filter for the by-name ingest fan-out. The name list itself lives in
* {@link ExternalTypeNames} (item 128) so the parsers apply the identical exclusion when emitting
* reference edges — two copies of this list would drift, and the drift would show up as placeholder
* nodes appearing and disappearing between ingests.
*/
final class ExternalTypes {
private static final Set<String> NAMES = Set.of(
// java.lang
"OBJECT", "STRING", "CHARSEQUENCE", "INTEGER", "LONG", "DOUBLE", "FLOAT", "BOOLEAN", "BYTE",
"SHORT", "CHARACTER", "NUMBER", "STRINGBUILDER", "STRINGBUFFER", "THREAD", "RUNNABLE",
"EXCEPTION", "RUNTIMEEXCEPTION", "ILLEGALARGUMENTEXCEPTION", "ILLEGALSTATEEXCEPTION",
"THROWABLE", "ERROR", "CLASS", "ENUM", "ITERABLE", "COMPARABLE", "CLONEABLE", "VOID", "MATH",
"SYSTEM", "AUTOCLOSEABLE",
// java.util
"LIST", "ARRAYLIST", "LINKEDLIST", "MAP", "HASHMAP", "LINKEDHASHMAP", "TREEMAP",
"CONCURRENTHASHMAP", "SORTEDMAP", "NAVIGABLEMAP", "SET", "HASHSET", "LINKEDHASHSET", "TREESET",
"SORTEDSET", "COLLECTION", "COLLECTIONS", "OPTIONAL", "OPTIONALINT", "OPTIONALLONG", "ITERATOR",
"QUEUE", "DEQUE", "ARRAYDEQUE", "STACK", "VECTOR", "COMPARATOR", "ARRAYS", "OBJECTS", "UUID",
"DATE", "CALENDAR", "LOCALE", "RANDOM", "SCANNER", "PROPERTIES", "ENUMSET", "ENUMMAP", "BITSET",
// java.util.stream / function
"STREAM", "INTSTREAM", "LONGSTREAM", "DOUBLESTREAM", "COLLECTORS", "FUNCTION", "BIFUNCTION",
"CONSUMER", "BICONSUMER", "SUPPLIER", "PREDICATE", "BIPREDICATE", "UNARYOPERATOR", "BINARYOPERATOR",
// java.time
"LOCALDATE", "LOCALDATETIME", "LOCALTIME", "INSTANT", "DURATION", "PERIOD", "ZONEDDATETIME",
"OFFSETDATETIME", "ZONEID", "DAYOFWEEK", "MONTH", "YEAR", "CHRONOUNIT",
// java.io / nio
"FILE", "PATH", "PATHS", "FILES", "INPUTSTREAM", "OUTPUTSTREAM", "READER", "WRITER",
"BUFFEREDREADER", "IOEXCEPTION", "UNCHECKEDIOEXCEPTION",
// java.math
"BIGDECIMAL", "BIGINTEGER",
// java.util.concurrent / atomic
"ATOMICINTEGER", "ATOMICLONG", "ATOMICBOOLEAN", "ATOMICREFERENCE", "COMPLETABLEFUTURE", "FUTURE",
"EXECUTOR", "EXECUTORSERVICE", "EXECUTORS", "TIMEUNIT", "COUNTDOWNLATCH",
// logging
"LOGGER", "LOGGERFACTORY", "LOG",
// common frameworks: CDI / JPA / Quarkus / JAX-RS reactive
"ENTITYMANAGER", "SESSION", "STATELESSSESSION", "INSTANCE", "EVENT", "PROVIDER", "TYPELITERAL",
"UNI", "MULTI", "RESPONSE", "PANACHEQUERY", "PANACHEENTITY", "PANACHEENTITYBASE");
private ExternalTypes() {
}
@@ -54,6 +17,6 @@ final class ExternalTypes {
* True if {@code upperName} (an uppercased simple type name) is a JDK/stdlib/framework type.
*/
static boolean isExternal(String upperName) {
return NAMES.contains(upperName);
return ExternalTypeNames.isExternal(upperName);
}
}

View File

@@ -3,6 +3,7 @@ package com.agenticcode.codeserver.service;
import com.agenticcode.neo4jstore.graph.EnrichmentLevel;
import com.agenticcode.neo4jstore.graph.IngestDepth;
import com.agenticcode.neo4jstore.graph.ProjectInfo;
import com.agenticcode.neo4jstore.graph.ProjectIngestInfo;
import com.agenticcode.parsercore.ast.model.*;
import com.agenticcode.parsercore.ast.spi.LanguageParser.ParseResult;
import jakarta.enterprise.context.ApplicationScoped;
@@ -13,6 +14,8 @@ import org.jspecify.annotations.Nullable;
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.time.Instant;
import java.time.temporal.ChronoUnit;
import java.util.*;
import java.util.concurrent.*;
import java.util.stream.Collectors;
@@ -43,14 +46,24 @@ public class ProjectIngestService {
private final int maxDeepDepth;
private final int defaultDeepNodes;
private final boolean autoInvalidateEnabled;
// Item 126: stamped onto the project shell with every whole-root ingest, so a graph can be
// attributed to the server release that wrote it.
private final VersionInfo versionInfo;
// Item 130: dropped whenever the project shell changes, so the scope/staleness headers cannot
// report "clean" about a graph whose refresh has just started.
private final ProjectMetadataCache projectMetadata;
public ProjectIngestService(AstIngestService astIngestService,
VersionInfo versionInfo,
ProjectMetadataCache projectMetadata,
@ConfigProperty(name = "agenticcode.ingest.batch-size", defaultValue = "200") int batchSize,
@ConfigProperty(name = "agenticcode.deep-ingest.default-depth", defaultValue = "5") int defaultDeepDepth,
@ConfigProperty(name = "agenticcode.deep-ingest.max-depth", defaultValue = "20") int maxDeepDepth,
@ConfigProperty(name = "agenticcode.deep-ingest.default-nodes", defaultValue = "300") int defaultDeepNodes,
@ConfigProperty(name = "agenticcode.auto-invalidate.enabled", defaultValue = "true") boolean autoInvalidateEnabled) {
this.astIngestService = astIngestService;
this.versionInfo = versionInfo;
this.projectMetadata = projectMetadata;
this.batchSize = Math.max(1, batchSize);
this.maxDeepDepth = Math.max(1, maxDeepDepth);
this.defaultDeepDepth = Math.min(Math.max(1, defaultDeepDepth), this.maxDeepDepth);
@@ -281,6 +294,18 @@ public class ProjectIngestService {
return new IngestSummary.Truncation(reason, depthLimit, nodeLimit, hint);
}
/**
* @return {@code file}'s content hash, or {@code ""} when it cannot be read — which never equals a
* stored hash, so an unreadable file is re-parsed (and fails loudly there) rather than skipped.
*/
private static String hashOf(Path file) {
try {
return SourceHash.of(SourceFiles.read(file));
} catch (IOException e) {
return "";
}
}
/**
* Walks the entire project root and ingests every ingestible file. {@code deep} selects the
* enrichment level: {@code deep=false} (default) runs the fast call-graph pass (placeholder
@@ -291,7 +316,7 @@ public class ProjectIngestService {
* {@link IngestDepth#FULL}).
*/
public IngestSummary ingestAll(ProjectInfo project, boolean deep) throws IOException {
return ingestRoot(project, deep ? EnrichmentLevel.FULL : EnrichmentLevel.CALL_GRAPH, false);
return ingestRoot(project, deep ? EnrichmentLevel.FULL : EnrichmentLevel.CALL_GRAPH, false, false);
}
/**
@@ -304,7 +329,7 @@ public class ProjectIngestService {
* resolve; field-level dataflow is deferred to the on-demand deep ingest.
*/
public IngestSummary scanTier1(ProjectInfo project) throws IOException {
return ingestRoot(project, EnrichmentLevel.CALL_GRAPH, true);
return ingestRoot(project, EnrichmentLevel.CALL_GRAPH, true, false);
}
/**
@@ -314,7 +339,7 @@ public class ProjectIngestService {
* caller to deep-ingest a program first. Intended as the fast first pass after project creation.
*/
public IngestSummary ingestCallGraph(ProjectInfo project) throws IOException {
return ingestRoot(project, EnrichmentLevel.CALL_GRAPH, false);
return ingestRoot(project, EnrichmentLevel.CALL_GRAPH, false, false);
}
/**
@@ -327,8 +352,32 @@ public class ProjectIngestService {
* wipe.)
*/
public IngestSummary refreshProject(ProjectInfo project, boolean deep) throws IOException {
return refreshProject(project, deep, false);
}
/**
* Item 129: {@code changedOnly} re-parses only files whose content hash differs from the graph's
* (plus files with no stored hash). <b>Opt-in, and deliberately not the default</b> — three
* whole-walk behaviours are reduced in this mode:
*
* <ul>
* <li><b>Natural copycodes are inlined at parse time</b>, so a module whose {@code .cpy} changed
* parses differently while its own hash is unchanged. A changed copycode therefore
* <b>disables skipping for the whole run</b> (logged) rather than silently keeping stale
* expansions — the one case where "incremental" would corrupt the graph outright.</li>
* <li><b>Duplicate identity detection</b> groups the files it parsed; with a subset it can only
* confirm duplicates among changed files. Existing markers are never cleared (the query only
* MERGEs), so this loses discovery, not recorded facts.</li>
* <li><b>User-exit LoC annotation</b> (item 47) is re-stamped only on files that were re-parsed:
* a changed user-exit twin does not refresh an unchanged generated module's metrics.</li>
* </ul>
*
* <p>Enrichment is project-wide and still runs in full, so this cuts parse+persist time only.
*/
public IngestSummary refreshProject(ProjectInfo project, boolean deep, boolean changedOnly) throws IOException {
long startedAt = System.nanoTime();
IngestSummary summary = deep ? ingestAll(project, true) : ingestCallGraph(project);
IngestSummary summary = ingestRoot(project, deep ? EnrichmentLevel.FULL : EnrichmentLevel.CALL_GRAPH,
false, changedOnly);
sweepDeletedFileOrphans(project);
LOG.infof("Refresh finished: project='%s', mode=%s, files=%d, modules=%d, failed=%d, %d s",
project.name(), deep ? "deep" : "call-graph", summary.examinedFiles().size(),
@@ -372,17 +421,69 @@ public class ProjectIngestService {
return ingestModule(project, moduleName, maxDepth, maxNodes);
}
/**
* Item 129: re-ingests <b>only the named files</b> (relative paths), for the common case of a code
* change touching a handful of files where a whole-root refresh re-examines thousands.
*
* <p>Deep by construction: it runs the same BFS deep ingest the fan-out warm uses, so the named
* files and their dependencies land {@code FULL}. Two things it deliberately does <b>not</b> do,
* because they are only meaningful for a whole-root walk: the deleted-file sweep (item 43 —
* nothing here says which files disappeared) and the project-shell ingest stamp (item 126 — this
* walks a fraction of the tree, and moving {@code ingestedAt} would advertise the project as
* freshly walked).
*
* <p>Paths that match no file under the root are returned in the summary's {@code unresolved} list
* rather than dropped: "I ingested 2 of your 3 files" must be visible, or a typo'd path reads as a
* successful refresh.
*/
public IngestSummary refreshPaths(ProjectInfo project, Collection<String> sourceFiles) throws IOException {
Path root = Path.of(project.root()).toAbsolutePath().normalize();
List<String> requested = sourceFiles.stream()
.map(String::trim)
.filter(sf -> !sf.isEmpty())
.distinct()
.toList();
// Same guard as the source endpoints: these paths are client-supplied, so a '..' or an
// absolute path must not reach outside the project root.
List<String> unresolved = requested.stream()
.filter(sf -> !root.resolve(sf).normalize().startsWith(root)
|| !Files.isRegularFile(root.resolve(sf).normalize())
// A path that exists but is not an ingestible source file (pom.xml, a README)
// must be reported, not accepted: it was silently listed as examined while
// nothing about it could ever be ingested, which is precisely the "I ingested
// 2 of your 3 files" invisibility this list exists to prevent.
|| SourceFiles.classify(root.resolve(sf).normalize()) == null)
.toList();
List<String> ingestable = requested.stream().filter(sf -> !unresolved.contains(sf)).toList();
if (ingestable.isEmpty()) {
return new IngestSummary(0, unresolved, List.of(), List.of(), List.of(), null);
}
IngestSummary summary = ingestFiles(project, ingestable, null);
LOG.infof("Targeted refresh of '%s': %d file(s) requested, %d ingested, %d unresolved",
project.name(), requested.size(), summary.ingested(), unresolved.size());
return new IngestSummary(summary.ingested(), unresolved, summary.duplicates(), summary.failed(),
ingestable, summary.truncation());
}
/**
* Shared whole-root ingest. {@code level} selects how far enrichment goes, and the
* {@link IngestDepth} the ingested modules are tagged with ({@code FULL} only when field-level
* resolution ran, else {@code CALL_GRAPH}).
*/
private IngestSummary ingestRoot(ProjectInfo project, EnrichmentLevel level, boolean coarse) throws IOException {
private IngestSummary ingestRoot(ProjectInfo project, EnrichmentLevel level, boolean coarse,
boolean changedOnly) throws IOException {
long startedAt = System.nanoTime();
// Item 129: mark the pass in flight before touching anything. If it never reaches the record
// call at the end — crash, container stop, an aborted deep refresh — the marker stays set and
// every later answer can say the graph is half-updated instead of looking clean.
markIngestStarted(project, coarse ? "tier1" : level.name().toLowerCase(Locale.ROOT));
Path root = Path.of(project.root());
List<String> excludeDirs = ingestExcludeDirs(project);
List<Candidate> candidates = walk(root, excludeDirs);
List<Candidate> allCandidates = walk(root, excludeDirs);
CopycodeLibrary copycodes = CopycodeLibrary.scan(root, excludeDirs);
// Item 129: opt-in skip of files whose content is byte-identical to what the graph holds.
List<Candidate> candidates = changedOnly ? changedCandidates(project, root, excludeDirs, allCandidates)
: allCandidates;
Map<String, LocMetrics> userExit = UserExitMetrics.scan(root, project.userExitDir(), project.excludeDirs(), astIngestService);
List<String> examinedFiles = candidates.stream()
@@ -479,14 +580,128 @@ public class ProjectIngestService {
astIngestService.markIngestDepth(project.name(),
level.resolveFields() ? IngestDepth.FULL : IngestDepth.CALL_GRAPH, null).await().indefinitely();
}
String mode = coarse ? "tier1" : level.name().toLowerCase(Locale.ROOT);
long durationSeconds = TimeUnit.NANOSECONDS.toSeconds(System.nanoTime() - startedAt);
LOG.infof("Project ingest finished: project='%s', mode=%s, files=%d, persisted=%d, failed=%d, "
+ "duplicates=%d, %d s",
project.name(), coarse ? "tier1" : level.name().toLowerCase(Locale.ROOT), examinedFiles.size(),
ingested, failed.size(), duplicates.size(),
TimeUnit.NANOSECONDS.toSeconds(System.nanoTime() - startedAt));
project.name(), mode, examinedFiles.size(),
ingested, failed.size(), duplicates.size(), durationSeconds);
// Item 126: this is the only place that stamps the project shell, and it is reached only by the
// three whole-root passes (Tier-1 scan, call-graph refresh, deep refresh). By-name and fan-out
// ingests deliberately do not come through here — they walk a fraction of the tree, and moving
// ingestedAt for them would report the project as freshly walked when one module was deepened.
// Item 129: filesExamined counts what this pass actually walked. In changedOnly mode that is
// the changed subset, and the log line above says so — a smaller number here is the point, not
// a sign of a short walk.
recordIngest(project, mode, examinedFiles.size(), ingested, failed, durationSeconds);
return new IngestSummary(ingested, List.of(), duplicates, failed, examinedFiles, null);
}
/**
* Item 129: the subset of {@code candidates} whose on-disk content differs from the hash stored in
* the graph (or that has no stored hash at all — a new file, or one ingested before hashes existed).
*
* <p>Returns <b>every</b> candidate — i.e. skips nothing — when a Natural copycode has changed.
* Copycode text is inlined into the including module at parse time, so those modules parse
* differently while their own hashes are unchanged; skipping them would leave stale expansions in
* the graph with nothing to indicate it. Reading the copycodes' own hashes is not enough to know
* <em>which</em> modules include them at this point in the walk, and guessing wrong is silent
* corruption, so the whole optimisation stands down for that run.
*/
private List<Candidate> changedCandidates(ProjectInfo project, Path root, List<String> excludeDirs,
List<Candidate> candidates) {
Map<String, String> stored = astIngestService.sourceHashes(project.name()).await().indefinitely();
// Only Natural inlines copycodes at parse time, so only Natural needs the stand-down. Running
// the check for a Java project was actively harmful: `ac` carries .cpy files as Natural *test
// fixtures* that its Java walk never ingests, so they had no stored hash, counted as changed,
// and disabled skipping entirely — 487 of 509 unchanged files re-parsed. An unknown language
// (a legacy project) keeps the conservative behaviour.
if (!"java".equalsIgnoreCase(project.language() == null ? "" : project.language())
&& copycodeChanged(root, excludeDirs, stored)) {
LOG.infof("changedOnly refresh of '%s': a copycode changed, so every file is re-parsed "
+ "(copycode text is inlined at parse time; skipping would keep stale expansions)",
project.name());
return candidates;
}
List<Candidate> changed = new ArrayList<>();
for (Candidate candidate : candidates) {
String relative = relativeSourceFile(root, candidate.file());
@Nullable String known = stored.get(relative);
if (known == null || !known.equals(hashOf(candidate.file()))) {
changed.add(candidate);
}
}
LOG.infof("changedOnly refresh of '%s': %d of %d files changed since the last ingest",
project.name(), changed.size(), candidates.size());
return changed;
}
/**
* @return whether any {@code .cpy} member's content differs from the hash the graph holds for it.
* A copycode with no stored hash counts as changed: it may never have been ingested, and assuming
* otherwise is the unsafe direction.
*/
private boolean copycodeChanged(Path root, List<String> excludeDirs, Map<String, String> stored) {
try (Stream<Path> stream = Files.walk(root)) {
return stream.filter(Files::isRegularFile)
.filter(f -> f.getFileName().toString().toLowerCase(Locale.ROOT).endsWith(".cpy"))
.filter(f -> !SourceFiles.isExcluded(root, f, excludeDirs))
.anyMatch(f -> {
@Nullable String known = stored.get(relativeSourceFile(root, f));
return known == null || !known.equals(hashOf(f));
});
} catch (IOException e) {
LOG.warnf("Could not scan copycodes under '%s' for changes (%s); re-parsing everything",
root, e.toString());
return true;
}
}
/**
* Item 129: stamps the in-flight marker. Never fatal — a project whose bookkeeping cannot be written
* must still be ingestable; the cost of failing here is only that an interruption would go unmarked.
*/
private void markIngestStarted(ProjectInfo project, String mode) {
try {
astIngestService.markProjectIngestStarted(project.name(), mode,
Instant.now().truncatedTo(ChronoUnit.SECONDS).toString()).await().indefinitely();
projectMetadata.invalidate(project.name());
} catch (RuntimeException e) {
LOG.warnf(e, "Could not mark ingest start for project '%s'; an interrupted run will not be flagged",
project.name());
}
}
/**
* Item 126: writes the whole-root ingest's outcome onto the {@code (:Project)} shell. Failure
* <em>paths</em> are capped at {@link ProjectIngestInfo#MAX_FAILURES} while the count stays exact,
* so a truncated list can never be read as "these were all of them".
*
* <p>Never fatal: the graph is already written, and losing the bookkeeping must not turn a
* successful ingest into a failed request. A warning is logged instead — the missing metadata then
* shows up as {@code ingest: null}, which reads as "not recorded" rather than as a false fact.
*/
private void recordIngest(ProjectInfo project, String mode, int filesExamined, int filesPersisted,
List<IngestSummary.Failure> failed, long durationSeconds) {
List<String> failurePaths = failed.stream()
.map(f -> relativeSourceFile(Path.of(project.root()), Path.of(f.path())))
.limit(ProjectIngestInfo.MAX_FAILURES)
.toList();
ProjectIngestInfo ingest = new ProjectIngestInfo(
Instant.now().truncatedTo(ChronoUnit.SECONDS).toString(), mode,
filesExamined, filesPersisted, failed.size(), failurePaths,
failed.size() > failurePaths.size(), durationSeconds, versionInfo.version(),
// Reaching here means the pass completed; the write clears the in-flight marker.
false, null);
try {
astIngestService.recordProjectIngest(project.name(), ingest).await().indefinitely();
projectMetadata.invalidate(project.name());
} catch (RuntimeException e) {
LOG.warnf(e, "Could not record ingest metadata for project '%s'; it will report ingest=null",
project.name());
}
}
/**
* Ingests {@code moduleName} and its transitive dependencies using the configured default depth
* and node budget. See {@link #ingestModule(ProjectInfo, String, Integer, Integer)}.

View File

@@ -0,0 +1,72 @@
package com.agenticcode.codeserver.service;
import com.agenticcode.neo4jstore.graph.GraphRepository;
import com.agenticcode.neo4jstore.graph.ProjectInfo;
import jakarta.enterprise.context.ApplicationScoped;
import org.jboss.logging.Logger;
import org.jspecify.annotations.Nullable;
import java.util.Map;
import java.util.concurrent.ConcurrentHashMap;
/**
* Item 130: a short-lived cache of the {@code (:Project)} shell, so the scope/staleness response
* headers cost no database round trip per request.
*
* <p>Without it the header filter would add a Neo4j read to <em>every</em> call — unacceptable for
* endpoints that answer in tens of milliseconds. The TTL is short, and the ingest path
* {@link #invalidate(String) invalidates} explicitly rather than waiting it out: a stale
* {@code ingestedAt} is a cosmetic lag, but a stale {@code incomplete=false} while a refresh is
* running would point exactly the wrong way — it would say "clean" about a half-updated graph.
*
* <p>A read failure yields {@code null} (no headers) rather than an error: this is decoration on
* someone else's answer, and it must never turn a good response into a failed one.
*/
@ApplicationScoped
public class ProjectMetadataCache {
private static final Logger LOG = Logger.getLogger(ProjectMetadataCache.class);
/**
* How long a cached shell may be reused. Short enough that a missed invalidation self-corrects
* within one human-noticeable moment.
*/
private static final long TTL_MILLIS = 10_000;
private final GraphRepository graphRepository;
private final Map<String, Entry> cache = new ConcurrentHashMap<>();
public ProjectMetadataCache(GraphRepository graphRepository) {
this.graphRepository = graphRepository;
}
/**
* @return the project's shell, or {@code null} if it does not exist or could not be read
*/
public @Nullable ProjectInfo get(String project) {
Entry cached = cache.get(project);
long now = System.currentTimeMillis();
if (cached != null && now - cached.readAt() < TTL_MILLIS) {
return cached.info();
}
try {
@Nullable ProjectInfo info = graphRepository.getProject(project).await().indefinitely();
cache.put(project, new Entry(info, now));
return info;
} catch (RuntimeException e) {
LOG.debugf(e, "Could not read project '%s' for response headers", project);
return null;
}
}
/**
* Drops the cached shell — called when an ingest changes it, so the in-flight marker is visible
* immediately rather than up to {@link #TTL_MILLIS} late.
*/
public void invalidate(String project) {
cache.remove(project);
}
private record Entry(@Nullable ProjectInfo info, long readAt) {
}
}

View File

@@ -3,7 +3,7 @@ quarkus.http.port=8787
# AgenticCode's own release counter (not the Maven project version) — bump this by hand for each
# release. Single source of truth for the startup log line, GET /api/version, and the OpenAPI
# info version (referenced below via property expression, not duplicated).
agenticcode.version=198
agenticcode.version=265
# OpenAPI / Swagger UI (item 48) — the generated spec is the contract the web-UI TS client
# is generated against. Served at /q/openapi (yaml/json); Swagger UI at /q/swagger-ui in dev.
mp.openapi.extensions.smallrye.info.title=AgenticCode API

View File

@@ -0,0 +1,52 @@
package com.agenticcode.codeserver.api;
import jakarta.ws.rs.NotFoundException;
import jakarta.ws.rs.WebApplicationException;
import jakarta.ws.rs.core.Response;
import org.junit.jupiter.api.Test;
import static org.junit.jupiter.api.Assertions.*;
/**
* Item 136: an unexpected failure must still leave the server as structured JSON, and a deliberate
* one must pass through untouched.
*
* <p>The pass-through half is the one worth testing hardest: {@code Throwable} is the least specific
* exception type there is, so without it this mapper would convert every {@code 404}/{@code 400} the
* API answers today into a {@code 500}.
*/
class ApiExceptionMapperTest {
private final ApiExceptionMapper mapper = new ApiExceptionMapper();
@Test
void anUnexpectedFailureBecomesAStructuredInternalError() {
Response response = mapper.toResponse(new IllegalStateException("boom"));
assertEquals(500, response.getStatus());
ErrorResponse body = (ErrorResponse) response.getEntity();
assertEquals("INTERNAL_ERROR", body.code());
assertTrue(body.details().containsKey("errorId"), "the correlation id belongs in details");
assertFalse(body.error().contains("boom"),
"the exception message may name internals and must not be echoed to the client");
}
@Test
void aDeliberateNotFoundIsHandedBackUnchanged() {
Response built = ProjectResource.error(Response.Status.NOT_FOUND, "PROJECT_NOT_FOUND", "no such project");
Response response = mapper.toResponse(new WebApplicationException(built));
assertEquals(404, response.getStatus());
assertEquals("PROJECT_NOT_FOUND", ((ErrorResponse) response.getEntity()).code());
}
/**
* The runtime's own routing failures are {@code WebApplicationException}s too and must keep their
* status — an unknown path stays a 404, not a 500.
*/
@Test
void aRoutingFailureKeepsItsStatus() {
assertEquals(404, mapper.toResponse(new NotFoundException()).getStatus());
}
}

View File

@@ -95,7 +95,7 @@ class CoalescingIT {
assertTrue(state.isFull(), "the module is FULL after concurrent deep ingests");
assertEquals(IngestStatus.INGESTED.name(), state.status());
List<IdentifierMatch> realModules = graphRepository.searchIdentifier(PROJECT, MODULE, "MODULE", null, null, null, 50, 0)
List<IdentifierMatch> realModules = graphRepository.searchIdentifier(PROJECT, MODULE, "MODULE", null, null, null, false, 50, 0)
.await().indefinitely().stream()
.filter(m -> !m.sourceFile().isEmpty())
.toList();

View File

@@ -44,6 +44,8 @@ class CopycodeExpansionIT {
copyFixture("fixtures/natural/copycode/MYLDA.lda");
copyFixture("fixtures/natural/copycode/QUALCOPY.cpy");
copyFixture("fixtures/natural/copycode/QUALHOST.nat");
copyFixture("fixtures/natural/copycode/BROWSECPY.cpy");
copyFixture("fixtures/natural/copycode/BROWSEHOST.nat");
given().contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
@@ -171,6 +173,42 @@ class CopycodeExpansionIT {
+ "Full response was: " + body.getList("$"));
}
/**
* Items 120/121/123 in one fixture, because in the corpus they occur in one statement.
*
* <p>{@code BROWSEHOST} pulls in a browse copycode whose {@code CALLNAT} target exists only after
* substitution. Three separate defects each broke it on their own:
* <ul>
* <li><b>120</b> — the arguments run onto line 12, and only line 11 was read, so {@code &2&}
* survived substitution verbatim and the call was dropped;</li>
* <li><b>121</b> — {@code '''AGNT-CHG-CMP-SP'''} is <em>one</em> literal, but split into three
* arguments, shifting every later position by one;</li>
* <li><b>123</b> — the target is written {@code '"YAGCHBN0"'}, and the {@code CALLNAT} pattern
* accepted only {@code '} as a string delimiter.</li>
* </ul>
*
* <p>Measured over {@code upms}: fixing 120 alone recovers <b>0</b> edges — its own motivating
* example is a 123 case. All three together recover 2672 copycode-derived call pairs (+49%), 0
* lost. That is why the fixture asserts the call surfaces rather than asserting three mechanisms.
*/
@Test
void aBrowseIncludeWithMultiLineEscapedArgumentsResolvesItsCall() {
JsonPath body = given().pathParam("name", "BROWSEHOST")
.when().get("/api/projects/" + PROJECT + "/modules/{name}/callees")
.then().statusCode(200)
.body("items.name", hasItem("YAGCHBN0"))
// Item 121's phantom: the sort key bound to &2& and became a MODULE node of its own,
// so the endpoint reported a call that does not exist while hiding the one that does.
.body("items.name", not(hasItem("AGNT-CHG-CMP-SP")))
.extract().jsonPath();
Map<String, Object> site = body.getMap("items.find { it.name == 'YAGCHBN0' }.sites[0]");
assertEquals("BROWSECPY", site.get("viaCopycode"));
assertEquals(8, site.get("lineNo"), "the CALLNAT is on line 8 of BROWSECPY.cpy");
assertEquals(11, site.get("includedAt"),
"the INCLUDE is on line 11 of BROWSEHOST.nat — line 12 is its argument continuation");
}
/**
* Item 70: placeholder resolution must not throw the copycode provenance away.
*

View File

@@ -0,0 +1,214 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import jakarta.inject.Inject;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import org.neo4j.driver.Driver;
import org.neo4j.driver.Session;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.List;
import java.util.Map;
import static io.restassured.RestAssured.given;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertTrue;
/**
* Item 75-B: a copycode-resident node belongs to the module that includes it, not to the copycode.
*
* <p>{@code CopycodePreprocessor} splices a {@code .cpy} body into every including module before
* parsing, and the nodes it produces used to be MERGEd on {@code (type, name, sourceFile)} — so all
* includers shared <em>one</em> node. Two consequences, both of which this test pins:
*
* <ul>
* <li><b>{@code CONTAINS} became cyclic (items 75/75-C).</b> A copycode may open a block it does
* not close (the {@code END-FOR} lives in the includer — {@code YFRAMBC0.cpy} in {@code upms}
* does exactly this), so one node collected the nesting context of every expansion. Measured on
* {@code upms}, every cycle pairs a copycode node with statements of a <em>single</em> host that
* includes it repeatedly ({@code JX0031N0.nat} includes {@code YFRAMBC0} 16 times), which is why
* identity has to be per <em>expansion site</em> ({@code <hostFile>#<includePath>}) and not per
* module. {@code HOSTB} in this fixture reproduces that shape with two nested sites.</li>
* <li><b>The stale-edge reaps could not run on it.</b> Items 124/86/106 key on the source node's
* file; with a shared node, reaping during one module's refresh would delete edges the other
* includers contributed and never re-create them. The fixture's second half asserts the
* opposite property now holds: refreshing one host leaves the other host's copies alone.</li>
* </ul>
*/
@QuarkusTest
class CopycodeNodeOwnershipIT {
private static final String PROJECT = "item75b-copycode-ownership";
private static final String CPY = "SHAREDBLK.cpy";
@TempDir
static Path root;
@Inject
Driver driver;
@BeforeAll
static void createProject() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
// Opens a FOR it never closes: the END-FOR is supplied by each including module. This is the
// shape that made the shared node cyclic.
write(CPY, """
FOR #I = 1 TO 10
CALLNAT 'SHAREDSUB' #I
""");
write("HOSTA.nat", """
DEFINE DATA
LOCAL
1 #I (I2)
END-DEFINE
*
INCLUDE SHAREDBLK
END-FOR
*
END
""");
// Item 75-C: the same copycode included TWICE, at different nesting depths. Both expansions
// used to map onto one node (same file, same line), so that node was contained by the IF of
// the second site and contained the IF itself — the CONTAINS 2-cycle of item 75.
write("HOSTB.nat", """
DEFINE DATA
LOCAL
1 #I (I2)
END-DEFINE
*
INCLUDE SHAREDBLK
IF #I = 2
INCLUDE SHAREDBLK
END-FOR
END-IF
END-FOR
*
END
""");
write("SHAREDSUB.nat", """
DEFINE DATA
PARAMETER
1 #I (I2)
END-DEFINE
*
END
""");
given().contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
.when().post("/api/projects/" + PROJECT)
.then().statusCode(201);
given().when().post("/api/projects/" + PROJECT + "/refresh?deep=true")
.then().statusCode(200);
}
private static void write(String fileName, String content) {
try {
Files.write(root.resolve(fileName), content.getBytes(StandardCharsets.UTF_8));
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
private long count(String cypher) {
try (Session session = driver.session()) {
return session.run(cypher, Map.of("p", PROJECT)).single().get("c").asLong();
}
}
/**
* Guards against a vacuous pass: if the copycode were not expanded at all, every assertion below
* would hold trivially.
*/
@Test
void theCopycodeIsActuallyExpandedIntoBothHosts() {
List<String> owners;
try (Session session = driver.session()) {
owners = session.run("""
MATCH (n:AstNode {project: $p})
WHERE n.sourceFile ENDS WITH '.cpy'
RETURN DISTINCT n.ownerModule AS owner ORDER BY owner
""", Map.of("p", PROJECT))
.list(r -> r.get("owner").asString());
}
assertEquals(3, owners.size(),
"one expansion site for HOSTA and two for HOSTB, each with its own identity: " + owners);
assertTrue(owners.stream().allMatch(o -> o.startsWith("HOSTA.nat#") || o.startsWith("HOSTB.nat#")),
"an owner is <hostFile>#<includePath>: " + owners);
}
/**
* The point of the item: no node is shared between the two includers.
*/
@Test
void eachIncluderGetsItsOwnCopyOfTheCopycodeNodes() {
long shared = count("""
MATCH (n:AstNode {project: $p})
WHERE n.sourceFile ENDS WITH '.cpy' AND n.ownerModule = ''
RETURN count(n) AS c
""");
assertEquals(0, shared, "a copycode-resident node must carry the including module as its owner");
}
/**
* A MODULE declared inside a copycode must stay shared — every module lookup binds
* {@code (project, name, sourceFile)} and never {@code ownerModule}, so an owned MODULE node
* would be invisible to them. Asserted on the whole project because the fixture has no such
* module; the invariant is what matters, and it must hold for every node of that type.
*/
@Test
void moduleAndTableNodesAreNeverOwned() {
long owned = count("""
MATCH (n:AstNode {project: $p})
WHERE n.ownerModule <> '' AND n.type IN ['MODULE', 'DB_TABLE']
RETURN count(n) AS c
""");
assertEquals(0, owned, "MODULE/DB_TABLE nodes must remain shared");
}
/**
* The symptom item 75 is named after. Not vacuous any more: HOSTB includes the copycode at two
* differently-nested sites, which is exactly what makes the shared node contain the block that
* contains it.
*/
@Test
void containsIsAcyclic() {
assertEquals(0, count("""
MATCH (n:AstNode {project: $p})-[:CONTAINS]->(n)
RETURN count(n) AS c
"""), "no node may contain itself");
assertEquals(0, count("""
MATCH (a:AstNode {project: $p})-[:CONTAINS]->(b:AstNode {project: $p})-[:CONTAINS]->(a)
RETURN count(a) AS c
"""), "no two nodes may contain each other");
}
/**
* The reap hazard: {@code DELETE_STALE_FILE_NODES} and the item-86/106/124 edge reaps key on the
* source file, and the copycode file is in every includer's fresh-file set. Keyed on the file
* alone, refreshing HOSTA would delete HOSTB's copies (their {@code ingestGen} is a transaction
* old) together with their edges, and nothing would rebuild them.
*/
@Test
void refreshingOneHostLeavesTheOtherHostsCopiesIntact() {
String cypher = """
MATCH (n:AstNode {project: $p})
WHERE n.sourceFile ENDS WITH '.cpy' AND n.ownerModule STARTS WITH 'HOSTB.nat#'
OPTIONAL MATCH (n)-[r]-()
RETURN count(DISTINCT n) + count(r) AS c
""";
long before = count(cypher);
assertTrue(before > 0, "HOSTB must own copycode nodes before the refresh");
given().when().post("/api/projects/" + PROJECT + "/refresh?paths=HOSTA.nat")
.then().statusCode(200);
assertEquals(before, count(cypher), "refreshing HOSTA must not touch HOSTB's copycode nodes or edges");
}
}

View File

@@ -0,0 +1,206 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import io.restassured.path.json.JsonPath;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.List;
import static io.restassured.RestAssured.given;
import static org.junit.jupiter.api.Assertions.*;
/**
* The dispatch-<em>table</em> idiom — an array filled with literal program names, called through an
* indexed read of it — resolves to candidate call edges. 65 modules in {@code upms} dispatch this way.
*
* <p>{@code RESOLVE_DYNAMIC_CALLNAT_INTRA_INDIRECT} handles it: for the assignment writing the
* dispatch variable it follows the {@code READS} on that same line to the array, then takes every
* string literal written to the array as a target. Item 83's scalar fold does not reach these — the
* literals live one hop away, on the array node.
*
* <p><b>Written while investigating item 108, which claims this idiom is unsupported.</b> It is
* supported; what is missing there is only the {@code dispatch-table} <em>endpoint</em> reporting the
* rows. The reason `upms` shows nothing is item 107: those target modules are absent from the
* checkout entirely, and a literal naming no ingested module correctly yields no edge. This fixture
* exists because the behaviour had no test of its own, so nothing would have caught its loss.
*
* <p>Which index is live at runtime is not statically known, so resolution over-approximates to the
* set of literals ever assigned to the array — the same multi-target model item 82 allows a manual
* override. `TBRANCH` pins that this is not merely a simplification but the only correct answer: it
* builds the table in an `IF`/`ELSE`, so index 1 carries a different program per branch and any
* index-keyed pairing would be wrong.
*/
@QuarkusTest
class DispatchTableFoldIT {
private static final String PROJECT = "nat-dispatch-table-fold";
/**
* Straight table: four literals, called through an indexed read.
*/
private static final String ROUTER = """
* Router dispatching through a literal-filled table.
DEFINE DATA LOCAL
01 #WT-PROG (A8/1:4)
01 #W-ACT (A8)
01 #I-OBJ (I2)
END-DEFINE
*
PERFORM INIT-TABLE
*
#W-ACT := #WT-PROG (#I-OBJ)
CALLNAT #W-ACT
*
DEFINE SUBROUTINE INIT-TABLE
ASSIGN #WT-PROG (1) = 'TDISPA'
ASSIGN #WT-PROG (2) = 'TDISPB'
ASSIGN #WT-PROG (3) = 'TDISPC'
ASSIGN #WT-PROG (4) = 'TNOSUCH'
END-SUBROUTINE
*
END
""";
/**
* Table built in two branches: index 1 is TDISPA in one and TDISPB in the other. Also proves the
* resolver does not depend on the init preceding the call — here the subroutine is defined after
* the CALLNAT, as Natural routinely does.
*/
private static final String BRANCHED = """
* Router whose table depends on a runtime condition.
DEFINE DATA LOCAL
01 #WT-PROG (A8/1:2)
01 #W-ACT (A8)
01 #I-OBJ (I2)
01 #MODE (A4)
END-DEFINE
*
PERFORM INIT-TABLE
#W-ACT := #WT-PROG (#I-OBJ)
CALLNAT #W-ACT
*
DEFINE SUBROUTINE INIT-TABLE
IF #MODE = 'ADD'
ASSIGN #WT-PROG (1) = 'TDISPA'
ELSE
ASSIGN #WT-PROG (1) = 'TDISPB'
END-IF
ASSIGN #WT-PROG (2) = 'TDISPC'
END-SUBROUTINE
*
END
""";
/**
* A dispatch fed from a plain scalar, not an array — must stay untouched by this resolver.
*/
private static final String SCALARDSP = """
* Dispatch through a scalar the table resolver must not claim.
DEFINE DATA LOCAL
01 #W-ACT (A8)
END-DEFINE
*
#W-ACT := 'TDISPA'
CALLNAT #W-ACT
END
""";
private static final String TARGET = "DEFINE DATA LOCAL\nEND-DEFINE\nEND\n";
@TempDir
static Path root;
@BeforeAll
static void ingest() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
write("TROUTER.nat", ROUTER);
write("TBRANCH.nat", BRANCHED);
write("TSCALAR.nat", SCALARDSP);
write("TDISPA.nat", TARGET);
write("TDISPB.nat", TARGET);
write("TDISPC.nat", TARGET);
// TNOSUCH.nat deliberately absent: a literal that is not a real module must yield no edge.
given().contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
.when().post("/api/projects/" + PROJECT)
.then().statusCode(201);
given().when().post("/api/projects/" + PROJECT + "/refresh?deep=true")
.then().statusCode(200);
}
private static void write(String fileName, String content) {
try {
Files.writeString(root.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
private static JsonPath callees(String module) {
return given().pathParam("name", module)
.when().get("/api/projects/" + PROJECT + "/modules/{name}/callees?scope=external")
.then().statusCode(200).extract().jsonPath();
}
/**
* Every literal in the table becomes a candidate target.
*/
@Test
void everyTableLiteralBecomesACandidateTarget() {
List<String> names = callees("TROUTER").getList("items.name");
assertTrue(names.containsAll(List.of("TDISPA", "TDISPB", "TDISPC")),
"all three real table targets must be resolved: " + names);
}
/**
* A literal that names no ingested module produces no edge — the resolver does not invent nodes.
*/
@Test
void aLiteralThatIsNotARealModuleYieldsNoEdge() {
assertFalse(callees("TROUTER").getList("items.name").contains("TNOSUCH"),
"'TNOSUCH' is a literal in the table but no module exists — it must not become an edge");
}
/**
* The resolved edges are marked inferred, and distinguishable from item 83's string fold.
*/
@Test
void resolvedEdgesAreMarkedInferredNotStatic() {
JsonPath body = callees("TROUTER");
assertEquals("CALLNAT_DYNAMIC", body.getString("items.find { it.name == 'TDISPA' }.edgeKind"),
"a table dispatch is inferred, so it must not masquerade as a static CALLNAT");
}
/**
* The branched table: both branch values are candidates. This is the case that makes the
* candidate-set model necessary rather than merely convenient — index 1 has two different targets,
* so no index-keyed answer could be right.
*/
@Test
void aTableBuiltInTwoBranchesContributesBothValues() {
List<String> names = callees("TBRANCH").getList("items.name");
assertTrue(names.contains("TDISPA") && names.contains("TDISPB"),
"index 1 is TDISPA in one branch and TDISPB in the other; both are possible: " + names);
assertTrue(names.contains("TDISPC"), "the unbranched index must resolve too: " + names);
}
/**
* The scalar case is item 83's fold, and it must keep resolving on its own — the two paths are
* separate, so a change to the indirect resolver must not silently take the scalar one with it.
*/
@Test
void aScalarFedDispatchStillResolvesOnItsOwnPath() {
assertTrue(callees("TSCALAR").getList("items.name").contains("TDISPA"),
"a directly-assigned literal target must resolve without the array path");
}
}

View File

@@ -62,6 +62,33 @@ class DuplicateIdentityIT {
}
}
/**
* Item 136: the duplicate marker is a node like any other, and every node must carry
* {@code startLine}/{@code endLine}. {@code MARK_DUPLICATE_IDENTITIES} was the one node-creating
* query that set neither, and the identifier search's row mapper coerced them unconditionally — so
* a page deep enough to reach a marker faulted with an unstructured 500 instead of returning rows.
*/
@Test
void aDuplicateMarkerCarriesLinesAndDoesNotFaultTheIdentifierSearch() {
given().queryParam("name", "DUPE").queryParam("limit", 500)
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("findAll { it.name == 'DUPE' }.startLine", everyItem(notNullValue()))
.body("findAll { it.name == 'DUPE' }.endLine", everyItem(notNullValue()));
}
/**
* The same page, fetched whole. Pins the actual reported symptom — a 500 several hundred rows in —
* rather than only the property that caused it.
*/
@Test
void aFullIdentifierPageOverTheWholeProjectDoesNotFault() {
given().queryParam("name", "E").queryParam("contains", true).queryParam("limit", 500)
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then().statusCode(200);
}
/**
* The unreferenced duplicate: used to answer 404, i.e. "no such module".
*/

View File

@@ -0,0 +1,165 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 125: a Java type declaration must be findable through {@code search/identifier} by its
* <b>short</b> name, not only by the fully-qualified identity the graph stores (item 117), and
* {@code ?contains=} must do what it does on {@code /search/value} instead of being dropped.
*
* <p>The bug this pins down was not "types are not indexed" — they were, under their FQN — but that
* the short form an agent actually types answered {@code []}, which is indistinguishable from "no
* such name". The match now also carries {@code simpleName} and {@code moduleKind}, so "is this a
* type or a method?" is answered by the search itself.
*/
@QuarkusTest
class IdentifierTypeDeclarationIT {
private static final String PROJECT = "identifier-type-decl-project";
private static final String FQN = "com.example.idt.PartnerUpdateLogic";
@TempDir
static Path root;
@BeforeAll
static void ingestFixtures() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
Path pkg = root.resolve("src/main/java/com/example/idt");
writeSource(pkg, "PartnerUpdateLogic.java", """
package com.example.idt;
public class PartnerUpdateLogic {
private String partnerName;
public void updatePartner() {
this.partnerName = "x";
}
}
""");
writeSource(pkg, "PartnerReadPort.java", """
package com.example.idt;
public interface PartnerReadPort {
String read();
}
""");
given()
.contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "java", null, null))
.when().post("/api/projects/" + PROJECT)
.then()
.statusCode(201);
given()
.when().post("/api/projects/" + PROJECT + "/refresh?deep=true")
.then()
.statusCode(200);
}
private static void writeSource(Path dir, String fileName, String content) {
try {
Files.createDirectories(dir);
Files.writeString(dir.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
private static io.restassured.specification.RequestSpecification search() {
return given().queryParam("type", "MODULE");
}
@Test
void theShortNameFindsTheTypeDeclaration() {
search()
.queryParam("name", "PartnerUpdateLogic")
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("name", hasItem(FQN))
.body("find { it.name == '" + FQN + "' }.simpleName", equalTo("PartnerUpdateLogic"))
.body("find { it.name == '" + FQN + "' }.moduleKind", equalTo("CLASS"));
}
@Test
void theQualifiedNameStillFindsIt() {
search()
.queryParam("name", FQN)
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("name", hasItem(FQN));
}
@Test
void moduleKindDistinguishesAnInterfaceFromAClass() {
search()
.queryParam("name", "PartnerReadPort")
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("moduleKind", everyItem(equalTo("INTERFACE")));
}
@Test
void containsMatchesASubstringOfTheShortNameCaseInsensitively() {
search()
.queryParam("name", "partnerupd").queryParam("contains", true)
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("name", hasItem(FQN));
}
@Test
void containsAlsoMatchesThePackageFragmentOfTheQualifiedName() {
search()
.queryParam("name", "com.example.idt").queryParam("contains", true)
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("name", hasItems(FQN, "com.example.idt.PartnerReadPort"));
}
@Test
void withoutContainsASubstringIsNotAMatch() {
search()
.queryParam("name", "PartnerUpd")
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("$", hasSize(0));
}
@Test
void aMethodMatchCarriesNoSimpleNameOrModuleKind() {
given()
.queryParam("name", "updatePartner")
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("find { it.type == 'FUNCTION' }.simpleName", nullValue())
.body("find { it.type == 'FUNCTION' }.moduleKind", nullValue());
}
@Test
void containsWithoutANameIsRejectedRatherThanDumpingEveryNode() {
given()
.queryParam("contains", true)
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(400)
.body("code", equalTo("MISSING_NAME"));
}
}

View File

@@ -0,0 +1,111 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.InputStream;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 140: a Java project with no JPA entity has no {@code DB_TABLE}, so every {@code DB_ACCESS}
* candidate the parser's over-approximating heuristic emitted is a false positive by construction
* and is reaped at the end of enrichment.
*
* <p>Both projects ingest the <em>same</em> two source files; the second adds one entity. That is
* the whole difference, and it is what makes this a test of the gate rather than of the parser: the
* reaper is project-level, so the identical false positives survive in the project that happens to
* own a table. {@code sql-statements} is the endpoint under test because it joins the table with
* {@code OPTIONAL MATCH} and is therefore the one that actually leaked them.
*/
@QuarkusTest
class JavaDbAccessNoEntityIT {
private static final String WITHOUT_ENTITY = "item140-no-entity";
private static final String WITH_ENTITY = "item140-with-entity";
@TempDir
static Path withoutEntityRoot;
@TempDir
static Path withEntityRoot;
@BeforeAll
static void ingest() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
copyFixture(withoutEntityRoot, "fixtures/java/nodb/TextUtils.java");
copyFixture(withoutEntityRoot, "fixtures/java/nodb/ReportRenderer.java");
copyFixture(withEntityRoot, "fixtures/java/nodb/TextUtils.java");
copyFixture(withEntityRoot, "fixtures/java/nodb/ReportRenderer.java");
copyFixture(withEntityRoot, "fixtures/java/nodb/Ledger.java");
ingestProject(WITHOUT_ENTITY, withoutEntityRoot);
ingestProject(WITH_ENTITY, withEntityRoot);
}
private static void ingestProject(String project, Path root) {
given().contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "java", null, null))
.when().post("/api/projects/" + project)
.then().statusCode(201);
given().when().post("/api/projects/" + project + "/refresh?deep=true")
.then().statusCode(200);
}
private static void copyFixture(Path root, String classpathResource) {
String fileName = classpathResource.substring(classpathResource.lastIndexOf('/') + 1);
try (InputStream in = JavaDbAccessNoEntityIT.class.getClassLoader().getResourceAsStream(classpathResource)) {
if (in == null) {
throw new IllegalStateException("Resource not found: " + classpathResource);
}
Files.write(root.resolve(fileName), in.readAllBytes());
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
/**
* Without an entity, {@code TextUtils.getColumn(...)} must not be reported as a database read.
*/
@Test
void noEntityMeansNoDbAccessAtAll() {
given().pathParam("name", "ReportRenderer")
.when().get("/api/projects/" + WITHOUT_ENTITY + "/modules/{name}/sql-statements")
.then().statusCode(200)
.body("$", empty());
}
/**
* db-accesses was already empty here (it joins the table with a plain MATCH), and stays empty —
* the reaper must not have made it worse by leaving a dangling row behind.
*/
@Test
void noEntityMeansNoDbAccessesEither() {
given().pathParam("name", "ReportRenderer")
.when().get("/api/projects/" + WITHOUT_ENTITY + "/modules/{name}/db-accesses")
.then().statusCode(200)
.body("$", empty());
}
/**
* The gate is project-level, so one entity anywhere in the project keeps every candidate alive —
* including the same static-getter false positives, which still arrive with a null table. This
* pins the deliberate limit of item 140's chosen rule rather than glossing over it.
*/
@Test
void oneEntityInTheProjectKeepsTheSameFalsePositives() {
given().pathParam("name", "ReportRenderer")
.when().get("/api/projects/" + WITH_ENTITY + "/modules/{name}/sql-statements")
.then().statusCode(200)
.body("$", not(empty()))
.body("statement", hasItem("TextUtils.getColumn(line, 0, 8)"))
.body("table", hasItem(nullValue()));
}
}

View File

@@ -18,6 +18,7 @@ import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.empty;
import static org.hamcrest.Matchers.not;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertNull;
/**
* Item 72: a dispatch row must report its <em>whole</em> guard chain, not just the innermost one.
@@ -45,6 +46,7 @@ class NestedDispatchGuardIT {
static void ingest() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
copyFixture("fixtures/natural/dispatch/NESTDISP.nat");
copyFixture("fixtures/natural/dispatch/DISPCPY.cpy");
given().contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
@@ -118,6 +120,56 @@ class NestedDispatchGuardIT {
assertEquals(List.of(List.of("TABL")), body.getList(row + ".guards.values"));
}
/**
* Item 122: a row spliced in from a copycode must name the copycode's file, not the host's.
*
* <p>Before this, {@code dispatch-table} was the only site-bearing endpoint carrying a bare
* {@code lineNo} — {@code callees}, {@code db-accesses}, {@code workfile-accesses} and
* {@code functions} all already carried the provenance quartet. So an agent resolved the number
* against the host file and landed somewhere arbitrary: here, line 8 of {@code NESTDISP.nat} is a
* comment in the module header. In {@code upms} this hit 26 of {@code VCOMIN50}'s 44 rows.
*/
@Test
void aCopycodeDerivedRowNamesTheCopycodeFile() {
JsonPath body = dispatchTable();
String row = "find { it.assignedValue == 'DESC-FROM-CPY' }";
assertEquals("DISPCPY.cpy", body.getString(row + ".sourceFile"),
"the MOVE is written in the copycode. Full response was: " + body.getList("$"));
assertEquals("DISPCPY", body.getString(row + ".viaCopycode"));
assertEquals(8, body.getInt(row + ".lineNo"), "the MOVE is on line 8 of DISPCPY.cpy");
assertEquals(35, body.getInt(row + ".includedAt"),
"and points back at the INCLUDE on line 35 of NESTDISP.nat");
}
/**
* The guard chain spans the file boundary: the outer {@code DECIDE} is in the host, the inner one
* in the copycode. Expansion happens before parsing, so this should hold — but it is the property
* that makes the row usable, and it is worth pinning rather than assuming.
*/
@Test
void theGuardChainSpansTheIncludeBoundary() {
JsonPath body = dispatchTable();
String row = "find { it.assignedValue == 'DESC-FROM-CPY' }";
assertEquals(List.of("#SHORT-VIEW", "#FIELD-NAME"), body.getList(row + ".guards.field"),
"the host's DECIDE guards the copycode's DECIDE");
assertEquals(List.of(List.of("CPYV"), List.of("TX-FROM-CPY")), body.getList(row + ".guards.values"));
}
/**
* A host-local row carries no copycode marker and reports the host file.
*/
@Test
void aHostLocalRowReportsTheHostFileWithNoCopycodeMarker() {
JsonPath body = dispatchTable();
String row = "find { it.assignedValue == 'CODE-TABL' }";
assertEquals("NESTDISP.nat", body.getString(row + ".sourceFile"));
assertNull(body.getString(row + ".viaCopycode"), "nothing included it, so there is no copycode to name");
assertEquals(23, body.getInt(row + ".lineNo"), "the MOVE is on line 23 of NESTDISP.nat");
}
/**
* The legacy fields keep their exact meaning (the innermost guard), so item 72 is additive: an
* existing consumer reading guardField/guardValues sees what it always saw.

View File

@@ -0,0 +1,169 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.*;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 126: {@code /projects} must carry what the last <b>whole-root</b> ingest did, so a negative
* answer is evidence rather than a guess. Before this, nothing in the API said whether a project had
* ever been fully ingested, so every "not found" had to be cross-checked against the file system.
*
* <p>The sharpest assertion here is {@link #aByNameRefreshDoesNotMoveIngestedAt()}: a partial ingest
* that moved the timestamp would report the project as freshly walked when one module was deepened —
* which is the very "looks complete but isn't" answer the item exists to remove.
*/
@QuarkusTest
@TestMethodOrder(MethodOrderer.OrderAnnotation.class)
class ProjectIngestMetadataIT {
private static final String PROJECT = "ingest-metadata-project";
@TempDir
static Path root;
@BeforeAll
static void createProject() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
writeSource("GOOD.nat", """
DEFINE DATA
LOCAL
1 #A (A8)
END-DEFINE
*
CALLNAT 'OTHER'
END
""");
writeSource("OTHER.nat", """
DEFINE DATA
LOCAL
1 #B (A8)
END-DEFINE
*
END
""");
given()
.contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
.when().post("/api/projects/" + PROJECT)
.then()
.statusCode(201);
}
private static void writeSource(String fileName, String content) {
try {
Files.writeString(root.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
private static io.restassured.response.Response project() {
return given().when().get("/api/projects/" + PROJECT);
}
@Test
@Order(1)
void theCreateTimeScanIsRecordedAsTier1() {
project().then()
.statusCode(200)
.body("ingest.mode", equalTo("tier1"))
.body("ingest.ingestedAt", notNullValue())
.body("ingest.filesExamined", equalTo(2))
.body("ingest.filesPersisted", equalTo(2))
.body("ingest.filesFailed", equalTo(0))
.body("ingest.failuresTruncated", equalTo(false))
.body("ingest.serverVersion", not(emptyString()));
}
@Test
@Order(2)
void aDeepRefreshRecordsTheFullModeAndTheNewFileCount() {
writeSource("THIRD.nat", """
DEFINE DATA
LOCAL
1 #C (A8)
END-DEFINE
*
END
""");
given().when().post("/api/projects/" + PROJECT + "/refresh?deep=true").then().statusCode(200);
project().then()
.statusCode(200)
.body("ingest.mode", equalTo("full"))
.body("ingest.filesExamined", equalTo(3))
.body("ingest.filesPersisted", equalTo(3));
}
/**
* The failure <em>list</em> and the failure <em>count</em> must agree whenever the list was not
* truncated — that invariant is what stops a short list from being read as "these were all of
* them". Deliberately not asserting that the malformed file below fails to parse: the Natural
* parser is tolerant by design, so a fixture that "looks broken" is not a reliable way to produce
* a failure, and a test that pretends otherwise would be testing the parser's mood.
*/
@Test
@Order(3)
void theFailureListAgreesWithTheFailureCount() {
writeSource("BROKEN.nat", "DEFINE DATA LOCAL\n1 #X (A8\n");
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
io.restassured.response.Response response = project();
response.then()
.statusCode(200)
.body("ingest.mode", equalTo("call_graph"))
.body("ingest.filesExamined", equalTo(4));
int failed = response.jsonPath().getInt("ingest.filesFailed");
int listed = response.jsonPath().getList("ingest.failures").size();
boolean truncated = response.jsonPath().getBoolean("ingest.failuresTruncated");
org.junit.jupiter.api.Assertions.assertEquals(truncated, listed < failed,
"failuresTruncated must say exactly whether the list is shorter than the count");
org.junit.jupiter.api.Assertions.assertTrue(listed <= failed,
"the failure list can never be longer than the failure count");
}
@Test
@Order(4)
void aConfigUpdateDoesNotClobberTheIngestMetadata() {
String before = project().jsonPath().getString("ingest.ingestedAt");
given()
.contentType("application/json")
.body(new ProjectResource.ProjectRequest("now described", null, null, null, null, null))
.when().put("/api/projects/" + PROJECT)
.then().statusCode(200);
project().then()
.statusCode(200)
.body("description", equalTo("now described"))
.body("ingest.ingestedAt", equalTo(before));
}
@Test
@Order(5)
void aByNameRefreshDoesNotMoveIngestedAt() {
String before = project().jsonPath().getString("ingest.ingestedAt");
int examinedBefore = project().jsonPath().getInt("ingest.filesExamined");
given().when().post("/api/projects/" + PROJECT + "/refresh/GOOD").then().statusCode(200);
project().then()
.statusCode(200)
.body("ingest.ingestedAt", equalTo(before))
.body("ingest.filesExamined", equalTo(examinedBefore));
}
@Test
@Order(6)
void theProjectListingCarriesTheSameMetadata() {
given().when().get("/api/projects")
.then()
.statusCode(200)
.body("find { it.name == '" + PROJECT + "' }.ingest.mode", notNullValue())
.body("find { it.name == '" + PROJECT + "' }.ingest.filesExamined", greaterThan(0));
}
}

View File

@@ -0,0 +1,271 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 130: the REST surface, composed from the class-level and method-level {@code @Path} that the
* parser now persists — the daily "which code runs for this URL" question, previously answerable only
* by joining two {@code /search/annotation} calls by hand.
*
* <p>Also covers the scope/staleness response headers, which exist so an <em>empty</em> answer can be
* read correctly: "no callers" means something different in a project that excludes {@code test}.
*/
@QuarkusTest
class RestEndpointsIT {
private static final String PROJECT = "rest-endpoints-project";
@TempDir
static Path root;
@BeforeAll
static void ingestFixtures() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
Path pkg = root.resolve("src/main/java/com/example/rest");
write(pkg, "PartnerResource.java", """
package com.example.rest;
import jakarta.ws.rs.GET;
import jakarta.ws.rs.POST;
import jakarta.ws.rs.Path;
@Path("/partners")
public class PartnerResource {
@GET
@Path("/{id}")
public String byId(String id) {
return id;
}
@POST
public String create(String body) {
return body;
}
public String notAnEndpoint() {
return "helper";
}
}
""");
write(pkg, "PathConstants.java", """
package com.example.rest;
public final class PathConstants {
public static final String ADMIN = "/admin";
private PathConstants() {
}
}
""");
write(pkg, "AdminResource.java", """
package com.example.rest;
import jakarta.ws.rs.DELETE;
import jakarta.ws.rs.Path;
@Path(AdminResource.BASE)
public class AdminResource {
static final String BASE = "/admin";
@DELETE
public String wipe() {
return "gone";
}
}
""");
write(pkg, "AbstractFileSvc.java", """
package com.example.rest;
import jakarta.ws.rs.POST;
import jakarta.ws.rs.Path;
@Path("/files/")
public abstract class AbstractFileSvc {
@POST
@Path("/upload")
public String upload(String body) {
return body;
}
}
""");
write(pkg, "ReportFileSvc.java", """
package com.example.rest;
public class ReportFileSvc extends AbstractFileSvc {
}
""");
given()
.contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "java", null, null))
.when().post("/api/projects/" + PROJECT)
.then()
.statusCode(201);
given().when().post("/api/projects/" + PROJECT + "/refresh?deep=true").then().statusCode(200);
}
private static void write(Path dir, String fileName, String content) {
try {
Files.createDirectories(dir);
Files.writeString(dir.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
private static io.restassured.response.Response endpoints() {
return given().when().get("/api/projects/" + PROJECT + "/rest-endpoints");
}
@Test
void theClassAndMethodPathsAreComposedIntoOnePath() {
endpoints().then()
.statusCode(200)
.body("find { it.handler == 'byId' }.path", equalTo("/partners/{id}"))
.body("find { it.handler == 'byId' }.httpMethod", equalTo("GET"))
.body("find { it.handler == 'byId' }.module", equalTo("com.example.rest.PartnerResource"));
}
@Test
void aMethodWithoutItsOwnPathInheritsTheClassPath() {
endpoints().then()
.statusCode(200)
.body("find { it.handler == 'create' }.path", equalTo("/partners"))
.body("find { it.handler == 'create' }.httpMethod", equalTo("POST"));
}
@Test
void aPathWrittenAsAConstantIsResolved() {
endpoints().then()
.statusCode(200)
.body("find { it.handler == 'wipe' }.path", equalTo("/admin"));
}
@Test
void aMethodWithNoHttpVerbIsNotAnEndpoint() {
endpoints().then()
.statusCode(200)
.body("handler", not(hasItem("notAnEndpoint")));
}
/**
* Both halves of the path carry their own slashes ({@code "/files/"} + {@code "/upload"}). The
* first implementation joined them with a single non-overlapping {@code replace("//", "/")},
* which turned {@code "///upload"} into {@code "//upload"} — visible only on real data, where
* {@code pur} produced paths like {@code //file}.
*/
@Test
void pathHalvesWithTheirOwnSlashesJoinWithExactlyOne() {
endpoints().then()
.statusCode(200)
.body("path", everyItem(not(containsString("//"))))
.body("find { it.handler == 'upload' }.path", equalTo("/files/upload"));
}
/**
* JAX-RS inherits {@code @Path} from a base class, and reporting the bare {@code /} for those is a
* wrong answer rather than a missing one.
*/
@Test
void aSubclassInheritsItsBaseClassPath() {
endpoints().then()
.statusCode(200)
.body("findAll { it.module.endsWith('ReportFileSvc') }.path", everyItem(equalTo("/files/upload")));
}
/**
* The graph can hold more than one {@code CONTAINS} edge between the same module and function
* (roadmap item 75), which multiplied endpoints into identical rows — 183 of 436 on {@code pur}.
*/
@Test
void everyRowIsUnique() {
java.util.List<java.util.Map<String, Object>> rows = endpoints().jsonPath().getList("$");
org.junit.jupiter.api.Assertions.assertEquals(rows.size(), new java.util.HashSet<>(rows).size(),
"rest-endpoints must not repeat a row: " + rows);
}
@Test
void everyRowPointsAtItsSource() {
endpoints().then()
.statusCode(200)
.body("sourceFile", everyItem(endsWith(".java")))
.body("startLine", everyItem(greaterThan(0)));
}
@Test
void theModuleFilterNarrowsToOneClass() {
given().queryParam("module", "AdminResource")
.when().get("/api/projects/" + PROJECT + "/rest-endpoints")
.then()
.statusCode(200)
.body("module", everyItem(equalTo("com.example.rest.AdminResource")));
}
/**
* An outbound {@code @RegisterRestClient} interface declares a call the application *makes*. It is
* flagged rather than presented as a served endpoint — `pur` had 3 of them reading as endpoints.
*/
@Test
void everyEndpointHereIsInboundNotAnOutboundRestClient() {
endpoints().then()
.statusCode(200)
.body("outbound", everyItem(equalTo(false)));
}
/**
* Item 129's copycode stand-down must not fire for a Java project. `ac` carries `.cpy` files as
* Natural *test fixtures* that its Java walk never ingests; they had no stored hash, counted as
* changed, and disabled skipping entirely — 487 of 509 unchanged files were re-parsed.
*/
@Test
void changedOnlyOnAJavaProjectIsNotDisabledByAStrayCopycodeFile() {
try {
Files.writeString(root.resolve("stray.cpy"), "* a Natural fixture in a Java project\n");
} catch (IOException e) {
throw new UncheckedIOException(e);
}
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
given()
.when().post("/api/projects/" + PROJECT + "/refresh?changedOnly=true")
.then()
.statusCode(200)
.body("examinedFiles", empty());
}
@Test
void everyProjectScopedResponseCarriesTheScopeAndFreshnessHeaders() {
given().when().get("/api/projects/" + PROJECT + "/modules")
.then()
.statusCode(200)
.header(ProjectScopeHeaderFilter.EXCLUDE_DIRS, notNullValue())
.header(ProjectScopeHeaderFilter.INGESTED_AT, notNullValue())
.header(ProjectScopeHeaderFilter.INCOMPLETE, equalTo("false"));
}
/**
* The headers must ride on a <em>bare array</em> response too — that is the whole reason they are
* headers and not body fields.
*/
@Test
void theHeadersAlsoRideOnBareArrayResponses() {
given().queryParam("name", "PartnerResource")
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.header(ProjectScopeHeaderFilter.EXCLUDE_DIRS, notNullValue())
.header(ProjectScopeHeaderFilter.INCOMPLETE, equalTo("false"));
}
}

View File

@@ -0,0 +1,172 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 128: every reference site of a type, not just its callers.
*
* <p>The fixture is built so each reference kind occurs in exactly one place: {@code Consumer}
* imports {@code Target}, declares a field of it, takes it as a parameter, returns it, is annotated
* with {@code Marker}, and calls a method on it. {@code Sub} extends it. A rename of {@code Target}
* must find all of those — {@code callers} finds only the call.
*/
@QuarkusTest
class SearchReferencesIT {
private static final String PROJECT = "search-references-project";
@TempDir
static Path root;
@BeforeAll
static void ingestFixtures() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
Path pkg = root.resolve("src/main/java/com/example/refs");
write(pkg, "Target.java", """
package com.example.refs;
public class Target {
public String describe() {
return "target";
}
}
""");
write(pkg, "Marker.java", """
package com.example.refs;
public @interface Marker {
}
""");
write(root.resolve("src/main/java/com/example/other"), "Consumer.java", """
package com.example.other;
import com.example.refs.Marker;
import com.example.refs.Target;
@Marker
public class Consumer {
private Target field;
public Target handle(Target incoming) {
return incoming;
}
public String use() {
return field.describe();
}
}
""");
write(root.resolve("src/main/java/com/example/other"), "Sub.java", """
package com.example.other;
import com.example.refs.Target;
public class Sub extends Target {
}
""");
given()
.contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "java", null, null))
.when().post("/api/projects/" + PROJECT)
.then()
.statusCode(201);
given().when().post("/api/projects/" + PROJECT + "/refresh?deep=true").then().statusCode(200);
}
private static void write(Path dir, String fileName, String content) {
try {
Files.createDirectories(dir);
Files.writeString(dir.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
private static io.restassured.response.Response references(String name, @org.jspecify.annotations.Nullable String kindOrNull) {
var request = given().queryParam("name", name).queryParam("limit", 200);
if (kindOrNull != null) {
request = request.queryParam("kind", kindOrNull);
}
return request.when().get("/api/projects/" + PROJECT + "/search/references");
}
@Test
void theImportOfATypeIsAReferenceSite() {
references("Target", "IMPORT").then()
.statusCode(200)
.body("sourceFile", hasItems(containsString("Consumer.java"), containsString("Sub.java")));
}
@Test
void aDeclaredFieldParameterAndReturnTypeAreReferenceSites() {
references("Target", "TYPE").then()
.statusCode(200)
.body("sourceFile", everyItem(containsString("Consumer.java")))
.body("size()", greaterThanOrEqualTo(2));
}
@Test
void anAnnotationUsageIsAReferenceSite() {
references("Marker", "ANNOTATION").then()
.statusCode(200)
.body("sourceFile", hasItem(containsString("Consumer.java")));
}
@Test
void inheritanceIsAReferenceSite() {
references("Target", "EXTENDS").then()
.statusCode(200)
.body("sourceFile", hasItem(containsString("Sub.java")));
}
@Test
void everyKindComesBackTogetherWhenNoKindIsGiven() {
references("Target", null).then()
.statusCode(200)
.body("kind", hasItems("IMPORT", "TYPE", "EXTENDS"))
.body("target", everyItem(equalTo("com.example.refs.Target")));
}
@Test
void theFullyQualifiedNameFindsTheSameSites() {
int viaShortName = references("Target", null).jsonPath().getList("$").size();
references("com.example.refs.Target", null).then()
.statusCode(200)
.body("size()", equalTo(viaShortName));
}
@Test
void eachSiteCarriesAUsableFileAndLine() {
references("Target", "IMPORT").then()
.statusCode(200)
.body("lineNo", everyItem(greaterThan(0)))
.body("inModule", everyItem(notNullValue()));
}
@Test
void anUnknownKindIsRejectedRatherThanAnsweredEmpty() {
references("Target", "NONSENSE").then()
.statusCode(400)
.body("code", equalTo("INVALID_KIND"));
}
@Test
void aMissingNameIsRejected() {
given().when().get("/api/projects/" + PROJECT + "/search/references")
.then()
.statusCode(400)
.body("code", equalTo("MISSING_NAME"));
}
}

View File

@@ -0,0 +1,310 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.Map;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 131: a capped search answer must say that it was capped.
*
* <p>The fixture declares 60 annotated classes against a default {@code limit} of 50, so the default
* page is short of the truth by construction — the shape of the real failure, where
* {@code search/annotation?name=Immutable} returned 50 of 95 rows and a real audit read the page as
* the whole set, reporting 17 entities as having lost the annotation when none had.
*/
@QuarkusTest
class SearchTruncationIT {
private static final String PROJECT = "search-truncation-project";
private static final int CLASSES = 60;
/**
* Item 135: more endpoints than the default page of 50, so paging them is a real question.
*/
private static final int ENDPOINTS = 60;
@TempDir
static Path root;
@BeforeAll
static void ingestFixtures() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
Path pkg = root.resolve("src/main/java/com/example/many");
write(pkg, "Marked.java", """
package com.example.many;
public @interface Marked {
}
""");
// Item 135: one project type every Thing imports and declares a field of, so the reference
// search has a set larger than the default page to page over. It lives in its own package on
// purpose: a same-package type needs no import, and TypeResolver cannot turn the bare simple
// name into an identity, so no reference edge is emitted for it at all (documented on
// JavaParser#addReferenceEdges).
write(root.resolve("src/main/java/com/example/support"), "Support.java", """
package com.example.support;
public class Support {
}
""");
for (int i = 0; i < CLASSES; i++) {
write(pkg, "Thing" + i + ".java", """
package com.example.many;
import com.example.support.Support;
@Marked
public class Thing%d {
private String shared = "repeated-literal";
private Support support;
public String shared() {
return shared;
}
}
""".formatted(i));
}
// Item 135: a REST surface to page over. Separate classes from the Thing fixture above so the
// annotation and identifier counts it pins stay untouched.
for (int i = 0; i < ENDPOINTS; i++) {
write(pkg, "Endpoint" + i + "Resource.java", """
package com.example.many;
import jakarta.ws.rs.GET;
import jakarta.ws.rs.Path;
@Path("/endpoint%d")
public class Endpoint%dResource {
@GET
@Path("/list")
public String list() {
return "";
}
}
""".formatted(i, i));
}
given()
.contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "java", null, null))
.when().post("/api/projects/" + PROJECT)
.then()
.statusCode(201);
given().when().post("/api/projects/" + PROJECT + "/refresh?deep=true").then().statusCode(200);
}
private static void write(Path dir, String fileName, String content) {
try {
Files.createDirectories(dir);
Files.writeString(dir.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
@Test
void aCappedAnnotationSearchReportsTheTotalAndSaysItWasCut() {
given().queryParam("name", "Marked")
.when().get("/api/projects/" + PROJECT + "/search/annotation")
.then()
.statusCode(200)
.body("size()", equalTo(50))
.header(AnalysisResource.TRUNCATED, equalTo("true"))
.header(AnalysisResource.TOTAL_COUNT, equalTo(String.valueOf(CLASSES)));
}
/**
* The count must not inherit the page's cap. Capping it would make the total equal the row count
* every time, which reads as "complete" and silently defeats the entire item.
*/
@Test
void theTotalExceedsTheRowsItDescribes() {
int rows = given().queryParam("name", "Marked")
.when().get("/api/projects/" + PROJECT + "/search/annotation")
.jsonPath().getList("$").size();
long total = Long.parseLong(given().queryParam("name", "Marked")
.when().get("/api/projects/" + PROJECT + "/search/annotation")
.header(AnalysisResource.TOTAL_COUNT));
org.junit.jupiter.api.Assertions.assertTrue(total > rows,
"total (" + total + ") must exceed the returned rows (" + rows + ")");
}
@Test
void anUncappedRequestIsNotReportedAsTruncated() {
given().queryParam("name", "Marked").queryParam("limit", 500)
.when().get("/api/projects/" + PROJECT + "/search/annotation")
.then()
.statusCode(200)
.body("size()", equalTo(CLASSES))
.header(AnalysisResource.TRUNCATED, equalTo("false"))
.header(AnalysisResource.TOTAL_COUNT, equalTo(String.valueOf(CLASSES)));
}
@Test
void countOnlyAnswersTheCountingQuestionWithoutTheRows() {
given().queryParam("name", "Marked").queryParam("countOnly", true)
.when().get("/api/projects/" + PROJECT + "/search/annotation")
.then()
.statusCode(200)
.body("count", equalTo(CLASSES))
.body("$", not(hasKey("rows")));
}
@Test
void theIdentifierSearchCarriesTheSameHeaders() {
given().queryParam("name", "shared").queryParam("type", "FUNCTION")
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.header(AnalysisResource.TOTAL_COUNT, equalTo(String.valueOf(CLASSES)))
.header(AnalysisResource.TRUNCATED, equalTo("true"));
}
@Test
void theIdentifierSearchCountOnlyWorks() {
given().queryParam("name", "shared").queryParam("type", "FUNCTION").queryParam("countOnly", true)
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then()
.statusCode(200)
.body("count", equalTo(CLASSES));
}
@Test
void theValueSearchCarriesTheSameHeaders() {
given().queryParam("value", "repeated-literal")
.when().get("/api/projects/" + PROJECT + "/search/value")
.then()
.statusCode(200)
.header(AnalysisResource.TOTAL_COUNT, notNullValue())
.header(AnalysisResource.TRUNCATED, notNullValue());
}
// --- item 135: the two endpoints item 131 forgot -------------------------------------------
/**
* The failure this item is about: {@code search/references} capped at the default 50 and said
* nothing at all — no total, no truncation flag. A rename scoped from that page would have missed
* every site past the fiftieth and looked complete doing it.
*/
@Test
void theReferenceSearchReportsItsTotalAndTruncation() {
int all = given().queryParam("name", "Support").queryParam("limit", 500)
.when().get("/api/projects/" + PROJECT + "/search/references")
.jsonPath().getList("$").size();
org.junit.jupiter.api.Assertions.assertTrue(all > 50,
"the fixture must produce more references than one default page, got " + all);
given().queryParam("name", "Support")
.when().get("/api/projects/" + PROJECT + "/search/references")
.then()
.statusCode(200)
.body("size()", equalTo(50))
.header(AnalysisResource.TRUNCATED, equalTo("true"))
.header(AnalysisResource.TOTAL_COUNT, equalTo(String.valueOf(all)));
}
@Test
void theReferenceSearchCountOnlyAgreesWithTheFullFetch() {
int rows = given().queryParam("name", "Support").queryParam("limit", 500)
.when().get("/api/projects/" + PROJECT + "/search/references")
.jsonPath().getList("$").size();
given().queryParam("name", "Support").queryParam("countOnly", true)
.when().get("/api/projects/" + PROJECT + "/search/references")
.then()
.statusCode(200)
.body("count", equalTo(rows));
}
/**
* The count must not inherit {@code $scanCap} — the same trap {@link #theTotalExceedsTheRowsItDescribes}
* pins for the annotation search.
*/
@Test
void theReferenceTotalExceedsTheRowsItDescribes() {
int rows = given().queryParam("name", "Support").queryParam("limit", 5)
.when().get("/api/projects/" + PROJECT + "/search/references")
.jsonPath().getList("$").size();
long total = Long.parseLong(given().queryParam("name", "Support").queryParam("limit", 5)
.when().get("/api/projects/" + PROJECT + "/search/references")
.header(AnalysisResource.TOTAL_COUNT));
org.junit.jupiter.api.Assertions.assertEquals(5, rows);
org.junit.jupiter.api.Assertions.assertTrue(total > rows,
"total (" + total + ") must exceed the returned rows (" + rows + ")");
}
/**
* {@code rest-endpoints} defaults to an uncapped limit, so it never lost rows — but it was equally
* silent about how many there are. The header has to be there either way.
*/
@Test
void theRestEndpointListReportsItsTotalWhenComplete() {
given().when().get("/api/projects/" + PROJECT + "/rest-endpoints")
.then()
.statusCode(200)
.body("size()", equalTo(ENDPOINTS))
.header(AnalysisResource.TRUNCATED, equalTo("false"))
.header(AnalysisResource.TOTAL_COUNT, equalTo(String.valueOf(ENDPOINTS)));
}
@Test
void aCappedRestEndpointPageSaysItWasCut() {
given().queryParam("limit", 10)
.when().get("/api/projects/" + PROJECT + "/rest-endpoints")
.then()
.statusCode(200)
.body("size()", equalTo(10))
.header(AnalysisResource.TRUNCATED, equalTo("true"))
.header(AnalysisResource.TOTAL_COUNT, equalTo(String.valueOf(ENDPOINTS)));
}
@Test
void theRestEndpointCountOnlyWorks() {
given().queryParam("countOnly", true)
.when().get("/api/projects/" + PROJECT + "/rest-endpoints")
.then()
.statusCode(200)
.body("count", equalTo(ENDPOINTS))
.body("$", not(hasKey("rows")));
}
/**
* Paging has to actually move: two consecutive pages of one must not be the same row. A count
* header on a page that never advances would be worse than no header, because it would look right.
*/
@Test
void consecutiveReferencePagesAreDisjoint() {
// Compared as whole rows, not by file: one class contributes both an IMPORT and a TYPE
// reference, so two adjacent rows legitimately share a sourceFile.
Map<String, ?> first = given().queryParam("name", "Support").queryParam("limit", 1).queryParam("offset", 0)
.when().get("/api/projects/" + PROJECT + "/search/references")
.jsonPath().getMap("[0]");
Map<String, ?> second = given().queryParam("name", "Support").queryParam("limit", 1).queryParam("offset", 1)
.when().get("/api/projects/" + PROJECT + "/search/references")
.jsonPath().getMap("[0]");
org.junit.jupiter.api.Assertions.assertNotEquals(first, second);
}
/**
* A page shorter than the limit is provably the end of the set, so no count query runs and the
* total is arithmetic — this pins that the cheap path reports the same numbers as the counted one.
*/
@Test
void aShortPageIsCompleteAndItsTotalIsExact() {
given().queryParam("name", "Marked").queryParam("limit", 50).queryParam("offset", 40)
.when().get("/api/projects/" + PROJECT + "/search/annotation")
.then()
.statusCode(200)
.body("size()", equalTo(20))
.header(AnalysisResource.TRUNCATED, equalTo("false"))
.header(AnalysisResource.TOTAL_COUNT, equalTo(String.valueOf(CLASSES)));
}
}

View File

@@ -0,0 +1,219 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.List;
import java.util.Map;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 124: a call edge must not outlive the call it was parsed from.
*
* <p>Item 58's sweep deletes stale <b>nodes</b>; an edge is only reaped when one of its endpoints goes
* with it. So an edge survives whenever both endpoints legitimately survive — which is exactly what a
* <b>parser fix</b> produces: the calling subroutine is untouched, and the old target is a placeholder
* ({@code sourceFile=""}) that is never file-swept. The corrected call is then merely <em>added</em>
* beside the wrong one, and both are served.
*
* <p>Found for real: after items 120/121/123 shipped, {@code DAGCHEN0/callees} in {@code upms} listed
* the correct {@code YAGCHBN0} <em>and</em> the pre-fix phantom {@code AGNT-CHG-CMP-SP} from the same
* call site, with 497 such edges surviving a full deep refresh.
*
* <p>The fixture edits only the CALLNAT target and keeps the enclosing subroutine, because that is the
* distinguishing case. {@code DerivedCallsModuleRefreshIT} removes the whole subroutine, so there the
* {@code FUNCTION} node disappears and {@code DETACH DELETE} takes the edge along — which is why the
* bug hid behind a green suite.
*/
@QuarkusTest
class StaleCallEdgeReapIT {
private static final String PROJECT = "nat-stale-call-reap";
private static final String TARGET_A = """
* First call target.
DEFINE DATA LOCAL
END-DEFINE
END
""";
private static final String TARGET_B = """
* Second call target.
DEFINE DATA LOCAL
END-DEFINE
END
""";
/** Calls RTARGETA from a subroutine that must survive the edit unchanged. */
private static final String CALLER_A = """
* Caller in its first state.
DEFINE DATA LOCAL
01 #TGT (A8)
END-DEFINE
*
PERFORM CALL-STEP
*
DEFINE SUBROUTINE CALL-STEP
CALLNAT 'RTARGETA'
END-SUBROUTINE
*
END
""";
/** Only the target changed — same subroutine, same line, same everything else. */
private static final String CALLER_B = """
* Caller in its first state.
DEFINE DATA LOCAL
01 #TGT (A8)
END-DEFINE
*
PERFORM CALL-STEP
*
DEFINE SUBROUTINE CALL-STEP
CALLNAT 'RTARGETB'
END-SUBROUTINE
*
END
""";
/** Calls a module that does not exist, so the target is an unresolved placeholder node. */
private static final String PHANTOM_CALLER = """
* Caller whose target does not exist — a placeholder, as a pre-fix parser artefact is.
DEFINE DATA LOCAL
END-DEFINE
*
PERFORM CALL-STEP
*
DEFINE SUBROUTINE CALL-STEP
CALLNAT 'RPHANTOM'
END-SUBROUTINE
*
END
""";
/** Same subroutine, now calling a real module: the phantom must be reaped, node and all. */
private static final String PHANTOM_CALLER_FIXED = """
* Caller whose target does not exist — a placeholder, as a pre-fix parser artefact is.
DEFINE DATA LOCAL
END-DEFINE
*
PERFORM CALL-STEP
*
DEFINE SUBROUTINE CALL-STEP
CALLNAT 'RTARGETA'
END-SUBROUTINE
*
END
""";
@TempDir
static Path root;
@BeforeAll
static void createProject() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
write("RTARGETA.nat", TARGET_A);
write("RTARGETB.nat", TARGET_B);
write("RCALLER.nat", CALLER_A);
write("RPHANT.nat", PHANTOM_CALLER);
given().contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
.when().post("/api/projects/" + PROJECT)
.then().statusCode(201);
}
private static void write(String fileName, String content) {
try {
Files.writeString(root.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
private static void refresh() {
given().when().post("/api/projects/" + PROJECT + "/refresh?deep=true").then().statusCode(200);
}
private static io.restassured.path.json.JsonPath callees(String module) {
return given().pathParam("name", module)
.when().get("/api/projects/" + PROJECT + "/modules/{name}/callees")
.then().statusCode(200).extract().jsonPath();
}
/**
* One test per transition rather than per assertion: each asserts a before/after pair on shared
* mutable server state, so splitting the halves would make the outcome depend on JUnit method order.
*/
@Test
void aRetargetedCallDoesNotKeepItsOldTarget() {
write("RCALLER.nat", CALLER_A);
refresh();
// Guard: prove the first edge exists, so the "gone" assertion cannot pass vacuously.
org.junit.jupiter.api.Assertions.assertTrue(
callees("RCALLER").getList("items.name").contains("RTARGETA"),
"the first target must be there before we can prove it is removed");
write("RCALLER.nat", CALLER_B);
refresh();
var names = callees("RCALLER").getList("items.name");
org.junit.jupiter.api.Assertions.assertTrue(names.contains("RTARGETB"),
"the new target must be present: " + names);
org.junit.jupiter.api.Assertions.assertFalse(names.contains("RTARGETA"),
"the old target outlived the call it was parsed from: " + names);
}
/**
* The placeholder case — the shape a fixed parser bug actually leaves behind. Both the edge and the
* now-edgeless placeholder node must go; the node is never file-swept, so without the item-124 node
* sweep it would keep surfacing in identifier search as an unresolved call target nothing calls.
*/
@Test
void aReapedPlaceholderTargetIsAlsoRemovedAsANode() {
write("RPHANT.nat", PHANTOM_CALLER);
refresh();
org.junit.jupiter.api.Assertions.assertTrue(
callees("RPHANT").getList("items.name").contains("RPHANTOM"),
"the phantom target must exist before we can prove it is swept");
write("RPHANT.nat", PHANTOM_CALLER_FIXED);
refresh();
var names = callees("RPHANT").getList("items.name");
org.junit.jupiter.api.Assertions.assertFalse(names.contains("RPHANTOM"),
"the stale placeholder edge survived: " + names);
given().queryParam("name", "RPHANTOM")
.when().get("/api/projects/" + PROJECT + "/search/identifier")
.then().statusCode(200)
.body("findAll { it.name == 'RPHANTOM' }", is(empty()));
}
/**
* The reap deletes <em>all</em> of a re-parsed file's call edges, including the two kinds enrichment
* builds rather than the parser: {@code folded} and {@code resolvedBy: 'manual'}. Both are rebuilt by
* finalize steps that run at every enrichment level — this pins that, because if it were false the
* reap would silently discard a user's manual override on every refresh.
*/
@Test
void aManualOverrideSurvivesTheReap() {
write("RCALLER.nat", CALLER_B);
refresh();
given().contentType("application/json")
.body(Map.of("originFile", "RPHANT.nat", "lineNo", 8,
"targets", List.of("RTARGETB"), "variable", "MANUAL-PIN"))
.when().post("/api/projects/" + PROJECT + "/dynamic-calls/overrides")
.then().statusCode(200);
refresh();
given().when().get("/api/projects/" + PROJECT + "/dynamic-calls/overrides")
.then().statusCode(200)
.body("findAll { it.originFile == 'RPHANT.nat' }", not(empty()));
}
}

View File

@@ -0,0 +1,201 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import org.junit.jupiter.api.*;
import org.junit.jupiter.api.io.TempDir;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.file.Files;
import java.nio.file.Path;
import static io.restassured.RestAssured.given;
import static org.hamcrest.Matchers.*;
/**
* Item 129: a refresh must be able to touch only what changed, and an interrupted one must not look
* like a clean graph.
*
* <p>Covers the three pieces separately because they are independent: {@code ?paths=} (targeted
* re-ingest), the {@code incomplete} marker (in-flight / aborted), and opt-in {@code ?changedOnly=}
* skipping. The assertions about what {@code changedOnly} deliberately does <em>not</em> do matter as
* much as the skipping itself — an incremental refresh that quietly kept stale copycode expansions
* would be worse than no incremental refresh at all.
*/
@QuarkusTest
@TestMethodOrder(MethodOrderer.OrderAnnotation.class)
class TargetedRefreshIT {
private static final String PROJECT = "targeted-refresh-project";
@TempDir
static Path root;
@BeforeAll
static void createProject() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
writeSource("MAIN.nat", """
DEFINE DATA
LOCAL
1 #A (A8)
END-DEFINE
*
CALLNAT 'HELPER'
END
""");
writeSource("HELPER.nat", """
DEFINE DATA
LOCAL
1 #B (A8)
END-DEFINE
*
END
""");
given()
.contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
.when().post("/api/projects/" + PROJECT)
.then()
.statusCode(201);
}
private static void writeSource(String fileName, String content) {
try {
Files.writeString(root.resolve(fileName), content);
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
@Test
@Order(1)
void pathsReIngestsOnlyTheNamedFile() {
given()
.when().post("/api/projects/" + PROJECT + "/refresh?paths=MAIN.nat")
.then()
.statusCode(200)
.body("examinedFiles", contains("MAIN.nat"))
.body("unresolved", empty());
}
@Test
@Order(2)
void aPathThatMatchesNothingIsReportedRatherThanDropped() {
given()
.when().post("/api/projects/" + PROJECT + "/refresh?paths=MAIN.nat,NOPE.nat")
.then()
.statusCode(200)
.body("examinedFiles", contains("MAIN.nat"))
.body("unresolved", contains("NOPE.nat"));
}
@Test
@Order(3)
void aPathEscapingTheRootIsRefusedAsUnresolved() {
given()
.when().post("/api/projects/" + PROJECT + "/refresh?paths=../outside.nat")
.then()
.statusCode(200)
.body("ingested", equalTo(0))
.body("unresolved", contains("../outside.nat"));
}
@Test
@Order(4)
void aTargetedRefreshDoesNotMoveTheProjectsIngestedAt() {
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
String before = given().when().get("/api/projects/" + PROJECT).jsonPath().getString("ingest.ingestedAt");
given().when().post("/api/projects/" + PROJECT + "/refresh?paths=MAIN.nat").then().statusCode(200);
given().when().get("/api/projects/" + PROJECT)
.then().body("ingest.ingestedAt", equalTo(before));
}
/**
* A path that exists but is not an ingestible source file must be reported, not accepted. It was
* silently listed as examined while nothing about it could be ingested — found by running
* {@code ?paths=pom.xml,...} against the real server, where it came back with an empty
* {@code unresolved}.
*/
@Test
@Order(4)
void aPathThatIsNotAnIngestibleSourceFileIsUnresolved() {
writeSource("notes.txt", "not source\n");
given()
.when().post("/api/projects/" + PROJECT + "/refresh?paths=MAIN.nat,notes.txt")
.then()
.statusCode(200)
.body("examinedFiles", contains("MAIN.nat"))
.body("unresolved", contains("notes.txt"));
}
@Test
@Order(5)
void aCompletedRefreshLeavesTheProjectNotIncomplete() {
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
given().when().get("/api/projects/" + PROJECT)
.then()
.statusCode(200)
.body("ingest.incomplete", equalTo(false))
.body("ingest.ingestedAt", notNullValue());
}
@Test
@Order(6)
void changedOnlyExaminesJustTheChangedFile() {
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
writeSource("HELPER.nat", """
DEFINE DATA
LOCAL
1 #B (A8)
1 #C (N4)
END-DEFINE
*
END
""");
given()
.when().post("/api/projects/" + PROJECT + "/refresh?changedOnly=true")
.then()
.statusCode(200)
.body("examinedFiles", contains("HELPER.nat"));
}
@Test
@Order(7)
void changedOnlyWithNoChangesExaminesNothing() {
given()
.when().post("/api/projects/" + PROJECT + "/refresh?changedOnly=true")
.then()
.statusCode(200)
.body("examinedFiles", empty())
.body("ingested", equalTo(0));
}
/**
* The correctness case that keeps {@code changedOnly} opt-in: copycode text is inlined into the
* including module at parse time, so a module whose {@code .cpy} changed parses differently while
* its own hash is unchanged. Skipping it would leave a stale expansion in the graph with nothing
* to indicate it — so a changed copycode must re-parse everything, not just itself.
*/
@Test
@Order(8)
void aChangedCopycodeDisablesSkippingForTheWholeRun() {
writeSource("SHARED.cpy", "* shared\n");
writeSource("USER.nat", """
DEFINE DATA
LOCAL
1 #D (A8)
END-DEFINE
*
INCLUDE SHARED
END
""");
given().when().post("/api/projects/" + PROJECT + "/refresh").then().statusCode(200);
writeSource("SHARED.cpy", "* shared, now different\n");
given()
.when().post("/api/projects/" + PROJECT + "/refresh?changedOnly=true")
.then()
.statusCode(200)
.body("examinedFiles", hasItems("MAIN.nat", "HELPER.nat", "USER.nat"));
}
}

View File

@@ -0,0 +1,17 @@
package com.example.nodb;
import jakarta.persistence.Entity;
import jakarta.persistence.Id;
import jakarta.persistence.Table;
/**
* The only difference between the two item-140 fixture projects: this entity gives one of them a
* {@code DB_TABLE}, which switches the reaper off.
*/
@Entity
@Table(name = "LEDGER")
public class Ledger {
@Id
public Long id;
}

View File

@@ -0,0 +1,13 @@
package com.example.nodb;
/**
* Touches no database at all — every call below is a static utility getter. Item 140 fixture.
*/
public class ReportRenderer {
public String render(String line) {
String key = TextUtils.getColumn(line, 0, 8);
String text = TextUtils.getTrimmed(TextUtils.getColumn(line, 8, 40));
return key + ": " + text;
}
}

View File

@@ -0,0 +1,20 @@
package com.example.nodb;
/**
* A plain static string utility. Item 140 fixture: its {@code get}-prefixed methods are exactly the
* shape that {@code JavaParser#addDbAccessCandidate} mistakes for a persistence read, because the
* read gate lets through any static receiver.
*/
public final class TextUtils {
private TextUtils() {
}
public static String getColumn(String line, int from, int to) {
return line.substring(from, to);
}
public static String getTrimmed(String value) {
return value == null ? "" : value.trim();
}
}

View File

@@ -0,0 +1,8 @@
* ----------------------------------------------------------------------
* Copycode member of the browse family, shaped after the upms idiom: the
* caller passes a sort key as &1&, the browse module's own name as &2& and
* its key field as &3&. The CALLNAT target is therefore never literal here —
* it exists only after substitution, which is what items 120/121/123 broke.
* ----------------------------------------------------------------------
ASSIGN #SORT-KEY = &1&
CALLNAT &2& &3&

View File

@@ -0,0 +1,14 @@
* Title : Host module exercising the three argument-parsing defects that
* kept 263 module-to-module calls out of the upms graph. Its INCLUDE spreads
* the arguments over two lines (item 120), writes one of them with Natural's
* doubled-quote escape '''X''' (item 121), and one with the double-quote
* delimiter '"X"' (item 123). Only when all three hold does YAGCHBN0 appear.
DEFINE DATA
LOCAL
01 #SORT-KEY (A20)
END-DEFINE
*
INCLUDE BROWSECPY '''AGNT-CHG-CMP-SP'''
'"YAGCHBN0"' '''YAGCHKEY'''
*
END

View File

@@ -0,0 +1,11 @@
* ----------------------------------------------------------------------
* Item 122: the guarded assignment lives HERE, spliced into the host's
* DECIDE branch by an INCLUDE. Its dispatch row's lineNo is a line of this
* file — line 8 of NESTDISP.nat is a comment in the module header.
* ----------------------------------------------------------------------
DECIDE ON FIRST VALUE OF #FIELD-NAME
VALUE 'TX-FROM-CPY'
MOVE 'DESC-FROM-CPY' TO #OUT-DESC
NONE
IGNORE
END-DECIDE

View File

@@ -31,6 +31,8 @@ DECIDE ON FIRST VALUE OF #SHORT-VIEW
END-DECIDE
VALUE 'APRF'
MOVE 'CODE-APRF' TO #OUT-CODE
VALUE 'CPYV'
INCLUDE DISPCPY
NONE
IGNORE
END-DECIDE

View File

@@ -67,10 +67,17 @@ public final class CypherQueries {
* fresh id (kept) while a renamed/removed field keeps its old id (deleted). Only files with a real
* {@code sourceFile} are reconciled; {@code sourceFile = ""} placeholders are shared across files
* and never swept here. Fixes stale identifier-index nodes that outlived a {@code refresh}.
*
* <p>Items 75-B/75-C: keyed on the {@code (sourceFile, ownerModule)} <em>pair</em>, not the file
* alone. A copycode-resident node carries its expansion site ({@code <hostFile>#<includePath>}) as
* {@code ownerModule}, so many nodes share one {@code sourceFile}; sweeping by file alone would
* delete every other includer's nodes (their {@code ingestGen} is one transaction old) together
* with their edges. A module's own nodes carry {@code ownerModule = ""}, so their sweep is
* unchanged.
*/
public static final String DELETE_STALE_FILE_NODES = """
UNWIND $sourceFiles AS sf
MATCH (n:AstNode {project: $project, sourceFile: sf})
UNWIND $files AS p
MATCH (n:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o})
WHERE n.ingestGen IS NULL OR n.ingestGen <> $ingestGen
DETACH DELETE n
""";
@@ -103,10 +110,10 @@ public final class CypherQueries {
* re-ingest, so a coarse Tier-1 (no-finalize) scan never strips resolved edges it cannot rebuild.
*/
public static final String DELETE_STALE_RESOLVED_FIELD_EDGES = """
UNWIND $sourceFiles AS f
MATCH (src:AstNode {project: $project, sourceFile: f})-[r:READS|WRITES]->(fld:AstNode)
UNWIND $files AS p
MATCH (src:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o})-[r:READS|WRITES]->(fld:AstNode)
WHERE (fld.type = 'VARIABLE' OR fld.type = 'CONSTANT')
AND fld.sourceFile <> "" AND fld.sourceFile <> f
AND fld.sourceFile <> "" AND fld.sourceFile <> p.f
DELETE r
""";
@@ -124,8 +131,8 @@ public final class CypherQueries {
* coarse re-ingest both re-emit them, so this is not gated on {@code reconcile}.
*/
public static final String DELETE_STALE_NATURAL_TABLE_ACCESS_EDGES = """
UNWIND $sourceFiles AS f
MATCH (src:AstNode {project: $project, sourceFile: f, language: 'natural'})-[r:READS|WRITES]->(t:AstNode {project: $project, sourceFile: ""})
UNWIND $files AS p
MATCH (src:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o, language: 'natural'})-[r:READS|WRITES]->(t:AstNode {project: $project, sourceFile: ""})
WHERE t.type IN ['DB_TABLE', 'WORKFILE']
DELETE r
""";
@@ -147,8 +154,67 @@ public final class CypherQueries {
* identically.
*/
public static final String DELETE_STALE_NATURAL_USING_EDGES = """
UNWIND $sourceFiles AS f
MATCH (src:AstNode {project: $project, sourceFile: f, language: 'natural'})-[r:INCLUDES]->(d:AstNode {project: $project, type: 'DATA_STRUCTURE'})
UNWIND $files AS p
MATCH (src:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o, language: 'natural'})-[r:INCLUDES]->(d:AstNode {project: $project, type: 'DATA_STRUCTURE'})
DELETE r
""";
/**
* Item 124: the {@code CALLS} counterpart of items 86 and 106 — reaps a re-parsed Natural file's
* call edges before the fresh ones are merged.
*
* <p>Item 58's node sweep only removes nodes the fresh parse no longer produces, so a call edge
* survives whenever <em>both</em> its endpoints legitimately survive. That is the normal case when a
* <b>parser fix</b> changes a call's target: the calling subroutine is unchanged (its {@code FUNCTION}
* node is re-merged and kept) and the old target is a placeholder ({@code sourceFile=""}, never
* file-swept), so nothing reaps the edge between them — the corrected call is merely <em>added</em>
* beside the wrong one. After items 120/121/123 shipped, {@code DAGCHEN0/callees} listed both the
* real {@code YAGCHBN0} and the pre-fix phantom {@code AGNT-CHG-CMP-SP} from the same call site;
* 497 such edges survived a full deep refresh of {@code upms}.
*
* <p>Reaches beyond item 86's placeholder-only scope on purpose: bug #63's `CALLNAT`-in-a-string
* matches resolved onto <b>real</b> modules, so a placeholder-only reap would leave that whole
* class of artefact behind. The one thing it must <em>not</em> touch is the dynamic-call
* resolvers' own output, hence the {@code WHERE}:
*
* <ul>
* <li><b>Reaped</b> — every edge to a placeholder target ({@code sourceFile=""}), plus every
* {@code PERFORM}/{@code CALLNAT}/{@code INCLUDE_MACRO} edge. All of these are emitted by the
* parser on every parse, at <em>both</em> tiers (the coarse scanner expands copycode exactly
* as the deep parser does), so the merge that follows re-creates the current ones and
* unchanged edges round-trip identically.</li>
* <li><b>Kept</b> — {@code CALLNAT_DYNAMIC} edges to a <em>real</em> module. The parser cannot
* know a dynamic target and always emits a placeholder, so such an edge is by construction
* enrichment-built: a fold, a manual override, or an intra/cross resolution. In {@code upms}
* 486 edges are of this kind and <b>392 of them carry no marker at all</b> — no
* {@code folded}, no {@code resolvedBy} — so the target's file is the only thing that
* distinguishes them from parser output.</li>
* </ul>
*
* <p>That exception is not theoretical: reaping them broke
* {@code AnalysisResourceIT#flowForwardPathWarmCrossesIntoDynamicallyDispatchedCallee}. The item-37a
* path-warm resolves a cross-module dispatch in one round and re-ingests in the next, and a
* <em>scoped</em> finalize does not reliably re-resolve an edge whose far side is outside its scope —
* so the reap deleted a resolution nothing rebuilt, and the dataflow trace stopped at the dispatch
* boundary.
*
* <p>Gated on {@code reconcile} (deep re-ingest), because {@code ..._INTRA_INDIRECT} and
* {@code ..._CROSS} are gated on {@code resolveFields}/{@code dataflow} and so do <em>not</em> run
* in a coarse finalize. Same rationale as item 74.
*
* <p><b>Item 75-B — the copycode gap is closed.</b> This used to key on the source node's file
* alone and therefore had to miss the 605 call edges whose source subroutine is defined
* <em>inside a copycode</em> (14 of them stale): those nodes were MERGEd per
* {@code (type, name, sourceFile)} and thus shared by every includer, so reaping during one
* module's refresh would have deleted edges the other includers contributed. Copycode-resident
* nodes now carry the includer's file as {@code ownerModule}, so the key is the
* {@code (sourceFile, ownerModule)} pair and the reap touches exactly the re-parsed module's own
* copies — which its own parse re-emits. Same for items 86 and 106 above.
*/
public static final String DELETE_STALE_NATURAL_CALL_EDGES = """
UNWIND $files AS p
MATCH (src:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o, language: 'natural'})-[r:CALLS]->(t)
WHERE t.sourceFile = "" OR r.callKind <> 'CALLNAT_DYNAMIC'
DELETE r
""";
@@ -231,6 +297,33 @@ public final class CypherQueries {
DELETE t
""";
/**
* Item 124, node half: the {@code MODULE} counterpart of item 88's table sweep. Once
* {@link #DELETE_STALE_NATURAL_CALL_EDGES} reaps the last call to a call-target placeholder, the
* placeholder <em>node</em> is left edgeless — it is never file-swept ({@code sourceFile=""}) — and
* still surfaces in {@code search/identifier} and the module inventory as an unresolved call target
* that nothing actually calls.
*
* <p>Guarded on {@code sourceFile = ""} so it can only ever remove a placeholder: a <b>real</b>
* parsed module that happens to call nothing and be called by nothing is a legitimate standalone
* program and must survive. Degree-0 only, so a placeholder still referenced by any call is kept.
* Runs beside item 88's table sweep, after all edge resolution.
*
* <p>{@code duplicatePaths IS NULL} excludes item 114's duplicate markers, which are a
* <em>second</em> reason for an edgeless placeholder to exist and are deliberately kept: an
* unreferenced duplicate identity is a degree-0 placeholder whose whole purpose is to record "this
* name exists in two files", so the module endpoints can answer {@code 409 DUPLICATE_IDENTITY}
* rather than {@code 404}. {@link #MARK_DUPLICATE_IDENTITIES} runs <em>before</em> finalize, so
* without this guard the sweep deleted the marker it had just written and the endpoints fell back
* to {@code 404} — caught by {@code DuplicateIdentityIT}. Markers are reaped on their own terms by
* {@link #CLEAR_DUPLICATE_MARKERS} when the source conflict goes away.
*/
public static final String DELETE_ORPHANED_PLACEHOLDER_MODULES = """
MATCH (t:AstNode {project: $project, type: 'MODULE', sourceFile: ""})
WHERE t.duplicatePaths IS NULL AND NOT (t)--()
DELETE t
""";
public static final String MODULE_SOURCE_FILE = """
MATCH (m:MODULE {project: $project})
// Item 117: this endpoint is not behind the resolving guard, so it accepts the identity or
@@ -248,6 +341,18 @@ public final class CypherQueries {
* slice. Returns {@code null} when no hash was stored (e.g. a legacy project), so the read falls
* back to serving the snippet unchecked.
*/
/**
* Item 129: every ingested file's stored content hash in one round trip, so a {@code changedOnly}
* refresh can decide what to re-parse without a query per file. One row per {@code sourceFile};
* files whose shell carries no hash (never coarse-scanned) are absent, and therefore treated as
* changed — the safe direction.
*/
public static final String SOURCE_HASHES = """
MATCH (n:AstNode {project: $project})
WHERE n.sourceHash IS NOT NULL AND n.sourceFile <> ""
RETURN DISTINCT n.sourceFile AS sourceFile, n.sourceHash AS hash
""";
public static final String SOURCE_HASH = """
MATCH (n:AstNode {project: $project, sourceFile: $sourceFile})
WHERE n.sourceHash IS NOT NULL
@@ -366,7 +471,16 @@ public final class CypherQueries {
public static final String MARK_DUPLICATE_IDENTITIES = """
UNWIND $duplicates AS d
MERGE (n:AstNode {type: d.type, name: d.name, sourceFile: '', project: $project, ownerModule: ''})
SET n:$(d.type), n.duplicatePaths = d.paths
SET n:$(d.type), n.duplicatePaths = d.paths,
// Item 136: every AstNode must carry startLine/endLine (CLAUDE.md), and this MERGE was the
// one node-creating query that did not. A marker has no line — 0 is a convention, not a
// truth — but the alternative, letting these two properties be null on a handful of nodes,
// forces null handling into every row mapper in the project; one of them missed it and
// faulted `search/identifier` with an unstructured 500 once a page reached row 487.
// coalesce, because the MERGE key is deliberately the same as MERGE_NODES': when something
// already references the skipped identity, marker and placeholder are one node and the
// placeholder's real lines must survive.
n.startLine = coalesce(n.startLine, 0), n.endLine = coalesce(n.endLine, 0)
""";
/**
@@ -1070,6 +1184,34 @@ public final class CypherQueries {
MERGE (fn)-[:WRITES {lineNo: a.startLine}]->(t))
""";
/**
* Item 140: in a project that has <b>no</b> {@code DB_TABLE} at all, every Java
* {@code DB_ACCESS} is a false positive by construction — a table node only ever comes from a
* JPA/Panache entity, so with none in the graph not a single candidate can resolve, and the
* over-approximating parse-time heuristic ({@code JavaParser#addDbAccessCandidate}, whose read
* gate lets through <em>any</em> static receiver) is left standing alone. Reaps them, so
* {@code sql-statements} — which joins the table with {@code OPTIONAL MATCH} and therefore does
* emit unresolved candidates as {@code table: null} rows — stays silent instead of reporting
* {@code UserContext.getCurrent()} as a database read.
*
* <p>Deliberately <b>Java only</b>: a Natural {@code DB_ACCESS} comes from a literal
* {@code READ}/{@code FIND}/{@code STORE} statement and is a database access whether or not its
* view resolved, so the same reaping would destroy real information there.
*
* <p>Must run after all three resolvers. Idempotent. Two known limits, both accepted: a project
* whose DB access is exclusively native SQL has no entity and thus no table, so its (already
* unresolvable) accesses are reaped too; and once such a project gains its first entity, the
* reaped nodes only return for files that are actually re-parsed — a {@code changedOnly}
* refresh will not bring them back, a full one will.
*/
public static final String REAP_JAVA_DB_ACCESS_WITHOUT_TABLES = """
OPTIONAL MATCH (t:DB_TABLE {type: 'DB_TABLE', project: $project})
WITH count(t) AS tables
WHERE tables = 0
MATCH (a:DB_ACCESS {type: 'DB_ACCESS', project: $project, language: 'java'})
DETACH DELETE a
""";
/**
* Dataflow enrichment: for each {@code CALLS} edge carrying an {@code args} list (the
* positional {@code CALLNAT} arguments), links each caller argument variable to the callee
@@ -1348,7 +1490,9 @@ public final class CypherQueries {
*/
public static final List<EdgeType> RESOLVABLE_EDGE_TYPES =
List.of(EdgeType.CALLS, EdgeType.INCLUDES, EdgeType.USES_TYPE, EdgeType.EXTENDS, EdgeType.IMPLEMENTS,
EdgeType.INJECTS, EdgeType.REFERENCES);
// Item 128: a mention's target is a placeholder until enrichment resolves it to the
// real module, exactly like a call's — otherwise every import would dangle.
EdgeType.INJECTS, EdgeType.REFERENCES, EdgeType.MENTIONS);
/**
* Scoped variant of {@link #LINK_ARGS_TO_PARAMS_JAVA}: callers in {@code $names} only.
@@ -1586,6 +1730,7 @@ public final class CypherQueries {
ON CREATE SET r2.folded = true
SET r2.dynamicVar = dyn.dynamicVar, r2.args = dyn.args
""";
/**
* Deletes the dynamic-call marker edges (the {@code CALLNAT_DYNAMIC} {@code CALLS} edges that
* point at a variable-named placeholder, {@code sourceFile = ""}) <em>only for call sites that
@@ -1906,7 +2051,10 @@ public final class CypherQueries {
CASE WHEN w.whenValues IS NULL THEN [w.whenValue] ELSE split(w.whenValues, '\\u001F') END
AS guardValues,
w.whenChainFields AS chainFields, w.whenChainValues AS chainValues,
t.name AS assignedField, w.value AS assignedValue, w.lineNo AS lineNo
t.name AS assignedField, w.value AS assignedValue, w.lineNo AS lineNo,
coalesce(w.originFile, m.sourceFile) AS sourceFile,
w.viaCopycode AS viaCopycode, w.includedAt AS includedAt,
w.includePath AS includePath
ORDER BY guardValue, assignedValue, lineNo
""";
@@ -2082,7 +2230,12 @@ public final class CypherQueries {
public static final String SEARCH_BY_VALUE = """
MATCH (n:AstNode {project: $project})
WHERE n.value IS NOT NULL AND trim(replace(n.value, "'", "")) = $value
RETURN 'NODE' AS kind, n.name AS name, n.value AS value, null AS module,
// Item 141: a comment block's text lives in `value`, so it would otherwise join this
// result set by default and move every existing completeness count (item 131's lesson).
// Opt-in only; the rows are marked 'COMMENT' so a comment hit is never read as code.
AND ($includeComments OR n.type <> 'COMMENT')
RETURN CASE WHEN n.type = 'COMMENT' THEN 'COMMENT' ELSE 'NODE' END AS kind,
n.name AS name, n.value AS value, null AS module,
n.sourceFile AS sourceFile, n.startLine AS startLine, n.endLine AS endLine
UNION
MATCH (m:AstNode {type: 'MODULE', project: $project})-[:CONTAINS*0..1]->(src:AstNode)-[w:WRITES]->(v:AstNode)
@@ -2091,6 +2244,79 @@ public final class CypherQueries {
m.sourceFile AS sourceFile, w.lineNo AS startLine, w.lineNo AS endLine
""";
/**
* Item 130: the project's REST surface — one row per handler method, with the endpoint path
* composed from the class-level and method-level {@code @Path} (item 130 persists both as
* {@code restPath}).
*
* <p>"Endpoint path → controller method → logic" is the daily question in a JAX-RS codebase and is
* derivable from the annotations, but it took two {@code /search/annotation} calls and a manual
* join to answer, because annotations are stored by name without their arguments.
*
* <p>A method with no {@code httpMethod} is not an endpoint (a sub-resource locator, a helper) and
* is excluded. Path segments are joined with exactly one {@code /}, and a class with no
* {@code @Path} contributes nothing to the prefix rather than a literal "null".
*/
private static final String REST_ENDPOINTS_CORE = """
MATCH (m:AstNode {project: $project, type: 'MODULE'})-[:CONTAINS]->(f:AstNode {type: 'FUNCTION'})
WHERE f.httpMethod IS NOT NULL
AND ($module IS NULL OR m.name = $module OR m.simpleName = $module)
// JAX-RS inherits @Path from a base class or interface, and this codebase uses that
// heavily (AbstractFileTransferUiSvc and friends). Without the ancestor lookup those
// endpoints all reported the bare path '/', which is a wrong answer, not a missing one.
// Nearest ancestor wins; the depth bound keeps this from walking deep hierarchies.
OPTIONAL MATCH ancestry = (m)-[:EXTENDS|IMPLEMENTS*1..4]->(base:AstNode {type: 'MODULE'})
WHERE base.restPath IS NOT NULL
WITH m, f, base, length(ancestry) AS depth
ORDER BY depth
WITH m, f, head(collect(base.restPath)) AS inheritedPath
WITH m, f, coalesce(m.restPath, inheritedPath, '') AS classPath,
coalesce(f.restPath, '') AS methodPath
// Normalise each half by stripping its own leading and trailing '/', then join the
// non-empty ones with exactly one '/'. The previous single replace('//', '/') was wrong:
// replacement is non-overlapping, so '///file' collapsed to '//file' rather than '/file'.
WITH m, f, [p IN [classPath, methodPath] WHERE p <> '' AND p <> '/' |
CASE WHEN left(p, 1) = '/' THEN substring(p, 1) ELSE p END] AS lead
WITH m, f, [p IN lead |
CASE WHEN size(p) > 0 AND right(p, 1) = '/'
THEN left(p, size(p) - 1) ELSE p END] AS parts
WITH m, f, '/' + reduce(acc = '', p IN parts |
CASE WHEN acc = '' THEN p ELSE acc + '/' + p END) AS path
// DISTINCT is load-bearing: the graph can hold more than one CONTAINS edge between the
// same module and function (see item 75), which multiplied every such endpoint into
// identical rows — 183 of 436 on `pur`.
""";
/**
* The row projection; {@link #REST_ENDPOINTS_COUNT} must stay distinct over the same columns.
*/
private static final String REST_ENDPOINTS_ROW = """
DISTINCT f.httpMethod AS httpMethod, path AS path,
m.name AS module, m.simpleName AS moduleSimpleName, f.name AS handler,
f.sourceFile AS sourceFile, f.startLine AS startLine,
// A @RegisterRestClient interface declares a call this application *makes*, not one
// it serves. Listing those as endpoints (3 in `pur`) states the traffic's direction
// backwards; they are flagged rather than dropped, because "who calls out to what"
// is a real question too.
coalesce(m.annotations, '') CONTAINS 'RegisterRestClient' AS outbound
""";
public static final String REST_ENDPOINTS = REST_ENDPOINTS_CORE + "RETURN " + REST_ENDPOINTS_ROW + """
ORDER BY path, httpMethod
LIMIT $scanCap
""";
/**
* Item 135: the endpoint count. Distinct over the full projected column set, not just
* {@code (module, handler)}: the narrower key would be equal here only by accident of the data
* model, and a count that disagrees with the page it describes is worse than none. Being distinct
* also makes it immune to item 75's duplicate {@code CONTAINS} edges — which it masks exactly as
* the row query does; item 75 is still open.
*/
public static final String REST_ENDPOINTS_COUNT = REST_ENDPOINTS_CORE + "WITH " + REST_ENDPOINTS_ROW + """
RETURN count(*) AS total
""";
private static String flowQuery(boolean forward, int maxDepth) {
String path = forward
? "(v)-[:ARG_TO_PARAM*1..%d]->(t:AstNode)"
@@ -2138,7 +2364,9 @@ public final class CypherQueries {
public static final String SEARCH_BY_VALUE_CONTAINS = """
MATCH (n:AstNode {project: $project})
WHERE n.value IS NOT NULL AND toLower(replace(n.value, "'", "")) CONTAINS toLower($value)
RETURN 'NODE' AS kind, n.name AS name, n.value AS value, null AS module,
AND ($includeComments OR n.type <> 'COMMENT')
RETURN CASE WHEN n.type = 'COMMENT' THEN 'COMMENT' ELSE 'NODE' END AS kind,
n.name AS name, n.value AS value, null AS module,
n.sourceFile AS sourceFile, n.startLine AS startLine, n.endLine AS endLine
UNION
MATCH (m:AstNode {type: 'MODULE', project: $project})-[:CONTAINS*0..1]->(src:AstNode)-[w:WRITES]->(v:AstNode)
@@ -2206,6 +2434,35 @@ public final class CypherQueries {
m.sourceFile AS sourceFile
ORDER BY lineNo, tag
""";
/**
* Item 141: the comment blocks of a module, each with the declaration it documents.
*
* <p>Matched by the module's own {@code sourceFile} rather than through the {@code DOCUMENTS}
* edge, because a comment documents a <em>declaration</em> (a field, a subroutine), and walking
* back from there to "which module" would have to re-derive containment for every row. A comment
* node only ever comes from the module's own file — copycode comments belong to the copycode,
* which is its own module (see {@code NaturalParser.withComments}) — so the file <em>is</em> the
* module scope.
*
* <p>{@code $kind} filters to one comment kind. With no filter, {@code **SAG} generator
* directives are left out: they are machine-written metadata rather than a human note, and they
* would otherwise be the bulk of the answer for every generated Natural module. Ask for
* {@code kind=SAG} to see them.
*/
public static final String MODULE_COMMENTS = """
MATCH (m:MODULE {name: $name, project: $project})
WHERE ($sourceFile = '' OR m.sourceFile = $sourceFile)
MATCH (c:AstNode {type: 'COMMENT', project: $project, sourceFile: m.sourceFile})
WHERE ($kind IS NULL AND c.commentKind <> 'SAG') OR c.commentKind = $kind
OPTIONAL MATCH (c)-[:DOCUMENTS]->(t:AstNode)
RETURN c.value AS text, c.commentKind AS kind, c.sourceFile AS sourceFile,
c.startLine AS startLine, c.endLine AS endLine,
t.name AS target, t.type AS targetType,
coalesce(c.truncated, 'false') AS truncated
ORDER BY startLine
SKIP $offset LIMIT $limit
""";
/**
* Finds nodes carrying a given annotation (item 29, Java only): matches case-insensitively as
* a substring against each name in the node's comma-joined {@code annotations} property (set at
@@ -2213,15 +2470,30 @@ public final class CypherQueries {
* annotation's own specific interpretation elsewhere, e.g. {@code @Entity}/{@code @Query}).
* Optional {@code $type} restricts to one {@code NodeType}.
*/
public static final String SEARCH_ANNOTATION = """
/**
* Item 131: shared predicate behind the annotation search's row and count projections.
*/
private static final String SEARCH_ANNOTATION_CORE = """
MATCH (n:AstNode {project: $project})
WHERE n.annotations IS NOT NULL
AND ($type IS NULL OR n.type = $type)
AND ANY(a IN split(n.annotations, ',') WHERE toLower(a) CONTAINS toLower($name))
""";
public static final String SEARCH_ANNOTATION = SEARCH_ANNOTATION_CORE + """
RETURN elementId(n) AS id, n.type AS type, n.name AS name, n.sourceFile AS sourceFile,
n.startLine AS startLine, n.endLine AS endLine, n.annotations AS annotations
ORDER BY n.name
""";
/**
* Item 131: the annotation search's total. This is the endpoint the item was written about —
* {@code @Immutable} returned 50 of 95 rows with nothing to say so, and a real audit concluded
* that 17 entities had lost the annotation when in fact zero had.
*/
public static final String SEARCH_ANNOTATION_COUNT = SEARCH_ANNOTATION_CORE + """
RETURN count(n) AS total
""";
/**
* Item 44: per-language LoC/SLoC rollup for a project. Aggregates the file-level shell nodes
* ({@code MODULE}/{@code DATA_STRUCTURE} with a non-empty {@code sourceFile}). Because a single
@@ -2296,17 +2568,104 @@ public final class CypherQueries {
// $module is a hard filter (return only that module's nodes); $priorityModule instead only
// pins that module's matches to the front so they survive a caller's limit/paginate truncation
// when a name recurs across many modules. ORDER BY makes the page deterministic (it was not before).
public static final String SEARCH_IDENTIFIER = """
/**
* Item 131: the shared predicate of the identifier search. The row projection and the
* {@code count(*)} projection are composed from this one constant on purpose — two hand-maintained
* copies would drift, and a total that disagrees with the rows it describes is worse than no total
* at all: it turns a visible truncation into a confident wrong number.
*/
private static final String SEARCH_IDENTIFIER_CORE = """
MATCH (n:AstNode {project: $project})
// Item 125: a Java type declaration's identity is its FQN (item 117), so matching n.name
// alone made a class findable only by its fully-qualified name — the short form an agent
// actually types answered [], which reads as "no such name" rather than "wrong form of the
// name". n.simpleName is matched alongside it. It is null for Natural, where the
// sigil-stripped n.name remains the only identity, so nothing changes there.
WITH n, (CASE WHEN left(n.name, 1) IN ['#', '&', '+'] THEN substring(n.name, 1) ELSE n.name END) AS bare
// $contains (item 125) switches the name predicate to a case-insensitive substring match,
// as /search/value already does — "every identifier containing 'upd'" was unanswerable.
// It matches the FQN too, so a package fragment also hits: over-matching is visible and
// filterable (?type=MODULE, moduleKind), silence is not.
WHERE ($name IS NULL OR
(CASE WHEN left(n.name, 1) IN ['#', '&', '+'] THEN substring(n.name, 1) ELSE n.name END) = $name)
CASE WHEN $contains
THEN toLower(bare) CONTAINS toLower($name)
OR (n.simpleName IS NOT NULL AND toLower(n.simpleName) CONTAINS toLower($name))
ELSE bare = $name OR n.simpleName = $name
END)
AND ($type IS NULL OR n.type = $type)
// Item 141: a COMMENT node's name is the synthetic `comment@<line>`, never prose - but it
// is still a name, and this query matches every node's name with no type filter. Excluded
// explicitly so `contains=true` can never drift into returning comment rows.
AND n.type <> 'COMMENT'
AND ($sourceFile IS NULL OR n.sourceFile = $sourceFile)
AND ($module IS NULL OR EXISTS {
MATCH (mod:MODULE {project: $project})
WHERE (mod.name = $module OR mod.simpleName = $module)
AND mod.sourceFile = n.sourceFile
})
""";
/**
* Item 128: <b>every reference site of a name</b> — not just its callers. Unions the edge kinds
* that mean "this file mentions that type": {@code CALLS} (call sites), {@code EXTENDS}/
* {@code IMPLEMENTS} (inheritance), {@code INJECTS} (CDI wiring), {@code INCLUDES} (Natural
* copycode) and {@code REFERENCES}, whose {@code refKind} distinguishes an {@code IMPORT}, a
* declared {@code TYPE} position, an {@code ANNOTATION} usage and the pre-existing class-literal
* edges (no {@code refKind}, reported as {@code CLASS_LITERAL}).
*
* <p>This is what makes "scope a rename" answerable. {@code callers} only sees calls, so a file
* that imports a class, declares a field of it or names it in an annotation was invisible — and
* the rename that missed it looked complete.
*
* <p>The target is matched by identity <em>or</em> short name (item 117/125), so both forms work.
* {@code $kind} narrows to one reference kind; {@code $scanCap} bounds the row set exactly as in
* {@link #SEARCH_IDENTIFIER}.
*/
private static final String SEARCH_REFERENCES_CORE = """
MATCH (t:AstNode {project: $project})
WHERE (t.name = $name OR t.simpleName = $name) AND t.type IN ['MODULE', 'DATA_STRUCTURE']
MATCH (s:AstNode {project: $project})-[r]->(t)
WHERE type(r) IN ['CALLS', 'EXTENDS', 'IMPLEMENTS', 'INJECTS', 'REFERENCES', 'INCLUDES', 'MENTIONS']
AND s.sourceFile <> ''
WITH s, r, t, CASE type(r)
WHEN 'CALLS' THEN 'CALL'
WHEN 'INCLUDES' THEN 'INCLUDE'
WHEN 'REFERENCES' THEN coalesce(r.refKind, 'CLASS_LITERAL')
WHEN 'MENTIONS' THEN coalesce(r.refKind, 'TYPE')
ELSE type(r)
END AS kind
WHERE $kind IS NULL OR kind = $kind
// One hop only: a reference is anchored either at the module itself or at a function
// inside it. A variable-length CONTAINS walk here would scan the whole containment tree
// for every row, and buys nothing this data model can use.
OPTIONAL MATCH (owner:AstNode {project: $project, type: 'MODULE'})-[:CONTAINS]->(s)
WITH s, r, t, kind,
CASE WHEN s.type = 'MODULE' THEN s.name ELSE owner.name END AS inModule
""";
/**
* The row projection: {@link #SEARCH_REFERENCES_COUNT} must stay distinct over the same columns.
*/
private static final String SEARCH_REFERENCES_ROW = """
DISTINCT s.sourceFile AS sourceFile, coalesce(r.lineNo, s.startLine) AS lineNo,
kind AS kind, inModule AS inModule, t.name AS target
""";
public static final String SEARCH_REFERENCES = SEARCH_REFERENCES_CORE + "RETURN " + SEARCH_REFERENCES_ROW + """
ORDER BY sourceFile ASC, lineNo ASC, kind ASC
LIMIT $scanCap
""";
/**
* Item 135: how many reference sites the search <em>would</em> return. Distinct over exactly the
* columns {@link #SEARCH_REFERENCES_ROW} projects — a narrower key would count fewer rows than the
* page delivers. Deliberately carries no {@code $scanCap}, for the reason spelled out on
* {@link #SEARCH_IDENTIFIER_COUNT}: a capped count equals the page size and reports nothing.
*/
public static final String SEARCH_REFERENCES_COUNT = SEARCH_REFERENCES_CORE + "WITH " + SEARCH_REFERENCES_ROW + """
RETURN count(*) AS total
""";
public static final String SEARCH_IDENTIFIER = SEARCH_IDENTIFIER_CORE + """
WITH n, CASE WHEN $priorityModule IS NOT NULL AND EXISTS {
MATCH (pm:MODULE {project: $project})
WHERE (pm.name = $priorityModule OR pm.simpleName = $priorityModule)
@@ -2314,9 +2673,43 @@ public final class CypherQueries {
} THEN 0 ELSE 1 END AS pinRank
RETURN elementId(n) AS id, n.type AS type, n.name AS name, n.sourceFile AS sourceFile,
n.startLine AS startLine, n.endLine AS endLine,
n.dataType AS dataType, n.value AS value, n.scope AS scope, n.unresolved AS unresolved
n.dataType AS dataType, n.value AS value, n.scope AS scope, n.unresolved AS unresolved,
// Item 125: the short form and the declaration kind (CLASS/INTERFACE/ENUM/RECORD,
// PROGRAM/SUBPROGRAM for Natural), so a caller can tell a type declaration from a
// method or a variable without a second round trip. Null for non-MODULE nodes.
n.simpleName AS simpleName, n.moduleKind AS moduleKind
ORDER BY pinRank ASC, n.sourceFile ASC, n.startLine ASC
// $scanCap bounds an unindexed CONTAINS over a large project: it is offset+limit, i.e.
// exactly the rows the caller's page can need, and Integer.MAX_VALUE when the caller asked
// for everything (limit <= 0). Slicing [offset, offset+limit) out of the first offset+limit
// ordered rows is identical to slicing it out of the full ordered list, so paginate()'s
// semantics — including "limit <= 0 means unlimited" — are untouched.
LIMIT $scanCap
""";
/**
* Item 131: how many rows the identifier search <em>would</em> return. Deliberately carries no
* {@code $scanCap}: capping the count to the page size would make it equal the row count every
* time and silently defeat the whole point of reporting a total.
*/
public static final String SEARCH_IDENTIFIER_COUNT = SEARCH_IDENTIFIER_CORE + """
RETURN count(n) AS total
""";
/**
* Item 131: the total for a value search, exact or substring. The union has to be wrapped in a
* {@code CALL} subquery to be counted, and it is a {@code UNION} (not {@code UNION ALL}) in both
* projections — so the count counts the same deduplicated rows the caller can actually page
* through, rather than a larger number no page will ever reach.
*/
public static String searchByValueCount(boolean substring) {
return """
CALL {
""" + (substring ? SEARCH_BY_VALUE_CONTAINS : SEARCH_BY_VALUE) + """
}
RETURN count(*) AS total
""";
}
/**
* Item 78: deletes a project's {@code AstNode}s <b>in batches</b>, leaving its {@code (:Project)}
* shell — the config — untouched. This is what {@code recreate} runs: the shell never leaves the
@@ -2503,7 +2896,17 @@ public final class CypherQueries {
public static final String GET_PROJECT = """
MATCH (p:Project {name: $name})
RETURN p.name AS name, p.description AS description, p.root AS root, p.excludeDirs AS excludeDirs,
p.language AS language, p.generatedDir AS generatedDir, p.userExitDir AS userExitDir
p.language AS language, p.generatedDir AS generatedDir, p.userExitDir AS userExitDir,
// Item 126: what the last whole-root ingest did. Null on a project last ingested
// before this was recorded — "never measured", which is not the same as zero.
p.ingestedAt AS ingestedAt, p.ingestMode AS ingestMode,
p.ingestFilesExamined AS ingestFilesExamined, p.ingestFilesPersisted AS ingestFilesPersisted,
p.ingestFilesFailed AS ingestFilesFailed, p.ingestFailures AS ingestFailures,
p.ingestFailuresTruncated AS ingestFailuresTruncated,
p.ingestDurationSeconds AS ingestDurationSeconds, p.ingestServerVersion AS ingestServerVersion,
// Item 129: true while a whole-root pass is running, and left true by one that never
// finished — the graph is then half-updated and every answer is drawn from that state.
p.ingestIncomplete AS ingestIncomplete, p.ingestStartedAt AS ingestStartedAt
""";
/**
@@ -2528,10 +2931,55 @@ public final class CypherQueries {
public static final String LIST_PROJECTS = """
MATCH (p:Project)
RETURN p.name AS name, p.description AS description, p.root AS root, p.excludeDirs AS excludeDirs,
p.language AS language, p.generatedDir AS generatedDir, p.userExitDir AS userExitDir
p.language AS language, p.generatedDir AS generatedDir, p.userExitDir AS userExitDir,
// Item 126: what the last whole-root ingest did. Null on a project last ingested
// before this was recorded — "never measured", which is not the same as zero.
p.ingestedAt AS ingestedAt, p.ingestMode AS ingestMode,
p.ingestFilesExamined AS ingestFilesExamined, p.ingestFilesPersisted AS ingestFilesPersisted,
p.ingestFilesFailed AS ingestFilesFailed, p.ingestFailures AS ingestFailures,
p.ingestFailuresTruncated AS ingestFailuresTruncated,
p.ingestDurationSeconds AS ingestDurationSeconds, p.ingestServerVersion AS ingestServerVersion,
// Item 129: true while a whole-root pass is running, and left true by one that never
// finished — the graph is then half-updated and every answer is drawn from that state.
p.ingestIncomplete AS ingestIncomplete, p.ingestStartedAt AS ingestStartedAt
ORDER BY p.name
""";
/**
* Item 126: stamps the {@code (:Project)} shell with what a <b>whole-root</b> ingest just did, so
* every later answer can be dated and its completeness judged from the API alone. Written only by
* the whole-root passes — a by-name or fan-out ingest walks a fraction of the tree, and letting it
* move {@code ingestedAt} would report the project as freshly walked when one module was deepened.
*
* <p>Separate from {@link #UPDATE_PROJECT} on purpose: that one {@code COALESCE}s user config, and
* these are server observations. Neither can clobber the other.
*/
/**
* Item 129: marks a whole-root ingest as <b>in flight</b> before it starts. {@link #RECORD_PROJECT_INGEST}
* clears it on success, so a pass that never finished — a crash, a container stop, an aborted deep
* refresh — leaves {@code ingestIncomplete = true} behind and every later answer can say so.
*
* <p>Before this, an interrupted deep refresh was indistinguishable from a clean graph: the earlier
* enrichment steps are already committed, so queries keep answering, just from a half-updated graph.
* The marker cannot self-heal (a killed process clears nothing), and that is the correct direction to
* fail — a false "incomplete" costs one refresh, a false "clean" costs trust in every answer.
*/
public static final String MARK_PROJECT_INGEST_STARTED = """
MATCH (p:Project {name: $name})
SET p.ingestIncomplete = true, p.ingestStartedAt = $startedAt, p.ingestMode = $mode
""";
public static final String RECORD_PROJECT_INGEST = """
MATCH (p:Project {name: $name})
SET p.ingestedAt = $ingestedAt, p.ingestMode = $mode,
p.ingestFilesExamined = $filesExamined, p.ingestFilesPersisted = $filesPersisted,
p.ingestFilesFailed = $filesFailed, p.ingestFailures = $failures,
p.ingestFailuresTruncated = $failuresTruncated,
p.ingestDurationSeconds = $durationSeconds, p.ingestServerVersion = $serverVersion,
// Item 129: reaching here means the pass completed, so the in-flight marker is cleared.
p.ingestIncomplete = false
""";
/**
* @return the callers query, grouping all call sites to the same caller into a single row
* with {@code lineNos}. Includes incoming {@code EXTENDS}/{@code IMPLEMENTS} edges (subtypes of

View File

@@ -1,5 +1,7 @@
package com.agenticcode.neo4jstore.graph;
import org.jspecify.annotations.Nullable;
import java.util.List;
/**
@@ -28,9 +30,22 @@ import java.util.List;
* a blank {@code guardField} also routes here (item 64)
* @param assignedField the field assigned in the branch (e.g. {@code #P-CALLED-PROG})
* @param assignedValue the assigned literal, quotes stripped (e.g. {@code WGEAGB0S})
* @param lineNo source line of the assignment
* @param lineNo source line of the assignment, <em>within {@link #sourceFile}</em>
* @param sourceFile item 122 — the file {@code lineNo} is in. Usually the module's own file, but for a
* row spliced in from a copycode it is that {@code .cpy}. Until this was carried
* through, every endpoint but this one reported provenance, so a copycode row's
* {@code lineNo} read as a line of the host module: 26 of {@code VCOMIN50}'s 44 rows
* pointed at lines 18/20, which there are change-history comments, while the real
* sites were {@code ISICINDE.cpy:18} and {@code ISICINDI.cpy:20}.
* @param viaCopycode the {@code .cpy} member this row was spliced from, or {@code null} when it is
* written directly in the module's own file
* @param includedAt 1-based line of the {@code INCLUDE} in the host module — where to look in the
* module itself. {@code null} unless {@code viaCopycode} is set.
* @param includePath the whole {@code INCLUDE} chain, host first; empty for a row written in the module.
* A nested include is otherwise invisible (item 104).
*/
public record DispatchEntry(List<DispatchGuard> guards, String guardField, String guardValue,
List<String> guardValues, String assignedField, String assignedValue,
int lineNo) {
int lineNo, String sourceFile, @Nullable String viaCopycode,
@Nullable Integer includedAt, List<IncludeStep> includePath) {
}

View File

@@ -1,9 +1,6 @@
package com.agenticcode.neo4jstore.graph;
import com.agenticcode.parsercore.ast.model.AstEdge;
import com.agenticcode.parsercore.ast.model.AstNode;
import com.agenticcode.parsercore.ast.model.EdgeType;
import com.agenticcode.parsercore.ast.model.NodeType;
import com.agenticcode.parsercore.ast.model.*;
import com.agenticcode.parsercore.ast.spi.LanguageParser.ParseResult;
import io.smallrye.mutiny.Uni;
import jakarta.enterprise.context.ApplicationScoped;
@@ -15,6 +12,7 @@ import org.neo4j.driver.Record;
import org.neo4j.driver.summary.SummaryCounters;
import java.util.*;
import java.util.function.Supplier;
import java.util.stream.Collectors;
/**
@@ -146,6 +144,12 @@ public class GraphRepository {
// (entity -> table); native SQL matches a literal DB_TABLE name directly, no dependency.
statements.add(new EnrichmentStep("resolve-java-query-jpql", CypherQueries.RESOLVE_JAVA_QUERY_JPQL));
statements.add(new EnrichmentStep("resolve-java-query-native-sql", CypherQueries.RESOLVE_JAVA_QUERY_NATIVE_SQL));
// Item 140: a project with no DB_TABLE cannot resolve a single Java DB_ACCESS candidate, so
// everything the over-approximating parse-time heuristic emitted there is a false positive.
// Reap it (Java only — a Natural READ/FIND is a real access even with an unresolved view).
// Must follow all three resolvers above.
statements.add(new EnrichmentStep("reap-java-db-access-without-tables",
CypherQueries.REAP_JAVA_DB_ACCESS_WITHOUT_TABLES));
// Reap self-EXTENDS/IMPLEMENTS edges (a class cannot extend/implement itself) before any step
// traverses the inheritance graph — clears stale name-collision edges a non-wiping refresh leaves.
statements.add(new EnrichmentStep("delete-self-inheritance-edges", CypherQueries.DELETE_SELF_INHERITANCE_EDGES));
@@ -263,6 +267,9 @@ public class GraphRepository {
statements.add(new EnrichmentStep("resolve-view-alias-access-nodes",
CypherQueries.RESOLVE_VIEW_ALIAS_ACCESS_NODES));
statements.add(new EnrichmentStep("delete-orphaned-placeholder-tables", CypherQueries.DELETE_ORPHANED_PLACEHOLDER_TABLES));
// Item 124: the same for call-target placeholders left edgeless by the CALLS reap — otherwise a
// phantom target of a since-fixed parser bug keeps showing up as an unresolved module.
statements.add(new EnrichmentStep("delete-orphaned-placeholder-modules", CypherQueries.DELETE_ORPHANED_PLACEHOLDER_MODULES));
// Item 40: flag surviving placeholders as (un)resolved by whether a real definition now exists.
// Runs last so it sees the fully redirected/reaped graph. Project-wide, cheap, idempotent.
statements.add(new EnrichmentStep("stamp-unresolved-placeholders", CypherQueries.STAMP_UNRESOLVED_PLACEHOLDERS));
@@ -932,6 +939,32 @@ public class GraphRepository {
GraphRepository::toPayloadField));
}
/**
* Item 141: the comment blocks of a module, ordered by line. {@code kind} filters to one comment
* kind; with none, {@code **SAG} generator directives are excluded (see
* {@link CypherQueries#MODULE_COMMENTS}).
*/
public Uni<List<CommentBlock>> moduleComments(String project, String moduleName, String sourceFile,
@Nullable String kind, int limit, int offset) {
Map<String, @Nullable Object> params = new HashMap<>(moduleParams(project, moduleName, sourceFile));
params.put("kind", kind);
params.put("limit", limit);
params.put("offset", offset);
return read(CypherQueries.MODULE_COMMENTS, params, GraphRepository::toCommentBlock);
}
private static CommentBlock toCommentBlock(Record record) {
return new CommentBlock(
record.get("text").asString(""),
record.get("kind").asString(""),
record.get("sourceFile").asString(""),
record.get("startLine").asInt(),
record.get("endLine").asInt(),
record.get("target").isNull() ? null : record.get("target").asString(),
record.get("targetType").isNull() ? null : record.get("targetType").asString(),
Boolean.parseBoolean(record.get("truncated").asString("false")));
}
private static PayloadField toPayloadField(Record record) {
return new PayloadField(
record.get("tag").asString(),
@@ -1078,19 +1111,74 @@ public class GraphRepository {
}
/**
* Paginated variant of {@link #searchIdentifier(String, String, String, String, String, String)} for the endpoint.
* Item 126: the last whole-root ingest's facts, or {@code null} when the project shell carries no
* {@code ingestedAt} — i.e. it was last ingested before this was recorded. Null rather than a
* zero-filled record: "never measured" and "measured as none" are different answers, and
* conflating them is the failure mode the item is about.
*/
private static @Nullable ProjectIngestInfo toProjectIngestInfo(Record record) {
@Nullable String ingestedAt = nullableString(record, "ingestedAt");
boolean incomplete = record.containsKey("ingestIncomplete") && !record.get("ingestIncomplete").isNull()
&& record.get("ingestIncomplete").asBoolean();
// Item 129: a project whose very first whole-root pass is still running (or died) has no
// ingestedAt yet, but the in-flight marker is the most important thing to report about it —
// so it is not folded into the "never recorded" null case below.
if (ingestedAt == null && !incomplete) {
return null;
}
if (ingestedAt == null) {
return new ProjectIngestInfo("", nullableString(record, "ingestMode") != null
? record.get("ingestMode").asString() : "", 0, 0, 0, List.of(), false, 0L, "",
true, nullableString(record, "ingestStartedAt"));
}
List<String> failures = record.containsKey("ingestFailures") && !record.get("ingestFailures").isNull()
? record.get("ingestFailures").asList(value -> value.asString())
: List.of();
return new ProjectIngestInfo(ingestedAt,
nullableString(record, "ingestMode") != null ? record.get("ingestMode").asString() : "",
intOrZero(record, "ingestFilesExamined"), intOrZero(record, "ingestFilesPersisted"),
intOrZero(record, "ingestFilesFailed"), failures,
record.containsKey("ingestFailuresTruncated") && !record.get("ingestFailuresTruncated").isNull()
&& record.get("ingestFailuresTruncated").asBoolean(),
record.containsKey("ingestDurationSeconds") && !record.get("ingestDurationSeconds").isNull()
? record.get("ingestDurationSeconds").asLong() : 0L,
nullableString(record, "ingestServerVersion") != null
? record.get("ingestServerVersion").asString() : "",
incomplete, nullableString(record, "ingestStartedAt"));
}
/**
* Paginated variant of {@link #searchIdentifier(String, String, String, String, String, String, boolean, int)}
* for the endpoint.
*
* @param contains item 125: match names that <em>contain</em> {@code identifierName}
* (case-insensitive) instead of equalling it
*/
public Uni<List<IdentifierMatch>> searchIdentifier(String project, @Nullable String identifierName,
@Nullable String type, @Nullable String sourceFile,
@Nullable String module, @Nullable String priorityModule,
int limit, int offset) {
return searchIdentifier(project, identifierName, type, sourceFile, module, priorityModule)
boolean contains, int limit, int offset) {
// Item 125: fetch at most the rows this page can consume, so an unindexed CONTAINS over a
// large project cannot materialise its whole match set here. paginate() then slices exactly
// as before — including its "limit <= 0 means unlimited" rule, which is why the cap is
// Integer.MAX_VALUE in that case rather than 0.
int scanCap = limit > 0 ? (int) Math.min((long) Math.max(offset, 0) + limit, Integer.MAX_VALUE)
: Integer.MAX_VALUE;
return searchIdentifier(project, identifierName, type, sourceFile, module, priorityModule, contains, scanCap)
.map(list -> paginate(list, limit, offset));
}
public Uni<List<IdentifierMatch>> searchIdentifier(String project, @Nullable String identifierName,
@Nullable String type, @Nullable String sourceFile,
@Nullable String module, @Nullable String priorityModule) {
return searchIdentifier(project, identifierName, type, sourceFile, module, priorityModule,
false, Integer.MAX_VALUE);
}
public Uni<List<IdentifierMatch>> searchIdentifier(String project, @Nullable String identifierName,
@Nullable String type, @Nullable String sourceFile,
@Nullable String module, @Nullable String priorityModule,
boolean contains, int scanCap) {
Map<String, @Nullable Object> params = new java.util.HashMap<>();
params.put("project", project);
// Sigil-insensitive: strip a leading Natural sigil (# user, & AIV, + GDA) so a search for
@@ -1103,26 +1191,40 @@ public class GraphRepository {
// Pin this module's matches to the front so they survive limit/paginate truncation when a
// name recurs across many modules (the caller's local declaration would otherwise be lost).
params.put("priorityModule", priorityModule);
// Item 125: substring vs exact name match, and the row cap the query's LIMIT reads.
params.put("contains", contains);
params.put("scanCap", scanCap);
return read(CypherQueries.SEARCH_IDENTIFIER, params, record -> new IdentifierMatch(
record.get("id").asString(),
NodeType.valueOf(record.get("type").asString()),
record.get("name").asString(),
record.get("sourceFile").asString(),
record.get("startLine").asInt(),
record.get("endLine").asInt(),
// Item 136: asInt() on a NULL throws Uncoercible and escaped as an unstructured 500.
// Item 114's duplicate markers were created without lines; that is fixed at the write
// side too, but a row mapper must never be the thing that faults an endpoint.
intOrZero(record, "startLine"),
intOrZero(record, "endLine"),
record.get("dataType").isNull() ? null : record.get("dataType").asString(),
record.get("value").isNull() ? null : record.get("value").asString(),
record.get("scope").isNull() ? null : record.get("scope").asString(),
record.get("unresolved").isNull() ? null : record.get("unresolved").asBoolean()));
record.get("unresolved").isNull() ? null : record.get("unresolved").asBoolean(),
record.get("simpleName").isNull() ? null : record.get("simpleName").asString(),
record.get("moduleKind").isNull() ? null : record.get("moduleKind").asString()));
}
/**
* @param substring when {@code true}, uses {@link CypherQueries#SEARCH_BY_VALUE_CONTAINS}
* (case-insensitive substring, item 29) instead of the default exact match.
* @param substring when {@code true}, uses {@link CypherQueries#SEARCH_BY_VALUE_CONTAINS}
* (case-insensitive substring, item 29) instead of the default exact match.
* @param includeComments item 141: when {@code true}, comment blocks are searched too and their
* hits come back as {@code kind = "COMMENT"}. Default {@code false} — the
* text of a comment is not the same evidence as a literal in code, and
* folding it in silently would move every existing completeness count.
*/
public Uni<List<ValueMatch>> searchByValue(String project, String value, boolean substring) {
public Uni<List<ValueMatch>> searchByValue(String project, String value, boolean substring,
boolean includeComments) {
String query = substring ? CypherQueries.SEARCH_BY_VALUE_CONTAINS : CypherQueries.SEARCH_BY_VALUE;
return read(query, Map.of("project", project, "value", value), GraphRepository::toValueMatch);
return read(query, Map.of("project", project, "value", value, "includeComments", includeComments),
GraphRepository::toValueMatch);
}
/**
@@ -1313,7 +1415,229 @@ public class GraphRepository {
: record.get("excludeDirs").asList(value -> value.asString());
return new ProjectInfo(record.get("name").asString(), description, root, excludeDirs,
nullableString(record, "language"), nullableString(record, "generatedDir"),
nullableString(record, "userExitDir"));
nullableString(record, "userExitDir"), toProjectIngestInfo(record));
}
/**
* Item 131: the identifier search as a {@link Page} — the rows plus how many there are in total.
*
* <p>The total costs a second query only when it can matter. A page that came back <b>shorter</b>
* than {@code limit} is provably the end of the result set, so the total is arithmetic
* ({@code offset + rows}); only a <b>full</b> page — the one case where rows may have been cut —
* pays for a {@code count(*)}. For {@code contains=true}, which is an unindexed label scan
* (1.2-2.3s on {@code upms}), that difference is the whole cost of the feature.
*
* <p>A total that divides evenly by {@code limit} makes the last full page report
* {@code truncated} and the next page come back empty: one wasted call, never a wrong answer.
*/
public Uni<Page<IdentifierMatch>> searchIdentifierPage(String project, @Nullable String identifierName,
@Nullable String type, @Nullable String sourceFile,
@Nullable String module, @Nullable String priorityModule,
boolean contains, int limit, int offset) {
return searchIdentifier(project, identifierName, type, sourceFile, module, priorityModule,
contains, limit, offset)
.flatMap(rows -> withTotal(rows, limit, offset,
() -> countIdentifier(project, identifierName, type, sourceFile, module, contains)));
}
private Uni<Long> countIdentifier(String project, @Nullable String identifierName, @Nullable String type,
@Nullable String sourceFile, @Nullable String module, boolean contains) {
Map<String, @Nullable Object> params = new HashMap<>();
params.put("project", project);
params.put("name", stripLeadingSigil(identifierName));
params.put("type", type);
params.put("sourceFile", sourceFile);
params.put("module", module);
params.put("contains", contains);
return count(CypherQueries.SEARCH_IDENTIFIER_COUNT, params);
}
/**
* Item 131: the value search as a {@link Page}. See {@link #searchIdentifierPage} for why the
* count is conditional.
*/
public Uni<Page<ValueMatch>> searchByValuePage(String project, String value, boolean substring,
boolean includeComments, int limit, int offset) {
return searchByValue(project, value, substring, includeComments, limit, offset)
.flatMap(rows -> withTotal(rows, limit, offset,
() -> count(CypherQueries.searchByValueCount(substring),
Map.of("project", project, "value", value, "includeComments", includeComments))));
}
/**
* Item 131: the annotation search as a {@link Page} — the endpoint this item was written about.
*/
public Uni<Page<AnnotationMatch>> searchAnnotationPage(String project, String name, @Nullable String type,
int limit, int offset) {
return searchAnnotation(project, name, type, limit, offset)
.flatMap(rows -> withTotal(rows, limit, offset, () -> {
Map<String, @Nullable Object> params = new HashMap<>();
params.put("project", project);
params.put("name", name);
params.put("type", type);
return count(CypherQueries.SEARCH_ANNOTATION_COUNT, params);
}));
}
/**
* Wraps {@code rows} in a {@link Page}, running {@code counter} only when the page is full (and a
* limit was given at all — {@code limit <= 0} means the caller asked for everything, so what came
* back is by definition the whole set).
*/
private <T> Uni<Page<T>> withTotal(List<T> rows, int limit, int offset, Supplier<Uni<Long>> counter) {
if (limit <= 0 || rows.size() < limit) {
long total = (long) Math.max(offset, 0) + rows.size();
return Uni.createFrom().item(new Page<>(rows, total, false));
}
return counter.get().map(total -> new Page<>(rows, total,
total > (long) Math.max(offset, 0) + rows.size()));
}
/**
* Runs a query whose single row is a {@code total} column.
*/
private Uni<Long> count(String query, Map<String, @Nullable Object> params) {
return read(query, params, record -> record.get("total").asLong())
.map(totals -> totals.isEmpty() ? 0L : totals.get(0));
}
/**
* Item 130: the project's REST endpoints (path + verb + handler), optionally narrowed to one
* declaring class.
*/
public Uni<List<RestEndpoint>> restEndpoints(String project, @Nullable String module, int limit, int offset) {
int scanCap = limit > 0 ? (int) Math.min((long) Math.max(offset, 0) + limit, Integer.MAX_VALUE)
: Integer.MAX_VALUE;
Map<String, @Nullable Object> params = new HashMap<>();
params.put("project", project);
params.put("module", module);
params.put("scanCap", scanCap);
return read(CypherQueries.REST_ENDPOINTS, params, record -> new RestEndpoint(
record.get("httpMethod").asString(),
record.get("path").asString(),
record.get("module").asString(),
record.get("moduleSimpleName").isNull() ? null : record.get("moduleSimpleName").asString(),
record.get("handler").asString(),
record.get("sourceFile").asString(),
intOrZero(record, "startLine"),
!record.get("outbound").isNull() && record.get("outbound").asBoolean()))
.map(list -> paginate(list, limit, offset));
}
/**
* Item 135: the REST surface as a {@link Page}. Item 131 shipped without this endpoint, so the
* answer carried no total at all. With the default (uncapped) limit no count query runs — see
* {@link #withTotal}.
*/
public Uni<Page<RestEndpoint>> restEndpointsPage(String project, @Nullable String module,
int limit, int offset) {
return restEndpoints(project, module, limit, offset)
.flatMap(rows -> withTotal(rows, limit, offset, () -> {
Map<String, @Nullable Object> params = new HashMap<>();
params.put("project", project);
params.put("module", module);
return count(CypherQueries.REST_ENDPOINTS_COUNT, params);
}));
}
/**
* Item 128: every reference site of {@code name} — imports, declared type positions, annotation
* usages, calls, inheritance and wiring — with a {@code kind} discriminator per row.
*
* @param kind optional filter on that discriminator ({@code IMPORT}, {@code TYPE}, {@code CALL}, …)
*/
public Uni<List<ReferenceSite>> searchReferences(String project, String name, @Nullable String kind,
int limit, int offset) {
int scanCap = limit > 0 ? (int) Math.min((long) Math.max(offset, 0) + limit, Integer.MAX_VALUE)
: Integer.MAX_VALUE;
Map<String, @Nullable Object> params = new HashMap<>();
params.put("project", project);
params.put("name", name);
params.put("kind", kind);
params.put("scanCap", scanCap);
return read(CypherQueries.SEARCH_REFERENCES, params, record -> new ReferenceSite(
record.get("sourceFile").asString(),
record.get("lineNo").asInt(),
record.get("kind").asString(),
record.get("inModule").isNull() ? null : record.get("inModule").asString(),
record.get("target").asString()))
.map(list -> paginate(list, limit, offset));
}
/**
* Item 135: the reference search as a {@link Page}. This is the one that actually lost rows — it
* capped at the default 50 with no signal whatsoever, which is precisely the failure item 131 was
* written to remove.
*/
public Uni<Page<ReferenceSite>> searchReferencesPage(String project, String name, @Nullable String kind,
int limit, int offset) {
return searchReferences(project, name, kind, limit, offset)
.flatMap(rows -> withTotal(rows, limit, offset, () -> {
Map<String, @Nullable Object> params = new HashMap<>();
params.put("project", project);
params.put("name", name);
params.put("kind", kind);
return count(CypherQueries.SEARCH_REFERENCES_COUNT, params);
}));
}
/**
* Item 129: {@code sourceFile -> sourceHash} for the whole project, the input to a
* {@code changedOnly} refresh's skip decision. A file missing from this map has no stored hash and
* counts as changed.
*/
public Uni<Map<String, String>> sourceHashes(String project) {
return read(CypherQueries.SOURCE_HASHES, Map.of("project", project),
record -> Map.entry(record.get("sourceFile").asString(), record.get("hash").asString()))
.map(entries -> entries.stream().collect(java.util.stream.Collectors.toMap(
Map.Entry::getKey, Map.Entry::getValue, (a, b) -> a)));
}
private static int intOrZero(Record record, String key) {
return record.containsKey(key) && !record.get(key).isNull() ? record.get(key).asInt() : 0;
}
/**
* Item 126: records what a whole-root ingest just did on the {@code (:Project)} shell. The failure
* list is capped at {@link ProjectIngestInfo#MAX_FAILURES} with an explicit truncation flag — the
* count stays exact, so a short list never reads as the whole story.
*/
/**
* Item 129: marks a whole-root pass as in flight. Cleared by {@link #recordProjectIngest}, so an
* interrupted run leaves the marker set and stops looking like a clean graph.
*/
public Uni<Void> markProjectIngestStarted(String project, String mode, String startedAt) {
Map<String, @Nullable Object> params = new HashMap<>();
params.put("name", project);
params.put("mode", mode);
params.put("startedAt", startedAt);
return Uni.createFrom().item(() -> {
try (Session session = driver.session()) {
session.executeWriteWithoutResult(tx -> tx.run(CypherQueries.MARK_PROJECT_INGEST_STARTED, params));
}
return project;
}).replaceWithVoid();
}
public Uni<Void> recordProjectIngest(String project, ProjectIngestInfo ingest) {
Map<String, @Nullable Object> params = new HashMap<>();
params.put("name", project);
params.put("ingestedAt", ingest.ingestedAt());
params.put("mode", ingest.mode());
params.put("filesExamined", ingest.filesExamined());
params.put("filesPersisted", ingest.filesPersisted());
params.put("filesFailed", ingest.filesFailed());
params.put("failures", ingest.failures());
params.put("failuresTruncated", ingest.failuresTruncated());
params.put("durationSeconds", ingest.durationSeconds());
params.put("serverVersion", ingest.serverVersion());
return Uni.createFrom().item(() -> {
try (Session session = driver.session()) {
session.executeWriteWithoutResult(tx -> tx.run(CypherQueries.RECORD_PROJECT_INGEST, params));
}
return project;
}).replaceWithVoid();
}
private static @Nullable String nullableString(Record record, String key) {
@@ -1352,21 +1676,56 @@ public class GraphRepository {
*/
private static String mergeKey(AstNode node, boolean positional, String project, String ownerFile) {
String base = node.type() + "" + node.name() + "" + node.sourceFile() + "" + project
+ "" + placeholderOwner(node, ownerFile);
+ "" + nodeOwner(node, ownerFile);
return positional ? base + "" + node.startLine() : base;
}
/**
* Item 76: field-level placeholders (VARIABLE/CONSTANT with sourceFile="") get a per-module
* identity via the referencing module file (ownerFile), so a field name referenced by hundreds
* of modules is no longer collapsed onto one shared node (which made finalize step 19 ~55 min on
* upms). Real nodes and MODULE/DB_TABLE/DATA_STRUCTURE placeholders keep ownerModule="" and merge
* exactly as before.
* The node's owning module file, or {@code ""} for a node that is legitimately shared. Two
* populations get a per-module identity:
*
* <ul>
* <li><b>Item 76</b> — field-level placeholders (VARIABLE/CONSTANT with {@code sourceFile=""}),
* so a field name referenced by hundreds of modules is not collapsed onto one shared node
* (which made finalize step 19 ~55 min on upms).</li>
* <li><b>Items 75-B/75-C</b> — copycode-resident nodes, identified <em>per expansion site</em>
* as {@code <hostFile>#<includePath>}. {@code CopycodePreprocessor} splices a {@code .cpy}
* body into the including module before parsing and remaps the resulting nodes back onto the
* copycode file, so a node whose {@code sourceFile} differs from the parse's own module file
* came from a copycode — no extension test needed. Sharing those nodes blocked the stale-edge
* reaps of items 124/86/106 (they could not delete an edge other includers had contributed to
* the same shared node) and made {@code CONTAINS} cyclic, because a copycode that opens a
* block it does not close accumulated the nesting context of every site onto one node.</li>
* </ul>
*
* <p>{@code MODULE} and {@code DB_TABLE} are excluded deliberately. A module <em>can</em> be
* declared inside a copycode ({@code ZDTSTBP6} in {@code ZDTSTBC6.cpy}), and every module lookup
* binds {@code (project, name, sourceFile)} but never {@code ownerModule} — an owned MODULE node
* would be invisible to those queries and re-created beside itself on the next merge. Tables are
* shared by design.
*/
private static String placeholderOwner(AstNode node, String ownerFile) {
return node.sourceFile().isEmpty()
&& (node.type() == NodeType.VARIABLE || node.type() == NodeType.CONSTANT)
? ownerFile : "";
private static String nodeOwner(AstNode node, String ownerFile) {
if (node.type() == NodeType.MODULE || node.type() == NodeType.DB_TABLE) {
return "";
}
if (node.sourceFile().isEmpty()) {
return node.type() == NodeType.VARIABLE || node.type() == NodeType.CONSTANT ? ownerFile : "";
}
if (node.sourceFile().equals(ownerFile)) {
return "";
}
// Item 75-C: the including *module* is not fine-grained enough. One module can include the same
// copycode at many differently-nested sites (JX0031N0.nat includes YFRAMBC0 16 times), and a
// copycode can include another repeatedly (VPARTC02.cpy includes L4NLOGIC 136 times) — all of
// which share the host's INCLUDE line, so `includedAt` cannot tell them apart either. The
// expansion site is identified by the whole include chain, and only per-site identity keeps a
// copycode that opens a block it does not close from collecting every site's nesting context
// (the CONTAINS cycles of item 75).
@Nullable Map<String, String> props = node.properties();
@Nullable String site = props == null ? null : props.get(CopycodeProperties.INCLUDE_PATH);
// No chain recorded: fall back to per-module identity, i.e. the pre-item-75-C behaviour. That
// re-collapses the sites (the known, survivable defect) rather than inventing an identity.
return site == null || site.isEmpty() ? ownerFile : ownerFile + "#" + site;
}
private static Map<String, @Nullable Object> edgeParams(AstEdge edge) {
@@ -1724,7 +2083,7 @@ public class GraphRepository {
params.put("name", node.name());
params.put("sourceFile", node.sourceFile());
params.put("project", project);
params.put("ownerModule", placeholderOwner(node, ownerFile));
params.put("ownerModule", nodeOwner(node, ownerFile));
params.put("language", node.language());
params.put("startLine", node.startLine());
params.put("endLine", node.endLine());
@@ -2088,7 +2447,11 @@ public class GraphRepository {
record.get("assignedField").asString(),
record.get("assignedValue").isNull() ? ""
: record.get("assignedValue").asString().replace("'", "").trim(),
record.get("lineNo").asInt()));
record.get("lineNo").asInt(),
record.get("sourceFile").asString(""),
record.get("viaCopycode").isNull() ? null : record.get("viaCopycode").asString(),
intOrNull(record.get("includedAt")),
toIncludePath(record.get("includePath"))));
}
private static SqlStatement toSqlStatement(Record record) {
@@ -2163,8 +2526,10 @@ public class GraphRepository {
// exactly why a plain long suffices where a UUID used to be needed.
long[] nextNid = {0L};
int[] skippedEdges = {0};
// Real source files touched by this parse, for the item-58 stale-node sweep below.
Set<String> freshFiles = new HashSet<>();
// Real (sourceFile, ownerModule) pairs touched by this parse, for the item-58 stale-node sweep
// and the item-86/106/124 edge reaps below. Item 75-B: the pair, not the file alone — several
// includers' copycode nodes share one sourceFile and only the re-parsed owner's may be swept.
Set<Map<String, Object>> freshFiles = new LinkedHashSet<>();
// Item 76: owning modules whose placeholders this parse refreshed, for the placeholder sweep.
Set<String> freshPlaceholderOwners = new HashSet<>();
// Different files emit the same placeholder (e.g. a CALLNAT/USING target, sourceFile="")
@@ -2189,12 +2554,11 @@ public class GraphRepository {
Map<String, @Nullable Object> params = toParams(node, project, ownerFile);
params.put("nid", nid);
(positional ? positionalNodes : namedNodes).add(params);
String owner = nodeOwner(node, ownerFile);
if (!node.sourceFile().isEmpty()) {
freshFiles.add(node.sourceFile());
}
String phOwner = placeholderOwner(node, ownerFile);
if (!phOwner.isEmpty()) {
freshPlaceholderOwners.add(phOwner);
freshFiles.add(Map.of("f", node.sourceFile(), "o", owner));
} else if (!owner.isEmpty()) {
freshPlaceholderOwners.add(owner);
}
} else {
nidByParserId.put(parserId, canonical);
@@ -2230,11 +2594,20 @@ public class GraphRepository {
// current edges; unchanged ones round-trip identically.
if (!freshFiles.isEmpty()) {
tx.run(CypherQueries.DELETE_STALE_NATURAL_TABLE_ACCESS_EDGES,
Map.of("project", project, "sourceFiles", List.copyOf(freshFiles)));
Map.of("project", project, "files", List.copyOf(freshFiles)));
// Item 106: same for USING edges, which resolve onto REAL data-area nodes and so are not
// covered by the placeholder-only reap above.
tx.run(CypherQueries.DELETE_STALE_NATURAL_USING_EDGES,
Map.of("project", project, "sourceFiles", List.copyOf(freshFiles)));
Map.of("project", project, "files", List.copyOf(freshFiles)));
// Item 124: same for CALLS. A parser fix that changes a call's target leaves both endpoints
// alive (the calling subroutine is unchanged; the old target is a never-swept placeholder),
// so the corrected call was only ever added beside the wrong one. Deep re-ingest only —
// two dynamic-call resolvers are skipped in a coarse finalize and could not rebuild what
// this deletes.
if (reconcile) {
tx.run(CypherQueries.DELETE_STALE_NATURAL_CALL_EDGES,
Map.of("project", project, "files", List.copyOf(freshFiles)));
}
}
for (Map.Entry<EdgeType, List<Map<String, @Nullable Object>>> entry : edgesByType.entrySet()) {
tx.run(CypherQueries.mergeEdgesBatch(entry.getKey()),
@@ -2245,7 +2618,7 @@ public class GraphRepository {
// statements) so a refresh purges stale nodes instead of leaving them to shadow the new ones.
if (reconcile && !freshFiles.isEmpty()) {
tx.run(CypherQueries.DELETE_STALE_FILE_NODES, Map.of("project", project,
"sourceFiles", List.copyOf(freshFiles), "ingestGen", ingestGen));
"files", List.copyOf(freshFiles), "ingestGen", ingestGen));
}
// Item 76: sweep per-module field placeholders (sourceFile="") that a re-ingested owner file
// no longer produces. The file sweep above skips sourceFile="" nodes; without this an
@@ -2265,7 +2638,7 @@ public class GraphRepository {
// follows and re-links them.
if (reconcile && !freshFiles.isEmpty()) {
tx.run(CypherQueries.DELETE_STALE_RESOLVED_FIELD_EDGES,
Map.of("project", project, "sourceFiles", List.copyOf(freshFiles)));
Map.of("project", project, "files", List.copyOf(freshFiles)));
}
}
@@ -2275,10 +2648,11 @@ public class GraphRepository {
}
/**
* Paginated variant of {@link #searchByValue(String, String, boolean)} for the endpoint.
* Paginated variant of {@link #searchByValue(String, String, boolean, boolean)} for the endpoint.
*/
public Uni<List<ValueMatch>> searchByValue(String project, String value, boolean substring, int limit, int offset) {
return searchByValue(project, value, substring).map(list -> paginate(list, limit, offset));
public Uni<List<ValueMatch>> searchByValue(String project, String value, boolean substring,
boolean includeComments, int limit, int offset) {
return searchByValue(project, value, substring, includeComments).map(list -> paginate(list, limit, offset));
}
/**

View File

@@ -13,8 +13,16 @@ import org.jspecify.annotations.Nullable;
* @param unresolved item 40: {@code true} when this match is an unresolved reference placeholder
* ({@code sourceFile} is blank and no real definition of the name is ingested);
* {@code false} for a resolved placeholder, {@code null} for a normal (real) node
* @param simpleName item 125: a Java type declaration's short form, where {@code name} is its
* fully-qualified identity (item 117). {@code null} for every non-{@code MODULE} node and
* for Natural modules, whose {@code name} already is the short form
* @param moduleKind item 125: the declaration kind of a {@code MODULE} match — {@code CLASS},
* {@code INTERFACE}, {@code ENUM}, {@code RECORD} for Java, {@code PROGRAM},
* {@code SUBPROGRAM} for Natural — so "is this name a type or a method?" is answered by
* the search itself rather than by a follow-up call. {@code null} for non-{@code MODULE} nodes
*/
public record IdentifierMatch(String id, NodeType type, String name, String sourceFile, int startLine, int endLine,
@Nullable String dataType, @Nullable String value, @Nullable String scope,
@Nullable Boolean unresolved) {
@Nullable Boolean unresolved, @Nullable String simpleName,
@Nullable String moduleKind) {
}

View File

@@ -0,0 +1,26 @@
package com.agenticcode.neo4jstore.graph;
import java.util.List;
/**
* Item 131: a page of results together with how many there are in total, so a caller can tell a
* complete answer from a truncated one.
*
* <p>The three search endpoints returned a bare JSON array capped at {@code limit=50} with no total
* and no flag. That is not a cosmetic gap: a real UPMS→PUR audit read 50 of 95 {@code @Immutable}
* rows as the whole set and recorded that 17 entities had lost the annotation, when zero had. The
* finding cost a day and was purely an artefact of the cut.
*
* @param rows the requested page
* @param total how many rows match in total, ignoring {@code limit}/{@code offset}
* @param truncated whether {@code total} exceeds what this page shows — i.e. rows were left out
*/
public record Page<T>(List<T> rows, long total, boolean truncated) {
/**
* @return a page that is known to be complete, for callers that fetched everything
*/
public static <T> Page<T> complete(List<T> rows) {
return new Page<>(rows, rows.size(), false);
}
}

View File

@@ -20,12 +20,23 @@ import java.util.List;
* They are set together or not at all; {@code null} when the project has no user-exit split.
*/
public record ProjectInfo(String name, @Nullable String description, String root, List<String> excludeDirs,
@Nullable String language, @Nullable String generatedDir, @Nullable String userExitDir) {
@Nullable String language, @Nullable String generatedDir, @Nullable String userExitDir,
@Nullable ProjectIngestInfo ingest) {
/**
* Legacy convenience constructor for projects without a language / user-exit split.
*/
public ProjectInfo(String name, @Nullable String description, String root, List<String> excludeDirs) {
this(name, description, root, excludeDirs, null, null, null);
this(name, description, root, excludeDirs, null, null, null, null);
}
/**
* Config-only constructor: everything above is what the user configured, {@code ingest} is what the
* server observed. Kept separate on purpose — this record is also the <em>input</em> to an ingest
* (see {@code ProjectRootResolver}), and a run must never be fed the state it is about to replace.
*/
public ProjectInfo(String name, @Nullable String description, String root, List<String> excludeDirs,
@Nullable String language, @Nullable String generatedDir, @Nullable String userExitDir) {
this(name, description, root, excludeDirs, language, generatedDir, userExitDir, null);
}
}

View File

@@ -0,0 +1,54 @@
package com.agenticcode.neo4jstore.graph;
import org.jspecify.annotations.Nullable;
import java.util.List;
/**
* Item 126: what the last <b>whole-root</b> ingest of a project did, recorded on the {@code (:Project)}
* shell so a caller can tell an answer's age and completeness from the API instead of crawling the
* file system.
*
* <p>Without this, a "not found" was never evidence: an empty result could mean the name does not
* exist, or that the project was never fully ingested, and nothing in any response told the two
* apart. {@code null} on {@link ProjectInfo#ingest()} means <b>never recorded</b> (a project ingested
* before this was introduced) — deliberately not zeros, which would state a fact that was never
* measured.
*
* <p><b>Only whole-root passes write this</b> — create-time Tier-1 scan, {@code refresh},
* {@code refresh?deep=true}. A by-name deep ingest, a {@code refresh/{name}} or the fan-out warm
* ingest real files but see a fraction of the tree, so letting them stamp {@code ingestedAt} would
* report the project as freshly walked when one module was deepened — the "looks complete but isn't"
* answer this item exists to remove.
*
* <p><b>This is not a freshness guarantee.</b> It says when the walk ran, not that the graph still
* matches disk: files edited afterwards are stale while {@code ingestedAt} still looks recent.
* Content-hash-based staleness is item 129.
*
* @param ingestedAt UTC ISO-8601 instant the ingest finished
* @param mode {@code "tier1"} (coarse create-time scan), {@code "call_graph"}
* ({@code refresh}) or {@code "full"} ({@code refresh?deep=true})
* @param filesExamined source files the walk found and attempted
* @param filesPersisted files actually parsed and written (examined minus failed and minus
* duplicate-skipped identities)
* @param filesFailed how many files failed to parse — the <b>full</b> count, even when
* {@code failures} below is truncated
* @param failures the failing paths (relative to the project root), capped
* @param failuresTruncated {@code true} when {@code failures} lists fewer paths than
* {@code filesFailed}, so a short list is never mistaken for the whole story
* @param durationSeconds how long the pass took
* @param serverVersion the {@code agenticcode.version} of the server that wrote the graph — the
* server release, not a per-parser grammar version
*/
public record ProjectIngestInfo(String ingestedAt, String mode, int filesExamined, int filesPersisted,
int filesFailed, List<String> failures, boolean failuresTruncated,
long durationSeconds, String serverVersion, boolean incomplete,
@Nullable String startedAt) {
/**
* Cap on the persisted {@code failures} list: a misconfigured root can fail thousands of files and
* a graph property is not a log. The count stays exact and {@link #failuresTruncated} says the list
* was cut, so the truncation is never silent.
*/
public static final int MAX_FAILURES = 200;
}

View File

@@ -0,0 +1,21 @@
package com.agenticcode.neo4jstore.graph;
import org.jspecify.annotations.Nullable;
/**
* Item 128: one place a name is mentioned, returned by {@link CypherQueries#SEARCH_REFERENCES}.
*
* @param sourceFile the referencing file, relative to the project root
* @param lineNo the line the reference sits on
* @param kind how it is referenced — {@code CALL}, {@code IMPORT}, {@code TYPE} (a declared
* field/parameter/return type), {@code ANNOTATION}, {@code EXTENDS},
* {@code IMPLEMENTS}, {@code INJECTS}, {@code CLASS_LITERAL} ({@code X.class}), or
* {@code INCLUDE} (Natural copycode)
* @param inModule the module the reference sits in, or {@code null} when the referencing node has
* no containing module
* @param target the referenced module's identity (the FQN for Java), so a short-name query shows
* what it actually resolved to
*/
public record ReferenceSite(String sourceFile, int lineNo, String kind, @Nullable String inModule,
String target) {
}

View File

@@ -0,0 +1,20 @@
package com.agenticcode.neo4jstore.graph;
import org.jspecify.annotations.Nullable;
/**
* Item 130: one REST endpoint of a project — the composed path, its HTTP verb, and the handler that
* serves it, so "which code runs for {@code POST /partners}" is one call rather than a manual join of
* two annotation searches.
*
* @param httpMethod the verb ({@code GET}, {@code POST}, …)
* @param path class-level {@code @Path} + method-level {@code @Path}, joined
* @param module the declaring class's identity (FQN)
* @param moduleSimpleName its short name, or {@code null} for a Natural module
* @param handler the handler method
* @param sourceFile the file it lives in, relative to the project root
* @param startLine the handler's first line
*/
public record RestEndpoint(String httpMethod, String path, String module, @Nullable String moduleSimpleName,
String handler, String sourceFile, int startLine, boolean outbound) {
}

View File

@@ -0,0 +1,35 @@
package com.agenticcode.parsercore.ast.model;
/**
* Property keys a Natural copycode expansion puts on the nodes and edges it produces (roadmap items
* 46a/66/104). They live here, in the shared model, because they are written by the parser
* (`ac-parser-natural`) and read by the store (`ac-neo4j-store`) — item 75-C made
* {@link #INCLUDE_PATH} part of a node's <em>merge identity</em>, so a typo on either side would
* silently change which nodes are one node.
*/
public final class CopycodeProperties {
/**
* Name of the copycode member a node/edge was spliced from ({@code null}/absent for host lines).
*/
public static final String VIA_COPYCODE = "viaCopycode";
/**
* 1-based line of the {@code INCLUDE} <em>in the host module</em> that started the expansion.
* Item 104: always the host's line, at every nesting level — which is why it does <em>not</em>
* identify an expansion site on its own (one host {@code INCLUDE} can expand a member many
* times through nesting: {@code VPARTC02.cpy} includes {@code L4NLOGIC} 136 times). Use
* {@link #INCLUDE_PATH} for identity.
*/
public static final String INCLUDED_AT = "includedAt";
/**
* The whole {@code INCLUDE} chain as {@code file:line>file:line>…}, host first — the unique
* identity of one expansion site, and therefore part of the node merge key for
* copycode-resident nodes (item 75-C).
*/
public static final String INCLUDE_PATH = "includePath";
private CopycodeProperties() {
}
}

View File

@@ -40,5 +40,28 @@ public enum EdgeType {
* enrichment run, because the stale-node sweep only deletes nodes and would let an edge outlive the
* call it came from.
*/
CALLS_MODULE
CALLS_MODULE,
/**
* Item 128: a plain <b>mention</b> of a type — an {@code import}, a declared field/parameter/return
* type, or an annotation usage — carrying a {@code refKind} property saying which.
*
* <p>Deliberately <em>not</em> folded into {@link #REFERENCES}. The call-graph traversals
* ({@code callers}, {@code callees}, {@code call-tree}, {@code ego-graph}) follow {@code REFERENCES}
* as wiring, so reusing it made an {@code import} show up as a caller — a regression caught by
* {@code javaCrossClassCallGraphSpansFiles}, which saw {@code edgeKind: REFERENCES} where a method
* call belonged. A separate type keeps "mentions this" out of "calls this" by construction, rather
* than by remembering to filter it in every query that already exists.
*/
MENTIONS,
/**
* Item 141: a {@link NodeType#COMMENT} block → the declaration it documents (the nearest
* following {@code FUNCTION}/{@code FIELD}/{@code VARIABLE}, else the enclosing {@code FUNCTION},
* else the {@code MODULE}). Exactly one per comment block, so a comment is never orphaned.
*
* <p>Deliberately <em>not</em> {@link #CONTAINS}. {@code CONTAINS} is walked by the functions
* listing, the {@code SEARCH_BY_VALUE} assignment arm ({@code CONTAINS*0..1}), the ego-graph and
* the stale sweep; hanging comments off it would leak comment rows into queries that never asked
* for them. Same reasoning as {@link #MENTIONS} vs {@link #REFERENCES} in item 128.
*/
DOCUMENTS
}

View File

@@ -0,0 +1,67 @@
package com.agenticcode.parsercore.ast.model;
import java.util.Locale;
import java.util.Set;
/**
* JDK/stdlib and common-framework type names (item J6), in {@code ac-parser-core} so that both the
* parsers and the server-side ingest apply the <b>same</b> exclusion.
*
* <p>Two callers with one reason. A by-name ingest follows every referenced type, and JDK/framework
* types never resolve to a file in the project — without this they dominate the {@code unresolved}
* list and inflate the dependency fan-out. Item 128's reference index has the sharper version of the
* same problem: an edge per mention of {@code List} or {@code Logger} would mint a placeholder node
* that is created, persisted and swept again on <em>every</em> ingest.
*
* <p>Matched by uppercased simple name (a qualified name is reduced to its last segment first). A
* deliberate heuristic: a project class named like a JDK type is also skipped — acceptable, and
* vanishingly rare.
*/
public final class ExternalTypeNames {
private static final Set<String> NAMES = Set.of(
// java.lang
"OBJECT", "STRING", "CHARSEQUENCE", "INTEGER", "LONG", "DOUBLE", "FLOAT", "BOOLEAN", "BYTE",
"SHORT", "CHARACTER", "NUMBER", "STRINGBUILDER", "STRINGBUFFER", "THREAD", "RUNNABLE",
"EXCEPTION", "RUNTIMEEXCEPTION", "ILLEGALARGUMENTEXCEPTION", "ILLEGALSTATEEXCEPTION",
"THROWABLE", "ERROR", "CLASS", "ENUM", "ITERABLE", "COMPARABLE", "CLONEABLE", "VOID", "MATH",
"SYSTEM", "AUTOCLOSEABLE",
// java.util
"LIST", "ARRAYLIST", "LINKEDLIST", "MAP", "HASHMAP", "LINKEDHASHMAP", "TREEMAP",
"CONCURRENTHASHMAP", "SORTEDMAP", "NAVIGABLEMAP", "SET", "HASHSET", "LINKEDHASHSET", "TREESET",
"SORTEDSET", "COLLECTION", "COLLECTIONS", "OPTIONAL", "OPTIONALINT", "OPTIONALLONG", "ITERATOR",
"QUEUE", "DEQUE", "ARRAYDEQUE", "STACK", "VECTOR", "COMPARATOR", "ARRAYS", "OBJECTS", "UUID",
"DATE", "CALENDAR", "LOCALE", "RANDOM", "SCANNER", "PROPERTIES", "ENUMSET", "ENUMMAP", "BITSET",
// java.util.stream / function
"STREAM", "INTSTREAM", "LONGSTREAM", "DOUBLESTREAM", "COLLECTORS", "FUNCTION", "BIFUNCTION",
"CONSUMER", "BICONSUMER", "SUPPLIER", "PREDICATE", "BIPREDICATE", "UNARYOPERATOR", "BINARYOPERATOR",
// java.time
"LOCALDATE", "LOCALDATETIME", "LOCALTIME", "INSTANT", "DURATION", "PERIOD", "ZONEDDATETIME",
"OFFSETDATETIME", "ZONEID", "DAYOFWEEK", "MONTH", "YEAR", "CHRONOUNIT",
// java.io / nio
"FILE", "PATH", "PATHS", "FILES", "INPUTSTREAM", "OUTPUTSTREAM", "READER", "WRITER",
"BUFFEREDREADER", "IOEXCEPTION", "UNCHECKEDIOEXCEPTION",
// java.math
"BIGDECIMAL", "BIGINTEGER",
// java.util.concurrent / atomic
"ATOMICINTEGER", "ATOMICLONG", "ATOMICBOOLEAN", "ATOMICREFERENCE", "COMPLETABLEFUTURE", "FUTURE",
"EXECUTOR", "EXECUTORSERVICE", "EXECUTORS", "TIMEUNIT", "COUNTDOWNLATCH",
// logging
"LOGGER", "LOGGERFACTORY", "LOG",
// common frameworks: CDI / JPA / Quarkus / JAX-RS reactive
"ENTITYMANAGER", "SESSION", "STATELESSSESSION", "INSTANCE", "EVENT", "PROVIDER", "TYPELITERAL",
"UNI", "MULTI", "RESPONSE", "PANACHEQUERY", "PANACHEENTITY", "PANACHEENTITYBASE");
private ExternalTypeNames() {
}
/**
* @return whether {@code name} — a simple or fully-qualified type name, in any case — is a
* JDK/stdlib/framework type rather than something this project declares.
*/
public static boolean isExternal(String name) {
int lastDot = name.lastIndexOf('.');
String simple = lastDot >= 0 ? name.substring(lastDot + 1) : name;
return NAMES.contains(simple.toUpperCase(Locale.ROOT));
}
}

View File

@@ -33,5 +33,21 @@ public enum NodeType {
* (roadmap item 45), extracted from the {@code ADD-XML-LINE}/{@code COMPRESS '<' tag '>' value}
* emit idiom. Contained by the wrapper {@code MODULE}; exposed by {@code /modules/{name}/payload}.
*/
PAYLOAD_FIELD
PAYLOAD_FIELD,
/**
* Item 141: one contiguous <b>comment block</b> of a source file — a run of Natural full-line
* comments ({@code *}, {@code **SAG}), a single trailing {@code /*} comment, or a Java
* {@code //} / block / Javadoc comment.
*
* <p>The block text is the node's {@code value}; its {@code name} is the synthetic key
* {@code comment@<startLine>} and carries <b>no</b> comment text. That is deliberate:
* {@code SEARCH_IDENTIFIER} matches every node's {@code name} with no type filter, so prose in
* {@code name} would silently turn every {@code contains=true} identifier lookup into a full-text
* search. For the same reason {@code SEARCH_BY_VALUE} excludes this type unless the caller opts
* in with {@code includeComments}.
*
* <p>Attached to the declaration it documents by {@link EdgeType#DOCUMENTS}, never by
* {@code CONTAINS} — see that constant for why. Exposed by {@code /modules/{name}/comments}.
*/
COMMENT
}

View File

@@ -9,6 +9,10 @@ import com.github.javaparser.ast.CompilationUnit;
import com.github.javaparser.ast.ImportDeclaration;
import com.github.javaparser.ast.Node;
import com.github.javaparser.ast.body.*;
import com.github.javaparser.ast.comments.BlockComment;
import com.github.javaparser.ast.comments.Comment;
import com.github.javaparser.ast.comments.JavadocComment;
import com.github.javaparser.ast.comments.LineComment;
import com.github.javaparser.ast.expr.*;
import com.github.javaparser.ast.nodeTypes.NodeWithAnnotations;
import com.github.javaparser.ast.type.ClassOrInterfaceType;
@@ -434,6 +438,48 @@ public final class JavaParser implements LanguageParser {
return names.isEmpty() ? null : names;
}
/**
* The HTTP-method annotations JAX-RS defines. Kept as a set rather than matched by suffix so an
* unrelated annotation cannot masquerade as a verb.
*/
private static final Set<String> HTTP_METHOD_ANNOTATIONS =
Set.of("GET", "POST", "PUT", "DELETE", "PATCH", "HEAD", "OPTIONS");
/**
* Item 130: the {@code @Path} value declared on {@code declaration}, or {@code null} if it has
* none. Resolves a constant reference ({@code @Path(Paths.PARTNER)}) the same way a JPA
* {@code @Column} name is resolved; an expression that is neither a literal nor a resolvable
* constant yields {@code null} rather than a guess, because a wrong path is worse than a missing
* one for something whose whole purpose is routing.
*/
@Nullable
private static String restPath(NodeWithAnnotations<?> declaration, String className,
Map<String, String> constants) {
for (AnnotationExpr annotation : declaration.getAnnotations()) {
if (!annotation.getNameAsString().equals("Path")) {
continue;
}
@Nullable Expression member = annotationMember(annotation, "value");
return member == null ? null : resolveAnnotationString(member, className, constants);
}
return null;
}
/**
* Item 130: the HTTP verb a handler method is annotated with ({@code @GET}, {@code @POST}, …), or
* {@code null} when it carries none — which is how a sub-resource locator is told apart from an
* endpoint.
*/
@Nullable
private static String httpMethod(MethodDeclaration method) {
for (AnnotationExpr annotation : method.getAnnotations()) {
if (HTTP_METHOD_ANNOTATIONS.contains(annotation.getNameAsString())) {
return annotation.getNameAsString();
}
}
return null;
}
/**
* Returns the value expression of annotation member {@code member}, handling normal
* (multi-pair) and single-member annotations; {@code null} for marker annotations or a
@@ -855,6 +901,141 @@ public final class JavaParser implements LanguageParser {
}
}
/**
* Item 128: emits the <b>reference sites</b> a call graph does not see — {@code import} statements,
* declared type positions (field, parameter, return) and annotation usages — as {@code REFERENCES}
* edges carrying a {@code kind} property ({@code IMPORT}/{@code TYPE}/{@code ANNOTATION}). Together
* with the existing {@code CALLS}/{@code EXTENDS}/{@code IMPLEMENTS}/{@code INJECTS} edges this is
* what {@code /search/references} answers from: "every place this name is mentioned", which is the
* honest basis for scoping a rename.
*
* <p><b>Deliberately bounded.</b> Local-variable types and generic type arguments are not indexed:
* they multiply the edge count for far less value than the positions above. The practical
* consequence is worth stating plainly — a reference within the <em>same package</em> has no import,
* so for a rename inside one package this index rests on the declared-type positions alone.
*
* <p>Static and asterisk imports are skipped (as {@link TypeResolver} skips them): neither names a
* single type unambiguously, and guessing which segment is the type would put wrong lines in a
* result that exists to be trusted.
*/
private static void addReferenceEdges(CompilationUnit unit, TypeDeclaration<?> type, AstNode typeNode,
Map<String, AstNode> referencedModules, TypeResolver types,
List<AstNode> nodes, List<AstEdge> edges) {
// Imports belong to the file, not to each type in it, so they are emitted once — from the
// first top-level type. Emitting them per nested class would multiply identical rows.
// getTypes() is the file's top-level declarations; the first of them owns the import block.
if (type.isTopLevelType() && !unit.getTypes().isEmpty() && unit.getTypes().get(0) == type) {
String ownPackage = unit.getPackageDeclaration().map(pd -> pd.getNameAsString()).orElse("");
for (ImportDeclaration imp : unit.getImports()) {
if (imp.isAsterisk() || imp.isStatic()) {
continue;
}
String qualified = imp.getNameAsString();
if (!isProjectType(qualified, ownPackage)) {
continue;
}
int line = imp.getBegin().map(pos -> pos.line).orElse(typeNode.startLine());
reference(qualified, "IMPORT", line, typeNode, referencedModules, nodes, edges);
}
}
for (FieldDeclaration field : type.getFields()) {
int line = field.getBegin().map(pos -> pos.line).orElse(typeNode.startLine());
reference(types.resolve(baseTypeName(field.getElementType().asString())), "TYPE", line,
typeNode, referencedModules, nodes, edges);
addAnnotationReferences(field.getAnnotations(), typeNode, referencedModules, types, nodes, edges);
}
for (CallableDeclaration<?> callable : callables(type)) {
int line = callable.getBegin().map(pos -> pos.line).orElse(typeNode.startLine());
if (callable instanceof MethodDeclaration method) {
reference(types.resolve(baseTypeName(method.getType().asString())), "TYPE", line,
typeNode, referencedModules, nodes, edges);
}
for (Parameter parameter : callable.getParameters()) {
int parameterLine = parameter.getBegin().map(pos -> pos.line).orElse(line);
reference(types.resolve(baseTypeName(parameter.getType().asString())), "TYPE", parameterLine,
typeNode, referencedModules, nodes, edges);
}
addAnnotationReferences(callable.getAnnotations(), typeNode, referencedModules, types, nodes, edges);
}
addAnnotationReferences(type.getAnnotations(), typeNode, referencedModules, types, nodes, edges);
}
private static List<CallableDeclaration<?>> callables(TypeDeclaration<?> type) {
List<CallableDeclaration<?>> callables = new ArrayList<>(type.getMethods());
callables.addAll(type.getConstructors());
return callables;
}
private static void addAnnotationReferences(List<AnnotationExpr> annotations, AstNode typeNode,
Map<String, AstNode> referencedModules, TypeResolver types,
List<AstNode> nodes, List<AstEdge> edges) {
for (AnnotationExpr annotation : annotations) {
int line = annotation.getBegin().map(pos -> pos.line).orElse(typeNode.startLine());
reference(types.resolve(annotation.getNameAsString()), "ANNOTATION", line, typeNode,
referencedModules, nodes, edges);
}
}
/**
* Emits one {@code REFERENCES} edge, skipping the JDK/framework names that would otherwise mint a
* placeholder node per mention — created, persisted and swept again on every single ingest.
*/
private static void reference(String target, String kind, int line, AstNode typeNode,
Map<String, AstNode> referencedModules, List<AstNode> nodes,
List<AstEdge> edges) {
if (target.isEmpty() || ExternalTypeNames.isExternal(target)) {
return;
}
AstNode targetNode = referencedModule(referencedModules, nodes, target);
// MENTIONS, not REFERENCES: the call-graph traversals follow REFERENCES as wiring, and an
// import is not a call. See EdgeType.MENTIONS.
edges.add(edge(EdgeType.MENTIONS, typeNode.id(), targetNode.id(), line, null,
Map.of("refKind", kind)));
}
/**
* @return {@code type} without array brackets and generic arguments — {@code List<Foo>[]} yields
* {@code List}. The type <em>argument</em> is deliberately dropped rather than indexed as a second
* reference; see {@link #addReferenceEdges}.
*/
private static String baseTypeName(String type) {
String base = type;
int generic = base.indexOf('<');
if (generic >= 0) {
base = base.substring(0, generic);
}
int bracket = base.indexOf('[');
if (bracket >= 0) {
base = base.substring(0, bracket);
}
return base.trim();
}
/**
* @return whether {@code qualified} (an import's FQN) plausibly names a type <em>inside this
* project</em>, judged by sharing the first two package segments with the importing file's own
* package ({@code com.uniqagroup.…} importing {@code com.uniqagroup.…}).
*
* <p>A heuristic, chosen over a hardcoded list of external package prefixes because it adapts to
* whatever organisation a project belongs to. It errs toward <em>excluding</em>: an internal type
* under a differently-rooted package is missed, which costs a row in the reference index —
* whereas including everything would mint a placeholder node for every {@code java.util} and
* framework import, on every ingest, only for the finalize sweep to delete them again.
*/
private static boolean isProjectType(String qualified, String ownPackage) {
String prefix = firstTwoSegments(ownPackage);
return !prefix.isEmpty() && qualified.startsWith(prefix + ".");
}
private static String firstTwoSegments(String packageName) {
int first = packageName.indexOf('.');
if (first < 0) {
return packageName;
}
int second = packageName.indexOf('.', first + 1);
return second < 0 ? packageName : packageName.substring(0, second);
}
private static boolean hasCdiScope(TypeDeclaration<?> type) {
return type.getAnnotations().stream().anyMatch(a -> CDI_SCOPES.contains(a.getNameAsString()));
}
@@ -1075,12 +1256,23 @@ public final class JavaParser implements LanguageParser {
moduleProps.put("description", firstLine);
}
});
// Collected before the module props below, because item 130's class-level @Path may be
// written as a constant reference (@Path(PurPaths.PARTNER)) and must resolve the same way
// a @Column name does.
Map<String, String> constants = collectConstants(type);
// Item 29: generic annotation capture for search_annotation, independent of any
// annotation's specific interpretation above (@Entity/@Query/repository base types/...).
@Nullable String typeAnnotations = annotationNames(type);
if (typeAnnotations != null) {
moduleProps.put("annotations", typeAnnotations);
}
// Item 130: the class-level @Path, so "which endpoint path reaches this handler" is
// answerable from the graph. Annotations are otherwise stored by name only, and a JAX-RS
// path lives half on the class and half on the method — composing it needed the source.
@Nullable String typePath = restPath(type, className, constants);
if (typePath != null) {
moduleProps.put("restPath", typePath);
}
AstNode typeNode = node(NodeType.MODULE, fqn, sourceFile,
type.getBegin().map(p -> p.line).orElse(1),
type.getEnd().map(p -> p.line).orElse(1),
@@ -1105,7 +1297,6 @@ public final class JavaParser implements LanguageParser {
}
// Pass 1: collect resolved constant values (needed before resolving the entity table name).
Map<String, String> constants = collectConstants(type);
// JPA entity: resolve the table name and link the class to its DB_TABLE. A Panache
// active-record entity (extends PanacheEntity[Base]) is an entity even without @Entity.
@@ -1212,6 +1403,16 @@ public final class JavaParser implements LanguageParser {
if (methodAnnotations != null) {
methodProps.put("annotations", methodAnnotations);
}
// Item 130: the method half of a JAX-RS endpoint — its own @Path (often absent, which
// means "the class path itself") and the HTTP verb annotation.
@Nullable String methodPath = restPath(method, className, constants);
if (methodPath != null) {
methodProps.put("restPath", methodPath);
}
@Nullable String httpMethod = httpMethod(method);
if (httpMethod != null) {
methodProps.put("httpMethod", httpMethod);
}
// Item 33: modifier-derived kind, so an agent can ask "what must a subclass
// implement/not override" without reading the base class source by hand.
methodProps.put("kind", method.isAbstract() ? "abstract" : method.isFinal() ? "final" : "overridable");
@@ -1332,11 +1533,106 @@ public final class JavaParser implements LanguageParser {
// J2: CDI injection + class-literal wiring edges.
addWiringEdges(type, typeNode, referencedModules, types, nodes, edges);
// Item 128: the reference index — imports, declared type positions, annotation usages.
addReferenceEdges(unit, type, typeNode, referencedModules, types, nodes, edges);
}
// Item 141: comments are file-level, so this runs once after the type loop — inside it, a file
// with nested types would emit every comment once per type.
addCommentNodes(unit, sourceFile, nodes, edges);
return new ParseResult(nodes, edges);
}
/**
* Item 141: one {@link NodeType#COMMENT} node per contiguous comment block, each with a
* {@link EdgeType#DOCUMENTS} edge to the declaration it documents.
*
* <p>Comments are free here: JavaParser already attaches them to the {@link CompilationUnit}
* (the default {@code ParserConfiguration.setAttributeComments(true)}), so this collects what the
* parse produced rather than re-lexing. Adjacent {@code //} lines are folded into one block —
* a five-line {@code //} preamble is one comment, not five.
*/
private static void addCommentNodes(CompilationUnit unit, String sourceFile,
List<AstNode> nodes, List<AstEdge> edges) {
List<Comment> comments = unit.getAllComments().stream()
.filter(c -> c.getBegin().isPresent() && c.getEnd().isPresent())
.sorted(Comparator.comparingInt(c -> c.getBegin().orElseThrow().line))
.toList();
if (comments.isEmpty()) {
return;
}
// Declaration nodes of THIS file only: placeholder modules for referenced classes carry
// sourceFile "" and are not a target a comment in this file could ever document.
List<AstNode> targets = nodes.stream()
.filter(n -> n.sourceFile().equals(sourceFile))
.filter(CommentBlocks::isDocumentable)
.toList();
if (targets.isEmpty()) {
return; // nothing parsed from this file — a comment with no target is not emitted
}
List<Comment> block = new ArrayList<>();
for (Comment comment : comments) {
boolean extendsBlock = !block.isEmpty()
&& comment instanceof LineComment
&& block.get(block.size() - 1) instanceof LineComment previous
&& previous.getEnd().orElseThrow().line + 1 == comment.getBegin().orElseThrow().line;
if (!extendsBlock && !block.isEmpty()) {
emitCommentBlock(block, sourceFile, targets, nodes, edges);
block = new ArrayList<>();
}
block.add(comment);
}
emitCommentBlock(block, sourceFile, targets, nodes, edges);
}
private static void emitCommentBlock(List<Comment> block, String sourceFile, List<AstNode> targets,
List<AstNode> nodes, List<AstEdge> edges) {
if (block.isEmpty()) {
return;
}
int startLine = block.get(0).getBegin().orElseThrow().line;
int endLine = block.get(block.size() - 1).getEnd().orElseThrow().line;
String text = block.stream().map(JavaParser::commentText).collect(Collectors.joining("\n"));
@Nullable AstNode commentNode = CommentBlocks.commentNode("java", sourceFile, startLine, endLine,
text, commentKind(block.get(0)));
if (commentNode == null) {
return; // an empty /**/ or a bare // documents nothing
}
nodes.add(commentNode);
edges.add(CommentBlocks.documentsEdge(commentNode,
CommentBlocks.documentedNode(targets, startLine, endLine)));
}
/**
* A comment's text with its decoration removed. {@code getContent()} hands back the inner text
* verbatim, which for a block/Javadoc comment still carries every line's indentation and leading
* {@code *} — so the stored text read {@code "* ServiceEndpoint: …\n\t * UpmsObject: …"}. Natural
* comments already lose their markers, and a reader (human or agent) wants the same thing here:
* the sentence, not the comment syntax.
*/
private static String commentText(Comment comment) {
if (comment instanceof LineComment) {
return comment.getContent().strip();
}
return comment.getContent().lines()
.map(line -> {
String stripped = line.strip();
return stripped.startsWith("*") ? stripped.substring(1).stripLeading() : stripped;
})
.collect(Collectors.joining("\n"))
.strip();
}
private static String commentKind(Comment comment) {
if (comment instanceof JavadocComment) {
return CommentProperties.KIND_JAVADOC;
}
return comment instanceof BlockComment ? CommentProperties.KIND_BLOCK : CommentProperties.KIND_LINE;
}
/**
* The handful of facts that differ between the four kinds of Java type declaration, resolved once
* so the ingest loop below can treat them uniformly.

View File

@@ -1,6 +1,7 @@
package com.agenticcode.parserjava;
import com.agenticcode.parsercore.ast.model.AstNode;
import com.agenticcode.parsercore.ast.model.CommentProperties;
import com.agenticcode.parsercore.ast.model.EdgeType;
import com.agenticcode.parsercore.ast.model.NodeType;
import com.agenticcode.parsercore.ast.spi.LanguageParser;
@@ -373,4 +374,126 @@ class JavaParserTest {
assertFalse(JavaParser.UNRESOLVED_FIELD_RECEIVER_PREFIX.chars().anyMatch(c -> c < 0x20),
"a control character here silently disables every Cypher guard that matches on it");
}
// ---------------------------------------------------------------------
// Item 141: comment blocks as graph nodes
// ---------------------------------------------------------------------
private static AstNode commentAt(LanguageParser.ParseResult result, int startLine) {
return result.nodes().stream()
.filter(n -> n.type() == NodeType.COMMENT && n.startLine() == startLine)
.findFirst()
.orElseThrow(() -> new AssertionError("No COMMENT node starting at line " + startLine
+ " (have: " + result.nodes().stream().filter(n -> n.type() == NodeType.COMMENT)
.map(n -> n.startLine() + ":" + n.value()).toList() + ")"));
}
@Test
void javadocBlockBecomesCommentNodeDocumentingTheType() {
String content = """
package p;
/**
* UpmsObject: PartnerCs, UpmsAdapter: Update
*/
class PartnerController {
}
""";
LanguageParser.ParseResult result = parser.parse("PartnerController.java", content);
AstNode comment = commentAt(result, 2);
assertEquals(CommentProperties.KIND_JAVADOC, prop(comment, CommentProperties.KIND));
assertEquals("UpmsObject: PartnerCs, UpmsAdapter: Update", comment.value(),
"the per-line * decoration is not part of the text");
assertEquals("comment@2", comment.name(), "the name carries no prose (search/identifier matches names)");
assertTrue(hasEdge(result, EdgeType.DOCUMENTS, comment, "p.PartnerController", NodeType.MODULE));
}
@Test
void adjacentLineCommentsAreOneBlockAndDocumentTheFollowingMethod() {
String content = """
package p;
class A {
// WPARTX0S — DPARTFN0
// over the field subset
void update() {
}
}
""";
LanguageParser.ParseResult result = parser.parse("A.java", content);
List<AstNode> comments = result.nodes().stream().filter(n -> n.type() == NodeType.COMMENT).toList();
assertEquals(1, comments.size(), "two adjacent // lines are one block");
AstNode comment = comments.get(0);
assertEquals(3, comment.startLine());
assertEquals(4, comment.endLine());
assertEquals("2", prop(comment, CommentProperties.LINE_COUNT));
assertEquals("WPARTX0S — DPARTFN0\nover the field subset", comment.value());
assertTrue(hasEdge(result, EdgeType.DOCUMENTS, comment, "update", NodeType.FUNCTION));
}
@Test
void trailingCommentInsideAMethodDocumentsThatMethodNotTheNextOne() {
String content = """
package p;
class A {
void first() {
// nothing follows inside this body
}
void second() {
}
}
""";
LanguageParser.ParseResult result = parser.parse("A.java", content);
assertTrue(hasEdge(result, EdgeType.DOCUMENTS, commentAt(result, 4), "first", NodeType.FUNCTION));
}
@Test
void separatedLineCommentsAreSeparateBlocks() {
String content = """
package p;
class A {
// first
int a;
// second
int b;
}
""";
LanguageParser.ParseResult result = parser.parse("A.java", content);
assertEquals(2, result.nodes().stream().filter(n -> n.type() == NodeType.COMMENT).count());
assertEquals("first", commentAt(result, 3).value());
assertEquals("second", commentAt(result, 6).value());
}
@Test
void aFileWithNoCommentsGetsNoCommentNodes() {
LanguageParser.ParseResult result = parser.parse("A.java", "package p;\nclass A {\n}\n");
assertEquals(0, result.nodes().stream().filter(n -> n.type() == NodeType.COMMENT).count());
}
@Test
void javadocKeepsItsLinesButNotItsAsterisks() {
String content = """
package p;
/**
* ServiceEndpoint: partner.update.partnercs.Update
*
* UpmsObject: PartnerCs
*/
class A {
}
""";
LanguageParser.ParseResult result = parser.parse("A.java", content);
assertEquals("ServiceEndpoint: partner.update.partnercs.Update\n\nUpmsObject: PartnerCs",
commentAt(result, 2).value());
}
}

View File

@@ -2,6 +2,7 @@ package com.agenticcode.parsernatural;
import com.agenticcode.parsercore.ast.model.AstEdge;
import com.agenticcode.parsercore.ast.model.AstNode;
import com.agenticcode.parsercore.ast.model.CopycodeProperties;
import com.agenticcode.parsercore.ast.spi.LanguageParser.ParseResult;
import org.jspecify.annotations.Nullable;
@@ -29,8 +30,13 @@ public final class CopycodePreprocessor {
private static final Pattern INCLUDE_STMT = Pattern.compile("(?i)^\\s*INCLUDE\\s+(\\S+)\\s*(.*)$");
private static final Pattern DEFINE_DATA = Pattern.compile("(?i)^\\s*DEFINE\\s+DATA\\b");
private static final Pattern END_DEFINE = Pattern.compile("(?i)^\\s*END-DEFINE\\b");
// A positional argument: a quoted literal or a whitespace-delimited token.
private static final Pattern ARG = Pattern.compile("'[^']*'|\\S+");
// A positional argument: a quoted literal or a whitespace-delimited token. Item 121: Natural
// escapes a quote by doubling it, so '''YAGCHBN0''' is the single literal 'YAGCHBN0'. Without the
// '' alternative it split into three arguments ('', 'YAGCHBN0', '') and shifted every later
// position by one — CALLNAT &2& then named whatever &1& was, an ADABAS sort key.
private static final Pattern ARG = Pattern.compile("'(?:''|[^'])*'|\\S+");
// Item 120: an INCLUDE argument continuation — a line made up of nothing but quoted literals.
private static final Pattern ARG_CONTINUATION = Pattern.compile("^\\s*(?:'(?:''|[^'])*'\\s*)+$");
private static final Pattern PARAM_REF = Pattern.compile("&(\\d+)&");
private static final int MAX_DEPTH = 10;
@@ -71,7 +77,20 @@ public final class CopycodePreprocessor {
if (!excludedMacros.contains(upper) && !stack.contains(upper)) {
CopycodeResolver.Copycode member = resolver.resolve(name);
if (member != null && !containsDefineData(member.content())) {
String substituted = substitute(member.content(), parseArgs(inc.group(2)));
List<String> args = parseArgs(inc.group(2));
// Item 120: a Natural INCLUDE may spread its arguments over several lines.
// Only the first was read, so &3& survived substitution verbatim and the
// CALLNAT it feeds was dropped — silently, and not even as an unresolved
// dynamic call. The member's own highest &n& gives the arity, so the
// lookahead is bounded by the copycode rather than by guesswork.
int arity = maxParamRef(member.content());
int lastArgLine = i;
while (args.size() < arity && lastArgLine + 1 < lines.length
&& isArgContinuation(stripInlineComment(lines[lastArgLine + 1]))) {
args.addAll(parseArgs(stripInlineComment(lines[lastArgLine + 1])));
lastArgLine++;
}
String substituted = substitute(member.content(), args);
stack.push(upper);
// includedAt keeps the HOST module's INCLUDE line at every nesting level;
// the chain records where each hop was written.
@@ -82,7 +101,11 @@ public final class CopycodePreprocessor {
includePath.isEmpty() ? step : includePath + ">" + step,
depth + 1, stack);
stack.pop();
continue; // drop the INCLUDE line itself
// Drop the INCLUDE line and the argument lines it consumed. Those lines are
// the statement's arguments, not code; emitting them (as happened before)
// fed stray literals to the parser.
i = lastArgLine;
continue;
}
}
}
@@ -105,10 +128,10 @@ public final class CopycodePreprocessor {
return props;
}
Map<String, String> merged = props == null ? new LinkedHashMap<>() : new LinkedHashMap<>(props);
merged.put("viaCopycode", origin.viaCopycode());
merged.put("includedAt", Integer.toString(origin.includedAt()));
merged.put(CopycodeProperties.VIA_COPYCODE, origin.viaCopycode());
merged.put(CopycodeProperties.INCLUDED_AT, Integer.toString(origin.includedAt()));
if (!origin.includePath().isEmpty()) {
merged.put("includePath", origin.includePath());
merged.put(CopycodeProperties.INCLUDE_PATH, origin.includePath());
}
return merged;
}
@@ -204,11 +227,40 @@ public final class CopycodePreprocessor {
private static String dequote(String arg) {
String s = arg.trim();
if (s.length() >= 2 && s.charAt(0) == '\'' && s.charAt(s.length() - 1) == '\'') {
return s.substring(1, s.length() - 1);
// Item 121: unescape the doubled quote, so '''X''' yields the literal 'X' the copycode
// expects — not the bare X, which would stop CALLNAT from recognising it as a literal.
return s.substring(1, s.length() - 1).replace("''", "'");
}
return s;
}
/**
* The highest {@code &n&} the member actually uses — i.e. the include's arity, and the bound on
* how far {@link #expandInto} may look ahead for arguments (item 120).
*/
private static int maxParamRef(String content) {
int max = 0;
Matcher m = PARAM_REF.matcher(content);
while (m.find()) {
max = Math.max(max, Integer.parseInt(m.group(1)));
}
return max;
}
/**
* Whether {@code line} continues an {@code INCLUDE}'s argument list: nothing but quoted literals.
*
* <p>Such lines do occur elsewhere in Natural — 3081 of them in {@code upms}, continuations of
* {@code WRITE} and of table initialisations — so this test is <em>not</em> safe on its own. It is
* safe where it is used: only ever on the line following an {@code INCLUDE} (or one of its own
* argument lines), and an {@code INCLUDE} terminates the preceding statement, so another
* statement's continuation can never appear there.
*/
private static boolean isArgContinuation(String line) {
String trimmed = line.trim();
return !trimmed.isEmpty() && !trimmed.startsWith("*") && ARG_CONTINUATION.matcher(line).matches();
}
private static boolean containsDefineData(String content) {
for (String line : content.split("\r?\n", -1)) {
if (DEFINE_DATA.matcher(stripInlineComment(line)).find()) {

View File

@@ -35,8 +35,12 @@ public final class NaturalCoarseScanner implements CoarseScanner {
private static final Pattern DEFINE_SUBROUTINE = Pattern.compile("(?i)^\\s*DEFINE\\s+SUBROUTINE\\s+(\\S+)");
private static final Pattern END_SUBROUTINE = Pattern.compile("(?i)^\\s*END-SUBROUTINE\\b");
private static final Pattern PERFORM = Pattern.compile("(?i)^\\s*PERFORM\\s+(\\S+)");
private static final Pattern CALLNAT = Pattern.compile("(?i)CALLNAT\\s+'([^']+)'");
private static final Pattern CALLNAT_DYNAMIC = Pattern.compile("(?i)CALLNAT\\s+(#?[A-Za-z][#A-Za-z0-9.\\-]*)");
// Item 123: both string delimiters, shared with the deep tier so the rule exists once.
private static final Pattern CALLNAT = NaturalLines.CALLNAT_LITERAL;
// The &n& alternative (item 120) catches an unbound copycode parameter — an unknown target, which
// belongs in /dynamic-calls/unresolved rather than being silently dropped.
private static final Pattern CALLNAT_DYNAMIC =
Pattern.compile("(?i)CALLNAT\\s+(&\\d+&|#?[A-Za-z][#A-Za-z0-9.\\-]*)");
private static final Pattern SELECT_FROM = Pattern.compile("(?i)^\\s*SELECT\\b");
// FROM must be a standalone SQL keyword, not the "-FROM" tail of a hyphenated Natural identifier
// (e.g. a SELECT column `DAT-CALC-FROM (*)`), which a bare `\bFROM` would misread as `FROM (*)`.
@@ -268,6 +272,9 @@ public final class NaturalCoarseScanner implements CoarseScanner {
// are attributed exactly as NaturalParser attributes them (caller = currentFunction ?: module).
@Nullable AstNode currentFunction = null;
boolean inDefineData = false;
// Item 75: inside an INIT<...>/CONST<...> value clause that spans several lines. Must mirror the
// deep parser exactly (item 59) or the coarse identifier index diverges from the deep parse.
boolean inMultiLineValue = false;
for (int i = 0; i < lines.length; i++) {
String line = stripInlineComment(lines[i]);
int lineNo = i + 1;
@@ -288,6 +295,17 @@ public final class NaturalCoarseScanner implements CoarseScanner {
continue;
}
if (inDefineData) {
// Item 75: skip the continuation lines of a multi-line INIT<...>/CONST<...> — they are
// values, not declarations, but they tokenize as one (`166 ,` -> level 166, name ",").
if (inMultiLineValue) {
if (NaturalFieldTokenizer.closesMultiLineValue(line)) {
inMultiLineValue = false;
}
continue;
}
if (NaturalFieldTokenizer.opensMultiLineValue(line)) {
inMultiLineValue = true;
}
// Copybook include (PARAMETER/LOCAL USING): a coarse INCLUDES reference to the data area.
Matcher include = INCLUDE.matcher(line);
if (include.find()) {
@@ -354,7 +372,7 @@ public final class NaturalCoarseScanner implements CoarseScanner {
Matcher callnat = CALLNAT.matcher(line);
if (NaturalLines.findOutsideStringLiteral(callnat, line)) {
AstNode target = placeholder(NodeType.MODULE, callnat.group(1), lineNo);
AstNode target = placeholder(NodeType.MODULE, NaturalLines.literalTarget(callnat), lineNo);
nodes.add(target);
edges.add(edge(EdgeType.CALLS, caller, target.id(), lineNo, Map.of("callKind", CallKind.CALLNAT.name())));
continue;

View File

@@ -40,6 +40,47 @@ public final class NaturalFieldTokenizer {
private static final Pattern CONST_VALUE = Pattern.compile("(?i)CONST\\s*<\\s*([^>]*)>");
private static final Pattern INIT_VALUE = Pattern.compile("(?i)INIT\\s*<\\s*([^>]*)>");
/**
* Opening of a {@code CONST<...>} / {@code INIT<...>} value clause, used to recognise the
* multi-line form (item 75).
*/
private static final Pattern VALUE_CLAUSE_OPEN = Pattern.compile("(?i)(?:CONST|INIT)\\s*<");
/**
* True when this line opens a {@code CONST<...>}/{@code INIT<...>} whose closing {@code >} is on a
* later line, e.g. an initialiser listing one value per line. The continuation lines are <em>not</em>
* field declarations, but they look like one to {@link #parse(String)} — {@code 166 , /*CREDIT NOTE}
* tokenizes as level 166, name {@code ","} — which is where the {@code DATA_STRUCTURE} named
* {@code ","} in the graph came from (item 75). A caller inside {@code DEFINE DATA} should skip every
* following line until {@link #closesMultiLineValue(String)} reports the clause closed.
*
* <p>The single-line form (the normal case) never triggers this: its {@code <} and {@code >} balance
* on the same line.
*/
public static boolean opensMultiLineValue(String line) {
return VALUE_CLAUSE_OPEN.matcher(line).find() && angleBalance(line) > 0;
}
/**
* True when this line closes a value clause left open by {@link #opensMultiLineValue(String)}.
*/
public static boolean closesMultiLineValue(String line) {
return angleBalance(line) < 0;
}
private static int angleBalance(String line) {
int balance = 0;
for (int i = 0; i < line.length(); i++) {
char c = line.charAt(i);
if (c == '<') {
balance++;
} else if (c == '>') {
balance--;
}
}
return balance;
}
private NaturalFieldTokenizer() {
}

View File

@@ -1,6 +1,7 @@
package com.agenticcode.parsernatural;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
/**
* Shared lexical helpers for a single line of Natural source, used by both ingest tiers — the deep
@@ -19,9 +20,32 @@ import java.util.regex.Matcher;
*/
final class NaturalLines {
/**
* A {@code CALLNAT} with a literal target (item 123). {@link #isInsideStringLiteral} has always
* known that Natural delimits a literal with {@code '} <em>or</em> {@code "}, but the call
* patterns in both tiers accepted only the apostrophe — so a {@code CALLNAT "YCOMIBNH"} matched
* nothing and the call vanished. That form is not exotic: it is what the browse copycodes
* produce, because {@code '"YCOMIBNH"'} is how a quoted string is passed as an {@code INCLUDE}
* argument (7232 uses in {@code upms}, against 2521 for the {@code '''X'''} spelling).
*
* <p>Group 1 is the delimiter, <b>group 2 the target</b>; a doubled delimiter inside is an escape.
*/
static final Pattern CALLNAT_LITERAL =
Pattern.compile("(?i)CALLNAT\\s+(['\"])((?:\\1\\1|(?!\\1).)+)\\1");
private NaturalLines() {
}
/**
* The target of a {@link #CALLNAT_LITERAL} match, with any doubled delimiter unescaped.
*
* @param matcher positioned on a match, e.g. by {@link #findOutsideStringLiteral}
*/
static String literalTarget(Matcher matcher) {
String delimiter = matcher.group(1);
return matcher.group(2).replace(delimiter + delimiter, delimiter);
}
/**
* @return {@code s} truncated at the first {@code /*}, which starts a Natural comment running to
* end of line.

View File

@@ -40,11 +40,15 @@ public final class NaturalParser implements LanguageParser {
private static final Pattern DEFINE_SUBROUTINE = Pattern.compile("(?i)^\\s*DEFINE\\s+SUBROUTINE\\s+(\\S+)");
private static final Pattern END_SUBROUTINE = Pattern.compile("(?i)^\\s*END-SUBROUTINE\\b");
private static final Pattern PERFORM = Pattern.compile("(?i)^\\s*PERFORM\\s+(\\S+)");
private static final Pattern CALLNAT = Pattern.compile("(?i)CALLNAT\\s+'([^']+)'");
// Item 123: both string delimiters, shared with the Tier-1 scanner so the rule exists once.
private static final Pattern CALLNAT = NaturalLines.CALLNAT_LITERAL;
// Dynamic CALLNAT: target module name held in a variable, e.g. CALLNAT #WIF ... . Tried only
// after the quoted form fails, so it never shadows a literal CALLNAT.
// after the quoted form fails, so it never shadows a literal CALLNAT. The &n& alternative (item
// 120) catches a copycode parameter that stayed unbound because the include supplied no argument
// for it: the target is genuinely unknown, so it belongs in /dynamic-calls/unresolved — being
// dropped instead is what made "not analysed" indistinguishable from "nothing found".
private static final Pattern CALLNAT_DYNAMIC =
Pattern.compile("(?i)CALLNAT\\s+(#?[A-Za-z][#A-Za-z0-9.\\-]*)");
Pattern.compile("(?i)CALLNAT\\s+(&\\d+&|#?[A-Za-z][#A-Za-z0-9.\\-]*)");
// A single CALLNAT argument: a quoted literal or a (possibly qualified) identifier.
private static final Pattern CALLNAT_ARG = Pattern.compile("'[^']*'|[#A-Za-z][#A-Za-z0-9.\\-]*");
// A continuation line that instead begins a new statement, ending a multi-line CALLNAT arg list.
@@ -196,6 +200,16 @@ public final class NaturalParser implements LanguageParser {
// token of the rest — `V 1VDB2-VERSIS_GENAGREE VERSVW_GENAGREE DA:00,00ED:00.00PF:`. There is no
// `VIEW OF` text here, so VIEW_OF (which matches the source form) never fires on these files. `V` is
// not a Natural data type, so the prefix is unambiguous.
// Item 75: a Natural *filler* is declared `<level><byteCount>X` with no name — `489X` is level 4, 89
// bytes of padding. Neither data-area pattern can see that: the occurrence/length column swallows part
// of the digits, so each such line produced a DATA_STRUCTURE literally named "X" at a different (bogus)
// level. The (type,name,sourceFile) merge key then collapsed all of them onto a single node, which
// contained itself — the CONTAINS self-loops of item 75 — and, because a swallowed level pops the whole
// group stack, the fields declared after a filler were silently re-parented under it (SNA27R01.pda:
// C63P0027-GESCHL hung under "X" instead of its real group). At least one count digit is required, so a
// field genuinely named X (`5X` = level 5, name X) is still parsed as a field.
private static final Pattern DATA_AREA_FILLER = Pattern.compile("^\\s*\\d{2,}X\\s*(?:/\\*.*)?$");
private static final Pattern DATA_AREA_VIEW_DDM = Pattern.compile("^([A-Za-z][\\w-]*)");
// Item 45: the XML-emit idiom `COMPRESS '<' <tagVar> '>' <valueVar> ...` inside an ADD-XML-LINE
// subroutine — the '<'/'>' literals wrapping a tag variable then a value variable identify the
@@ -886,7 +900,7 @@ public final class NaturalParser implements LanguageParser {
private static List<String> topLevelNames(String[] lines) {
List<String> names = new ArrayList<>();
for (String line : lines) {
if (DATA_AREA_COMMENT.matcher(line).find()) {
if (DATA_AREA_COMMENT.matcher(line).find() || DATA_AREA_FILLER.matcher(line).matches()) {
continue;
}
// Item 99: must use the same split as parseDataArea — a phantom level-1 group here suppresses
@@ -935,7 +949,7 @@ public final class NaturalParser implements LanguageParser {
String line = lines[i];
int lineNo = i + 1;
if (DATA_AREA_COMMENT.matcher(line).find()) {
if (DATA_AREA_COMMENT.matcher(line).find() || DATA_AREA_FILLER.matcher(line).matches()) {
continue;
}
// Item 99: the anchored export grammar first, the permissive pattern as a fallback.
@@ -1048,12 +1062,139 @@ public final class NaturalParser implements LanguageParser {
*/
public ParseResult parse(String sourceFile, String content, CopycodeResolver resolver) {
if (isDataArea(sourceFile)) {
return parseDataArea(sourceFile, content.split("\r?\n", -1));
return withComments(sourceFile, content, parseDataArea(sourceFile, content.split("\r?\n", -1)));
}
CopycodePreprocessor.Expansion expansion =
CopycodePreprocessor.expand(sourceFile, content, resolver, FrameworkMacros.names());
ParseResult raw = parseModule(sourceFile, expansion.lines().toArray(new String[0]));
return CopycodePreprocessor.remap(sourceFile, raw, expansion.origins());
return withComments(sourceFile, content, CopycodePreprocessor.remap(sourceFile, raw, expansion.origins()));
}
/**
* Item 141: adds one {@link NodeType#COMMENT} node per contiguous comment block of
* {@code content}, each with a {@link EdgeType#DOCUMENTS} edge to the declaration it documents.
*
* <p>Reads the module's <b>own</b> file text, not the copycode-expanded line array, and runs
* <em>after</em> the remap so the targets already carry real file positions. A copycode's
* comments would otherwise land on host line numbers and be duplicated once per includer
* (items 75-B/75-C); the copycode is ingested as its own module, so its comments are reachable
* there — once, with correct lines.
*/
private static ParseResult withComments(String sourceFile, String content, ParseResult result) {
List<AstNode> targets = result.nodes().stream()
.filter(n -> n.sourceFile().equals(sourceFile))
.filter(CommentBlocks::isDocumentable)
.toList();
if (targets.isEmpty()) {
return result; // nothing parsed from this file — a comment with no target is not emitted
}
List<AstNode> nodes = new ArrayList<>(result.nodes());
List<AstEdge> edges = new ArrayList<>(result.edges());
String[] lines = content.split("\r?\n", -1);
int blockStart = -1;
@Nullable String blockKind = null;
StringBuilder blockText = new StringBuilder();
for (int i = 0; i < lines.length; i++) {
String line = lines[i];
@Nullable String kind = fullLineCommentKind(line);
if (kind != null) {
// A run of full-line comments is one block, but a **SAG directive never merges with a
// human comment: the two are different kinds of text and stay separately filterable.
if (blockKind != null && !blockKind.equals(kind)) {
emitComment(sourceFile, blockStart, i, blockText.toString(), blockKind, targets, nodes, edges);
blockText.setLength(0);
blockStart = -1;
}
if (blockStart < 0) {
blockStart = i + 1;
} else {
blockText.append('\n');
}
blockKind = kind;
blockText.append(commentBody(line, kind));
continue;
}
if (blockStart >= 0) {
emitComment(sourceFile, blockStart, i, blockText.toString(), blockKind, targets, nodes, edges);
blockText.setLength(0);
blockStart = -1;
blockKind = null;
}
int inline = inlineCommentStart(line);
if (inline >= 0) {
emitComment(sourceFile, i + 1, i + 1, line.substring(inline + 2),
CommentProperties.KIND_NATURAL_INLINE, targets, nodes, edges);
}
}
if (blockStart >= 0) {
emitComment(sourceFile, blockStart, lines.length, blockText.toString(), blockKind, targets, nodes, edges);
}
return new ParseResult(nodes, edges);
}
private static void emitComment(String sourceFile, int startLine, int lastLine, String text,
@Nullable String kind, List<AstNode> targets,
List<AstNode> nodes, List<AstEdge> edges) {
// Both 1-based and inclusive. The clamp is item 137's rule: no node may carry endLine < startLine.
int endLine = Math.max(startLine, lastLine);
@Nullable AstNode comment = CommentBlocks.commentNode("natural", sourceFile, startLine, endLine,
text, kind == null ? CommentProperties.KIND_NATURAL_BANNER : kind);
if (comment == null) {
return; // a bare `*` separator line carries no text
}
nodes.add(comment);
edges.add(CommentBlocks.documentsEdge(comment, CommentBlocks.documentedNode(targets, startLine, endLine)));
}
/**
* @return the comment kind of a Natural <b>full-line</b> comment (first non-blank character is
* {@code *}), or {@code null} if the line is code or blank. A {@code **SAG} generator directive is
* reported as {@link CommentProperties#KIND_SAG} — it is machine-written metadata, and
* {@link #extractDescription} already mines it for the module description.
*/
private static @Nullable String fullLineCommentKind(String line) {
String stripped = line.strip();
if (stripped.isEmpty() || stripped.charAt(0) != '*') {
return null;
}
return stripped.startsWith("**SAG") ? CommentProperties.KIND_SAG : CommentProperties.KIND_NATURAL_BANNER;
}
/**
* The text of a full-line comment: a {@code **SAG} directive is kept verbatim (the marker is part
* of the datum), a human comment loses its leading {@code *} run and the space after it.
*/
private static String commentBody(String line, String kind) {
if (CommentProperties.KIND_SAG.equals(kind)) {
return line.strip();
}
String stripped = line.strip();
int i = 0;
while (i < stripped.length() && stripped.charAt(i) == '*') {
i++;
}
return stripped.substring(i).stripLeading();
}
/**
* @return the index of the {@code /*} that starts a trailing comment on a code line, or -1.
*
* <p>Quote-aware, unlike {@link NaturalLines#stripInlineComment}: {@code MOVE 'A/*B' TO #X} holds
* no comment. That the strip path is not quote-aware is a separate, pre-existing precision bug —
* there it merely truncates a line, here it would manufacture a comment node out of a literal.
*/
private static int inlineCommentStart(String line) {
boolean inQuote = false;
for (int i = 0; i < line.length() - 1; i++) {
char c = line.charAt(i);
if (c == '\'') {
inQuote = !inQuote; // a doubled '' toggles twice, which is the same as not toggling
} else if (!inQuote && c == '/' && line.charAt(i + 1) == '*') {
return i;
}
}
return -1;
}
private ParseResult parseModule(String sourceFile, String[] lines) {
@@ -1114,6 +1255,8 @@ public final class NaturalParser implements LanguageParser {
}
boolean inDefineData = false;
// Item 75: inside an INIT<...>/CONST<...> value clause that spans several lines.
boolean inMultiLineValue = false;
String defineScope = "";
int paramPosition = 0;
Deque<LevelEntry> groupStack = new ArrayDeque<>();
@@ -1154,6 +1297,19 @@ public final class NaturalParser implements LanguageParser {
}
if (inDefineData) {
// Item 75: the continuation lines of a multi-line INIT<...>/CONST<...> are values, not
// field declarations, but they tokenize as one (`166 ,` -> level 166, name ","). Skip
// them until the clause closes.
if (inMultiLineValue) {
if (NaturalFieldTokenizer.closesMultiLineValue(line)) {
inMultiLineValue = false;
}
continue;
}
if (NaturalFieldTokenizer.opensMultiLineValue(line)) {
inMultiLineValue = true;
}
Matcher scopeMatcher = SCOPE_KEYWORD.matcher(line);
if (scopeMatcher.matches()) {
defineScope = scopeMatcher.group(1).toUpperCase(Locale.ROOT);
@@ -1356,7 +1512,8 @@ public final class NaturalParser implements LanguageParser {
Matcher callnatMatcher = CALLNAT.matcher(line);
if (NaturalLines.findOutsideStringLiteral(callnatMatcher, line)) {
AstNode target = node(NodeType.MODULE, callnatMatcher.group(1), "", lineNo, lineNo, null, null);
AstNode target = node(NodeType.MODULE, NaturalLines.literalTarget(callnatMatcher), "",
lineNo, lineNo, null, null);
nodes.add(target);
// Stamp the call kind and capture the positional argument list after the quoted
// module name for dataflow.

View File

@@ -115,7 +115,11 @@ operand = identifier | constant | system-variable | system-function-call
operand1, operand2, operand3, ... = operand ; (* numbered per-statement operands *)
constant = numeric-constant | character-string | hex-constant ;
character-string = "'" { any-character } "'" ; (* refined in Section 21.3 *)
character-string = "'" { any-character } "'" ; (* SIMPLIFIED — see Section 21.3 before
writing any pattern against this: BOTH "'" and
'"' delimit, and a doubled delimiter escapes.
Roadmap items 121/123 were both caused by
matching this line instead of 21.3. *)
system-variable = "*" identifier ; (* e.g. *ISN, *COUNTER, *ETID, *TIMD, *CONVID;
superseded by the enumerated
`sysvar-name` form in Section 17 *)

View File

@@ -0,0 +1,142 @@
package com.agenticcode.parsernatural;
import org.junit.jupiter.api.Test;
import java.util.List;
import java.util.Map;
import java.util.Set;
import static org.junit.jupiter.api.Assertions.*;
/**
* Items 120/121. Copycode expansion had no unit test — it was covered only end-to-end through
* {@code CopycodeExpansionIT}, which is why two argument-parsing bugs survived: both are invisible
* unless you look at the substituted text, and an IT only sees whether some edge came out.
*
* <p>The fixtures are the real {@code upms} browse idiom, not invented shapes. Both bugs cost real
* edges there: 263 module→module calls across 154 modules were absent from the graph, and absent
* from {@code /dynamic-calls/unresolved} too — so nothing said "not analysed".
*/
class CopycodePreprocessorTest {
/**
* Resolves a fixed set of members; anything else is unknown, as in production.
*/
private static CopycodeResolver resolver(Map<String, String> members) {
return name -> {
String content = members.get(name.toUpperCase());
return content == null ? null : new CopycodeResolver.Copycode("cpy/" + name + ".cpy", content);
};
}
private static String expand(String host, Map<String, String> members) {
return String.join("\n",
CopycodePreprocessor.expand("HOST.nat", host, resolver(members), Set.of()).lines());
}
/**
* Item 121: {@code '''X'''} is one Natural literal, not three arguments. When it split, every
* later position shifted by one and {@code CALLNAT &2&} named whatever {@code &1&} was — in
* {@code DAGCHEN0} an ADABAS sort key, {@code AGNT-CHG-CMP-SP}, which then became a phantom
* MODULE node while the real call to {@code YAGCHBN0} was lost.
*/
@Test
void aDoubledQuoteIsAnEscapeNotAnArgumentBoundary() {
String out = expand(
"INCLUDE BC8 '''AGNT-CHG-CMP-SP''' '''YAGCHBN0'''\n",
Map.of("BC8", "ASSIGN SORT-KEY = &1&\nCALLNAT &2&\n"));
assertTrue(out.contains("ASSIGN SORT-KEY = 'AGNT-CHG-CMP-SP'"), out);
assertTrue(out.contains("CALLNAT 'YAGCHBN0'"), out);
}
/**
* Item 120: the arguments continue on the next line. Only the first was read, so {@code &2&}
* stayed literally {@code &2&}.
*/
@Test
void argumentsOnContinuationLinesAreBound() {
String out = expand(
"INCLUDE BC8 '\"AGNT-CHG-CMP-SP\"'\n"
+ " '\"YAGCHBN0\"' 'YAGCHKEY' 'YAGCHROW'\n",
Map.of("BC8", "ASSIGN SORT-KEY = &1&\nCALLNAT &2& &3& &4&\n"));
assertTrue(out.contains("CALLNAT \"YAGCHBN0\" YAGCHKEY YAGCHROW"), out);
assertFalse(out.contains("&2&"), "unbound parameter left in the expansion: " + out);
}
/**
* The consumed argument lines must not also survive as source: they are the statement's
* arguments, and feeding them to the parser as code produced stray literals.
*/
@Test
void consumedContinuationLinesAreNotEmittedAsCode() {
String out = expand(
"INCLUDE BC8 'A'\n 'B' 'C'\nEND\n",
Map.of("BC8", "CALLNAT &2& &3&\n"));
assertFalse(out.contains("'B' 'C'"), "argument line leaked into the source: " + out);
assertTrue(out.contains("END"), out);
}
/**
* The lookahead is bounded by the member's own highest {@code &n&}. Once the parameters are
* satisfied it must stop, even if the next line would also look like an argument list — this is
* what keeps it from eating a following statement's continuation.
*/
@Test
void theLookaheadStopsOnceTheMembersParametersAreBound() {
String out = expand(
"INCLUDE C1 'A'\n"
+ " 'not-an-argument'\n",
Map.of("C1", "CALLNAT &1&\n"));
assertTrue(out.contains("CALLNAT A"), out);
assertTrue(out.contains("'not-an-argument'"), "a line past the arity was swallowed: " + out);
}
/**
* A line consisting solely of quoted literals is <em>not</em> by itself an argument list — 3081
* such lines exist in {@code upms} as continuations of {@code WRITE} and of table initialisations.
* They are safe only because they never directly follow an {@code INCLUDE}; if the include is not
* expanded at all, nothing may be consumed.
*/
@Test
void anUnexpandedIncludeConsumesNothing() {
String host = "INCLUDE NOSUCH 'A'\n 'B' 'C'\n";
String out = expand(host, Map.of());
assertEquals(host.stripTrailing(), out.stripTrailing(), "an unresolved include altered the source");
}
/**
* A parameter the include genuinely supplies no argument for. The target is unknown — which is a
* fact worth reporting, so it must survive as a marker for the dynamic-call patterns to pick up,
* not be quietly discarded.
*/
@Test
void aParameterWithNoArgumentSurvivesAsAMarker() {
String out = expand(
"INCLUDE C1 'A'\n",
Map.of("C1", "CALLNAT &1&\nCALLNAT &7&\n"));
assertTrue(out.contains("CALLNAT A"), out);
assertTrue(out.contains("CALLNAT &7&"), "the unbound parameter was dropped: " + out);
}
/**
* Line origins stay in step with the emitted lines after argument lines are dropped.
*/
@Test
void originsRemainAlignedWithTheEmittedLines() {
CopycodePreprocessor.Expansion e = CopycodePreprocessor.expand("HOST.nat",
"PERFORM ONE\nINCLUDE C1 'A'\n 'B'\nPERFORM TWO\n",
resolver(Map.of("C1", "CALLNAT &1& &2&\n")), Set.of());
assertEquals(e.lines().size(), e.origins().size());
List<String> lines = e.lines();
int callnat = lines.indexOf(lines.stream().filter(l -> l.startsWith("CALLNAT")).findFirst().orElseThrow());
assertEquals("cpy/C1.cpy", e.origins().get(callnat).sourceFile());
assertEquals(2, e.origins().get(callnat).includedAt(), "includedAt must name the INCLUDE line in the host");
}
}

View File

@@ -102,6 +102,48 @@ class NaturalCoarseScannerTest {
assertEquals("CALLNAT", prop(call, "callKind"));
}
/**
* Item 123: Natural delimits a string with {@code '} <em>or</em> {@code "} —
* {@link NaturalLines#isInsideStringLiteral} has always said so, but the call pattern accepted
* only the apostrophe, so {@code CALLNAT "YCOMIBNH"} matched nothing and the call vanished. That
* form is what the browse copycodes expand to, because {@code '"YCOMIBNH"'} is how a quoted
* string is passed as an {@code INCLUDE} argument: 7232 uses in {@code upms}, and it left
* modules like {@code YCARPBN1} reporting zero callers.
*/
@Test
void aCallnatTargetMayUseEitherStringDelimiter() {
ParseResult r = scanner.scan("CALLER.nat", "CALLNAT \"YCOMIBNH\" #X\nEND\n");
AstNode callee = node(r, NodeType.MODULE, "YCOMIBNH");
assertEquals("", callee.sourceFile(), "an unresolved CALLNAT target is a placeholder");
assertEquals("CALLNAT", prop(edge(r, EdgeType.CALLS, "YCOMIBNH"), "callKind"),
"a double-quoted target is a literal call, not a dynamic one");
}
/**
* The widened delimiter must not reopen bug #63: a {@code CALLNAT} inside a string literal is
* text, not a call — including when that literal is double-quoted.
*/
@Test
void aCallnatInsideAStringLiteralIsStillNotACall() {
ParseResult r = scanner.scan("PGM.nat",
"PRINT '==> callnat before GHOSTA'\nPRINT \"==> callnat before GHOSTB\"\nEND\n");
assertFalse(hasNode(r, NodeType.MODULE, "GHOSTA"));
assertFalse(hasNode(r, NodeType.MODULE, "GHOSTB"));
}
/**
* Item 120: a copycode parameter the include supplied no argument for. The target is genuinely
* unknown, so it must surface as an unresolved dynamic call rather than be dropped — being
* dropped is what made "not analysed" look identical to "nothing found".
*/
@Test
void anUnboundCopycodeParameterIsReportedAsADynamicCall() {
ParseResult r = scanner.scan("PGM.nat", "CALLNAT &3& #X\nEND\n");
AstEdge call = edge(r, EdgeType.CALLS, "&3&");
assertEquals("CALLNAT_DYNAMIC", prop(call, "callKind"));
assertEquals("&3&", prop(call, "dynamicVar"));
}
@Test
void dynamicCallnatRecordsTheDispatchVariable() {
String src = "DEFINE DATA LOCAL\n01 #PGM (A8)\nEND-DEFINE\nCALLNAT #PGM #X\nEND\n";
@@ -306,4 +348,28 @@ class NaturalCoarseScannerTest {
.count(),
"UPDATE(label.) and DELETE(label.) each write the loop's table");
}
/**
* Item 75 / item 59: the coarse identifier index must skip the continuation lines of a multi-line
* {@code INIT<...>} exactly as the deep parser does — otherwise {@code 166 ,} enters the index as an
* identifier named {@code ","} and the two tiers disagree on the file's fields.
*/
@Test
void multiLineInitValuesDoNotEnterTheIdentifierIndex() {
String src = """
DEFINE DATA LOCAL
1 #V-LITERALES (N6/1:3)
INIT< 165 ,
166 ,
167
>
1 #TXT-TEXTO (A80)
END-DEFINE
END
""";
ParseResult r = scanner.scan("PGM.nat", src);
assertFalse(hasNode(r, NodeType.VARIABLE, ","), "INIT continuation values are not identifiers");
assertTrue(hasNode(r, NodeType.VARIABLE, "#V-LITERALES"));
assertTrue(hasNode(r, NodeType.VARIABLE, "#TXT-TEXTO"), "declaration after the closing > is indexed");
}
}

View File

@@ -2,6 +2,7 @@ package com.agenticcode.parsernatural;
import com.agenticcode.parsercore.ast.model.AstEdge;
import com.agenticcode.parsercore.ast.model.AstNode;
import com.agenticcode.parsercore.ast.model.CommentProperties;
import com.agenticcode.parsercore.ast.model.EdgeType;
import com.agenticcode.parsercore.ast.model.NodeType;
import com.agenticcode.parsercore.ast.spi.LanguageParser;
@@ -413,6 +414,84 @@ class NaturalParserTest {
assertEquals("N8", datStart.dataType());
}
/**
* Item 75: a Natural filler declaration ({@code <level><byteCount>X}, here {@code 51X}) is unnamed
* padding and must produce no node at all. It used to be read as a field literally named {@code X} at
* a bogus level 1, which popped the whole group stack — so every field after it was re-parented under
* that phantom, and because all fillers of a file collapsed onto one node by the merge key, the node
* ended up containing itself (the CONTAINS self-loops of item 75). A field genuinely named {@code X}
* ({@code 3X}) must still be parsed, which is why the filler pattern requires a byte count.
*/
@Test
void fillerDeclarationProducesNoNodeAndDoesNotReparentFollowingFields() throws IOException {
String content = readFixture("WGEAGL01_SAMPLE.pda");
LanguageParser.ParseResult result = parser.parse("WGEAGL01_SAMPLE.pda", content);
assertFalse(hasNode(result, NodeType.DATA_STRUCTURE, "X"), "filler 51X must not become a group");
assertTrue(result.edges().stream().noneMatch(e -> e.type() == EdgeType.CONTAINS
&& e.sourceId().equals(e.targetId())), "no self-containment");
// the field declared after the filler keeps its real parent (the REDEFINE group), not the filler
AstNode redefineGroup = findNode(result, NodeType.DATA_STRUCTURE, "#P-ADD-PARM");
AstNode codTypesys = findNode(result, NodeType.VARIABLE, "COD-TYPESYS");
assertEquals(redefineGroup.name(), parentName(result, codTypesys));
assertTrue(hasEdge(result, EdgeType.CONTAINS, redefineGroup, "COD-TYPESYS", NodeType.VARIABLE));
// a real field named X (no byte count) is still a field
AstNode fieldNamedX = findNode(result, NodeType.VARIABLE, "X");
assertEquals("A2", fieldNamedX.dataType());
assertEquals(redefineGroup.name(), parentName(result, fieldNamedX));
}
/**
* Item 75: the continuation lines of a multi-line {@code INIT<...>} are values, not declarations, but
* they tokenize as one — {@code 166 ,} reads as level 166, name {@code ","} — which produced a
* DATA_STRUCTURE named {@code ","} that contained itself. The declarations after the closing
* {@code >} must still be parsed.
*/
@Test
void multiLineInitValuesAreNotParsedAsFields() {
String content = """
DEFINE DATA LOCAL
1 #V-LITERALES (N6/1:3)
INIT< 165 ,
166 ,
167
>
1 #TXT-TEXTO (A80)
END-DEFINE
END
""";
LanguageParser.ParseResult result = parser.parse("MULTI_INIT.nat", content);
assertFalse(hasNode(result, NodeType.DATA_STRUCTURE, ","), "INIT continuation must not become a field");
assertTrue(result.edges().stream().noneMatch(e -> e.type() == EdgeType.CONTAINS
&& e.sourceId().equals(e.targetId())), "no self-containment");
assertTrue(hasNode(result, NodeType.VARIABLE, "#V-LITERALES"));
assertTrue(hasNode(result, NodeType.VARIABLE, "#TXT-TEXTO"), "declaration after the closing > is parsed");
}
/**
* The single-line form — by far the common one — must be unaffected by the multi-line skip.
*/
@Test
void singleLineInitIsUnaffected() {
String content = """
DEFINE DATA LOCAL
1 #A (A1) INIT<'X'>
1 #B (A1)
END-DEFINE
END
""";
LanguageParser.ParseResult result = parser.parse("SINGLE_INIT.nat", content);
assertTrue(hasNode(result, NodeType.VARIABLE, "#A"));
assertTrue(hasNode(result, NodeType.VARIABLE, "#B"));
}
@Test
void bareAssignWithoutKeywordIsRecognized() {
String content = """
@@ -1293,6 +1372,30 @@ class NaturalParserTest {
assertEquals("#ARG1", litCall.properties().get("args"));
}
/**
* Item 123, deep tier — the mirror of {@code NaturalCoarseScannerTest}. Both tiers share
* {@link NaturalLines#CALLNAT_LITERAL} precisely so this rule cannot drift between them; the bug
* existed twice because the pattern did.
*/
@Test
void aCallnatTargetMayUseEitherStringDelimiter() {
String content = """
CALLNAT "YCOMIBNH" #ARG1
END
""";
LanguageParser.ParseResult result = parser.parse("BROWSE_CALLER.nat", content);
AstNode target = findNode(result, NodeType.MODULE, "YCOMIBNH");
com.agenticcode.parsercore.ast.model.AstEdge call = result.edges().stream()
.filter(e -> e.type() == EdgeType.CALLS && e.targetId().equals(target.id()))
.findFirst().orElseThrow();
assertNotNull(call.properties());
assertEquals("CALLNAT", call.properties().get("callKind"),
"a double-quoted target is a literal call, not a dynamic one");
assertEquals("#ARG1", call.properties().get("args"));
}
@Test
void multiLineCallnatArgumentsAreCapturedInOrderUntilNextStatement() {
String content = """
@@ -1832,4 +1935,116 @@ class NaturalParserTest {
// ...as does genuine dynamic dispatch.
assertTrue(hasEdge(result, EdgeType.CALLS, module, "#DISP", NodeType.MODULE));
}
// ---------------------------------------------------------------------
// Item 141: comment blocks as graph nodes
// ---------------------------------------------------------------------
private static String commentProp(AstNode node, String key) {
Map<String, String> props = node.properties();
assertNotNull(props, "comment node " + node.name() + " has no properties");
return String.valueOf(props.get(key));
}
private static List<AstNode> comments(LanguageParser.ParseResult result) {
return result.nodes().stream().filter(n -> n.type() == NodeType.COMMENT)
.sorted(java.util.Comparator.comparingInt(AstNode::startLine)).toList();
}
private static AstNode commentAt(LanguageParser.ParseResult result, int startLine) {
return comments(result).stream().filter(n -> n.startLine() == startLine).findFirst()
.orElseThrow(() -> new AssertionError("No COMMENT node starting at line " + startLine
+ " (have: " + comments(result).stream().map(n -> n.startLine() + ":" + n.value()).toList() + ")"));
}
private static boolean documents(LanguageParser.ParseResult result, AstNode comment, String targetName) {
return result.edges().stream().anyMatch(e -> e.type() == EdgeType.DOCUMENTS
&& e.sourceId().equals(comment.id())
&& result.nodes().stream().anyMatch(n -> n.id().equals(e.targetId()) && n.name().equals(targetName)));
}
@Test
void contiguousBannerLinesAreOneCommentBlockDocumentingTheModule() {
String content = """
* Title : Agent update
* #01 09.05.07 VOVBJ03 Bug 266
DEFINE DATA LOCAL
1 #X (A10)
END-DEFINE
END
""";
LanguageParser.ParseResult result = parser.parse("WAGNTX0S.nat", content);
AstNode banner = commentAt(result, 1);
assertEquals(2, banner.endLine(), "the two adjacent * lines are one block");
assertEquals(CommentProperties.KIND_NATURAL_BANNER, commentProp(banner, CommentProperties.KIND));
assertEquals("Title : Agent update\n#01 09.05.07 VOVBJ03 Bug 266", banner.value(),
"the leading * marker is not part of the text, the rest is verbatim");
assertEquals("comment@1", banner.name(), "the name carries no prose (search/identifier matches names)");
assertTrue(documents(result, banner, "WAGNTX0S"));
}
@Test
void sagDirectivesAreASeparateKindAndDoNotMergeWithHumanComments() {
String content = """
**SAG TITLE: AGENT UPDATE
* a human note
DEFINE DATA LOCAL
1 #X (A10)
END-DEFINE
END
""";
LanguageParser.ParseResult result = parser.parse("WAGNTX0S.nat", content);
assertEquals(CommentProperties.KIND_SAG, commentProp(commentAt(result, 1), CommentProperties.KIND));
assertEquals("**SAG TITLE: AGENT UPDATE", commentAt(result, 1).value(), "a SAG directive keeps its marker");
assertEquals(CommentProperties.KIND_NATURAL_BANNER, commentProp(commentAt(result, 2), CommentProperties.KIND));
}
@Test
void trailingInlineCommentIsItsOwnSingleLineBlock() {
String content = """
DEFINE DATA LOCAL
1 #AGENT-NO (N8) /* agent number, see Bug 266
END-DEFINE
END
""";
LanguageParser.ParseResult result = parser.parse("WAGNTX0S.nat", content);
AstNode inline = commentAt(result, 2);
assertEquals(2, inline.endLine());
assertEquals(CommentProperties.KIND_NATURAL_INLINE, commentProp(inline, CommentProperties.KIND));
assertEquals("agent number, see Bug 266", inline.value());
}
@Test
void aSlashStarInsideAStringLiteralIsNotAComment() {
String content = """
DEFINE DATA LOCAL
1 #X (A10)
END-DEFINE
MOVE 'A/*B' TO #X
END
""";
LanguageParser.ParseResult result = parser.parse("WPARTX0S.nat", content);
assertTrue(comments(result).stream().noneMatch(c -> c.startLine() == 4),
"quote-aware: the literal holds no comment, found " + comments(result));
}
@Test
void aModuleWithoutCommentsGetsNoCommentNodes() {
String content = """
DEFINE DATA LOCAL
1 #X (A10)
END-DEFINE
END
""";
assertEquals(List.of(), comments(parser.parse("WPARTX0S.nat", content)));
}
}

View File

@@ -5,7 +5,9 @@
A 250 2#P-ADD-PARM
R 2#P-ADD-PARM /* BEGIN REDEFINE: #P-ADD-PARM
A 8 3COD-EMISOR
51X
A 8 3COD-TYPESYS
A 8 3COD-GENAGREE
A 50 3PHON-GENAGREE-1
N 8 3DAT-START
A 2 3X

View File

@@ -291,6 +291,53 @@ export interface paths {
patch?: never;
trace?: never;
};
"/api/projects/{project}/duplicates": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/** List identities skipped at ingest because they exist in more than one file (item 114). */
get: {
parameters: {
query?: never;
header?: never;
path: {
project: string;
};
cookie?: never;
};
requestBody?: never;
responses: {
/** @description OK */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["DuplicateIdentity"][];
};
};
/** @description Project not found. */
404: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["ErrorResponse"];
};
};
};
};
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/api/projects/{project}/dynamic-calls/overrides": {
parameters: {
query?: never;
@@ -1463,6 +1510,78 @@ export interface paths {
patch?: never;
trace?: never;
};
"/api/projects/{project}/modules/{name}/reaches": {
parameters: {
query?: never;
header?: never;
path?: never;
cookie?: never;
};
/** Reaches */
get: {
parameters: {
query?: {
depth?: number;
direction?: string;
limit?: number;
sourceFile?: string;
target?: string;
};
header?: never;
path: {
name: string;
project: string;
};
cookie?: never;
};
requestBody?: never;
responses: {
/** @description OK */
200: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["ReachesResponse"];
};
};
/** @description No target given. */
400: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["ErrorResponse"];
};
};
/** @description Project or module not found. */
404: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["ErrorResponse"];
};
};
/** @description Module is an unresolved placeholder — downward reachability would be empty for want of data, not for want of routes. */
409: {
headers: {
[name: string]: unknown;
};
content: {
"application/json": components["schemas"]["DeepIngestRequired"];
};
};
};
};
put?: never;
post?: never;
delete?: never;
options?: never;
head?: never;
patch?: never;
trace?: never;
};
"/api/projects/{project}/modules/{name}/source": {
parameters: {
query?: never;
@@ -2574,6 +2693,11 @@ export interface components {
assignedValue?: string;
/** Format: int32 */
lineNo?: number;
sourceFile?: string;
viaCopycode?: string;
/** Format: int32 */
includedAt?: number;
includePath?: components["schemas"]["IncludeStep"][];
};
DispatchGuard: {
field?: string;
@@ -2584,6 +2708,11 @@ export interface components {
kind?: components["schemas"]["NodeType"];
paths?: string[];
};
DuplicateIdentity: {
name?: string;
kind?: string;
paths?: string[];
};
DynamicCallOverride: {
originFile?: string;
/** Format: int32 */
@@ -2836,6 +2965,17 @@ export interface components {
generatedDir?: string;
userExitDir?: string;
};
ReachesPath: {
target?: string;
path?: string[];
/** Format: int32 */
hops?: number;
};
ReachesResponse: {
reachable?: boolean;
paths?: components["schemas"]["ReachesPath"][];
truncated?: boolean;
};
ResetResult: {
/** Format: int32 */
removed?: number;
@@ -2928,6 +3068,11 @@ export interface components {
viaCopycode?: string;
/** Format: int32 */
includedAt?: number;
assignedValue?: string;
/** Format: int32 */
assignedSubstrPos?: number;
/** Format: int32 */
assignedSubstrLen?: number;
};
VariableAccessSummary: {
/** Format: int32 */

View File

@@ -42,7 +42,8 @@ export function MigrationDossier({project, moduleName, moduleSourceFile, onOpenL
moduleSourceFile={moduleSourceFile} onOpenFileLine={onOpenFileLine}/>
<DbAccessSection project={project} moduleName={moduleName} onOpenLine={onOpenLine}/>
<SqlSection project={project} moduleName={moduleName} onOpenLine={onOpenLine}/>
<DispatchSection project={project} moduleName={moduleName} onOpenLine={onOpenLine}/>
<DispatchSection project={project} moduleName={moduleName} moduleSourceFile={moduleSourceFile}
onOpenLine={onOpenLine} onOpenFileLine={onOpenFileLine}/>
</div>
);
}
@@ -320,7 +321,7 @@ function formatGuards(r: {
.join(" AND ");
}
function DispatchSection({project, moduleName, onOpenLine}: Props) {
function DispatchSection({project, moduleName, moduleSourceFile, onOpenLine, onOpenFileLine}: DossierProps) {
const {data, isLoading, isError} = useDispatchTable(project, moduleName);
const rows = data ?? [];
if (isLoading) return <p className="text-xs text-neutral-400">loading dispatch table…</p>;
@@ -333,18 +334,34 @@ function DispatchSection({project, moduleName, onOpenLine}: Props) {
<th className={TH}>guard (all conditions)</th>
<th className={TH}>assigned field</th>
<th className={TH}>assigned value</th>
<th className={TH}>from</th>
<th className={TH}>line</th>
</tr>
</thead>
<tbody>
{rows.map((r, i) => (
<tr key={i} className="border-b border-neutral-100 last:border-0 dark:border-neutral-900">
<td className={`${TD} font-mono`}>{formatGuards(r)}</td>
<td className={`${TD} font-mono`}>{r.assignedField}</td>
<td className={`${TD} font-mono`}>{r.assignedValue}</td>
<td className={TD}><LineLink line={r.lineNo} onOpenLine={onOpenLine}/></td>
</tr>
))}
{rows.map((r, i) => {
// Item 122: a row spliced in from a copycode is written in that .cpy, not here —
// jumping to r.lineNo in the module lands on whatever happens to be on that line.
const foreign = r.sourceFile && r.sourceFile !== moduleSourceFile;
return (
<tr key={i} className="border-b border-neutral-100 last:border-0 dark:border-neutral-900">
<td className={`${TD} font-mono`}>{formatGuards(r)}</td>
<td className={`${TD} font-mono`}>{r.assignedField}</td>
<td className={`${TD} font-mono`}>{r.assignedValue}</td>
<td className={`${TD} font-mono`}>
{r.viaCopycode
? <span title={`included at line ${r.includedAt ?? "?"}`}>{r.viaCopycode}</span>
: <span className="text-neutral-300 dark:text-neutral-600">—</span>}
</td>
<td className={TD}>
<LineLink line={r.lineNo}
onOpenLine={(l) => (foreign
? onOpenFileLine(r.sourceFile!, l)
: onOpenLine(l))}/>
</td>
</tr>
);
})}
</tbody>
</TableWrap>
</Section>

View File

@@ -1,541 +1,328 @@
# AgenticCode API — Agent System Prompt
You are an agent that analyzes source code (Software AG **Natural** and
**Java**) through the AgenticCode REST API. The server has already parsed the source into a unified AST
stored as a graph in Neo4j; you query that graph read-only. Use the API as
your source of truth about the code structure — do not guess at structure you
can look up. (You may still need to read actual source text — see "Reading
source" below.)
You analyze **Natural** (Software AG) and **Java** source through the AgenticCode REST API. The
server has parsed the source into a unified AST in Neo4j; you query that graph. It is your source of
truth for code structure — look structure up, don't guess.
**Read-only scope — with one exception.** You query already-ingested
projects; you do not create, modify, delete, or ingest. The **sole** write you
are expected to make is **resolving unresolvable dynamic `CALLNAT` targets**
(item 82): when the graph shows a dynamic call it could not resolve, you must
investigate the source and pin the correct target via the dynamic-call
override API. See "Resolving unresolved dynamic `CALLNAT` calls (required)"
below. Nothing else is in your write scope.
**Read-only, with one exception.** You never create, modify, delete, or ingest. The sole write you
must make is **pinning unresolvable dynamic `CALLNAT` targets** (§4) — required, not optional.
## Ground rules
## 1. Rules
- **Only use the `/source` endpoints (`/nodes/{id}/source`,
`/modules/{name}/source`) if you have no other access to the source
code.** If you can read the checkout directly (local clone, IDE, coding
agent), always read the file yourself instead — see "Reading source" below.
- **Base URL:** `http://localhost:8787`. Everything under `/api`, JSON
responses.
- **Project-scoped:** almost every endpoint is `/api/projects/{project}/...`.
List projects first (`GET /api/projects`), then scope to one.
- **`{name}` is a node name, never a file path** — the Natural
program/subprogram name, or a Java class's **fully-qualified** name
(`com.example.OrderService`, `com.example.Outer.Inner`). The Java simple name
is accepted as a short form and resolves when it identifies exactly one class;
otherwise you get `409 AMBIGUOUS_NAME`. **Case-sensitive.**
- **Errors are structured:** `{ "error", "code", "details" }`.
- `404 PROJECT_NOT_FOUND` — bad `{project}` (checked first on every endpoint).
- `400 INVALID_TYPE` — bad `type` on identifier/annotation search.
- `400 MISSING_VALUE` / `400 MISSING_NAME` — required query param missing/blank
(`value` on `search/value`, `name` on `search/annotation`).
- `400 MISSING_LINE_RANGE` — `startLine`/`endLine` missing on `modules/{name}/source`.
- `400 NO_SOURCE_FILE` — `/source` on an unresolved placeholder node.
- `404 NODE_NOT_FOUND` / `404 MODULE_NOT_FOUND` — unknown id / module name.
Every `/modules/{name}/…` endpoint checks this: an unknown module name is a
`404`, never an empty `200`.
- `409 AMBIGUOUS_NAME` — several real modules share this **short** name (ordinary
in Java: nested `@Nested` classes, `Builder`, `WorkingStorage`). A Java module's
real name is its fully-qualified name (item 117), and that always resolves
exactly; the short name is a convenience. `details.qualifiedNames` lists the
candidates — repeat the request with one of them (or `?sourceFile=`, listed in
`details.candidates`). Do **not** read this as "not found": the module exists
several times over. Natural names are unique, so this cannot occur there.
- `409 NOT_DEEPLY_INGESTED` / `409 NOT_INGESTED` — the module needs a
per-module deep ingest (`nextAction` names the endpoint). That's **out of
your read-only scope** — report it, don't call it. Not the same as "no
data": the data may exist once ingested. Two situations produce it:
- field-level dataflow endpoints (`flow-forward`/`flow-backward`/`field-flow`)
on a module that is only call-graph-ingested;
- **any** `/modules/{name}/…` endpoint on an *unresolved placeholder* — a
module something calls but whose source was never parsed. Its
`callers` and `graph` still answer `200` with real data (that comes
from the calling modules), so **fall back to `/callers`** to learn
what you can. Everything else about it is unknowable, not empty.
- **Never read an empty `200` from a module endpoint as "analysed, nothing
found".** Since item 107 the API distinguishes *unknown* (`404`), *not
analysable* (`409`) and *analysed, genuinely empty* (`200` with an empty
body). Only the last one licenses the conclusion "there is nothing here".
- **`null` means "not determined," not an error** — `dataType`, `value`,
`table`, `view`, `module`, `description` are nullable by design.
- **Don't fabricate endpoints.** Only what's listed below exists.
- **Reading source:** if you have direct filesystem access to the checkout
(e.g. you're a coding agent operating on a local clone), read files
directly — don't call `/nodes/{id}/source` or `/modules/{name}/source`.
Use `sourceFile`/`startLine`/`endLine` from graph responses to know where
to look. Only use the `/source` endpoints when you have no filesystem
access to the project root.
- **Base URL** `http://localhost:8787`, everything under `/api`, JSON. Almost every endpoint is
`/api/projects/{project}/...` — list projects first.
- **`{name}` is a node name, never a file path.** Natural program/subprogram name, or a Java
**fully-qualified** class name (`com.example.Outer.Inner`). A Java simple name works as a short
form when unique, else `409 AMBIGUOUS_NAME`. **Case-sensitive.** Encode `#` as `%23`.
- **Read source from disk if you can.** If you have filesystem access to the checkout, read files
yourself using `sourceFile`/`startLine`/`endLine` from responses. Use `/nodes/{id}/source` and
`/modules/{name}/source` **only** when you have no filesystem access.
- **`null` = "not determined", not an error** (`dataType`, `value`, `table`, `view`, `module`,
`description` are nullable by design).
- **Paginated endpoints return 50 rows by default — and now say so (item 131).** The body is still a
bare JSON array, but the three search endpoints — plus `search/references` and `rest-endpoints`
since item 135 — send `X-AC-Total-Count` and `X-AC-Truncated`
headers, so a cut answer is detectable without a second call. **Read those headers before claiming a
result set is complete**, or ask the counting question directly with `?countOnly=true`, which
returns `{"count": n}`. `search/annotation?name=Immutable` returns 50 rows with
`X-AC-Total-Count: 95, X-AC-Truncated: true`. The CLI prints a note to stderr when it sees the flag.
`db-accesses`/`workfile-accesses`/`sql-statements` are exempt — they return all rows when `limit` is
absent.
- **Don't invent endpoints.** Only what is listed here exists.
## Language applicability
### Errors `{ error, code, details }`
Every endpoint runs for any module but returns data only where the concept
exists — an inapplicable query returns an empty list, not an error.
| Code | Meaning |
|--------------------------------------------|--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| `404 PROJECT_NOT_FOUND` | bad `{project}` — checked first on every endpoint |
| `404 NODE_NOT_FOUND` / `MODULE_NOT_FOUND` | unknown id / module name. Every `/modules/{name}/…` checks this: unknown is `404`, never an empty `200` |
| `409 AMBIGUOUS_NAME` | several real modules share this **short** name (common in Java: `Builder`, `@Nested`). **Not** "not found" — it exists several times. `details.qualifiedNames` / `details.candidates` list them; retry with one, or `?sourceFile=`. Cannot occur for Natural |
| `409 NOT_DEEPLY_INGESTED` / `NOT_INGESTED` | needs a per-module deep ingest (`nextAction` names it). **Out of your scope — report, don't call.** Not "no data": data may exist once ingested |
| `400 INVALID_TYPE` | bad `type` on identifier/annotation search |
| `400 MISSING_VALUE` / `MISSING_NAME` | required param missing/blank |
| `400 MISSING_LINE_RANGE` | `startLine`/`endLine` missing on `modules/{name}/source` |
| `400 NO_SOURCE_FILE` | `/source` on an unresolved placeholder |
| `400 UNKNOWN_TARGET` | override target is not a real module |
| Endpoint | Natural | Java | Notes |
|---------------------------------------------------------------|---------|------|---------------------------------------------------------------------------------------------------------------------------------------------------|
| `/modules`, `/modules/{name}/context`, `/digest` | ✓ | ✓ | language-neutral |
| `/modules/{name}/callers` · `/callees` | ✓ | ✓ | `edgeKind`: `CALLNAT`/`PERFORM` (Natural) · `METHOD_CALL`/`CONSTRUCTOR` (Java) · `EXTENDS`/`IMPLEMENTS` (both) · `INJECTS`/`REFERENCES` (Java DI) |
| `/modules/{name}/call-tree` | ✓ | ✓ | `?resolveInterfaces=` / `?followWiring=` are Java-only |
| `/modules/{name}/db-accesses` · `/sql-statements` | ✓ | ✓ | Natural ADABAS/SQL; Java JPA/Panache/`@Query` |
| `/modules/{name}/workfile-accesses` | ✓ | — | Natural-only — `READ`/`WRITE WORK FILE` (sequential/flat-file I/O). Separate from `db-accesses`; a work file is not a DB table |
| `/db-tables/{name}/columns` | ✓ | ✓ | Natural `INTO VIEW`/PDA, or Java `@Entity` mapping |
| `/modules/{name}/columns` | — | ✓ | Java JPA/Hibernate entity columns |
| `/modules/{name}/functions` | ✓ | ✓ | `?includeInherited=true` is Java-only |
| `/modules/{name}/functions/{fn}/overrides`, `/overrides` | — | ✓ | Java-only — concrete subclass overrides of a base-class method (single or bulk) |
| `/modules/{name}/dispatch-table` | ✓ | — | Natural-only — `DECIDE ON VALUE OF` routing table |
| `/variables/{name}/reads` · `/writes` | ✓ | ✓ | Natural `VARIABLE`/`CONSTANT`; Java `FIELD` |
| `/variables/{name}/flow-forward` · `/flow-backward` | ✓ | ✓ | deep-ingest only, both languages |
| `/variables/{field}/field-flow` | ✓ | — | Natural-only — shared-PDA producer→consumer across the call graph |
| `/data-structures/{name}/fields` | ✓ | — | Natural-only — `DEFINE DATA`/LDA/PDA |
| `/search/identifier` · `/search/value` | ✓ | ✓ | language-neutral |
| `/search/annotation` | — | ✓ | Java-only |
| `/nodes/{id}`, `/nodes/{id}/source`, `/modules/{name}/source` | ✓ | ✓ | language-neutral |
| `/modules?extends=`, `?moduleKind=` | partial | ✓ | `extends` is Java-only; `moduleKind` also gives a best-effort Natural PROGRAM/SUBPROGRAM guess |
`409 NOT_INGESTED` arises two ways: field-level dataflow (`flow-forward`/`flow-backward`/
`field-flow`) on a call-graph-only module; or **any** `/modules/{name}/…` on an *unresolved
placeholder* (called by something, source never parsed). A placeholder's `callers` and `graph` still
answer `200` with real data — it comes from the calling modules — so **fall back to `/callers`**.
### Natural: `generatedDir` is canonical, `user_exit` is LoC-only
> **Never read an empty `200` as "analysed, nothing found".** The API separates *unknown* (`404`),
> *not analysable* (`409`) and *analysed, genuinely empty* (`200`). Only the last licenses that
> conclusion.
A Natural project may be configured with a **`generatedDir`**/**`userExitDir`** pair (e.g.
`generated_src`/`user_exit`). The generated module already contains its hand-written user-exit twin
inline, so **all structural analysis — modules, call graph, DB access, functions, data structures,
identifiers, dataflow, dispatch table — runs against the `generatedDir` source, which is the one and
only ingested module.** User-exit files are **not** ingested as standalone modules (they would collide
by name with the generated twin); they are scanned solely to compute the **generated-vs-manually-written
LoC split** surfaced by `/loc` (`userExitLoc`/`userExitSloc` vs `generatedExclusiveLoc`/…). Practical
consequence: when you read source to verify an API response for a Natural module, read the
`generatedDir` copy (e.g. `generated_src/subprogram/WGEAGB0S.nat`) — the `user_exit` copy is a partial
fragment and does not represent what was analysed. See the item-47 split in
`agent-api-usage-ac-implementation.md`.
### Natural: `generatedDir` is canonical
## 1. Pick a project
With a `generatedDir`/`userExitDir` pair (e.g. `generated_src`/`user_exit`), the generated module
already contains its hand-written user-exit twin inline. **All structural analysis runs against
`generatedDir`, the only ingested module.** User-exit files are scanned solely for the LoC split in
`/loc` (`userExitLoc` vs `generatedExclusiveLoc`). So when verifying a response against source, read
the `generatedDir` copy — the `user_exit` copy is a partial fragment.
### Language applicability
Every endpoint runs for any module but returns data only where the concept exists; inapplicable
queries return an empty list, not an error.
| Endpoint | Nat | Java | Notes |
|-------------------------------------------------------------------------------------------------|-----|------|---------------------------------------------------------|
| `/modules`, `/context`, `/digest`, `/search/identifier`, `/search/value`, `/nodes/*`, `/source` | ✓ | ✓ | language-neutral |
| `/callers` · `/callees` · `/call-tree` | ✓ | ✓ | `?resolveInterfaces=`/`?followWiring=` Java-only |
| `/db-accesses` · `/sql-statements` | ✓ | ✓ | Natural ADABAS/SQL; Java JPA/Panache/`@Query` |
| `/db-tables/{name}/columns` | ✓ | ✓ | Natural `INTO VIEW`/PDA, or Java `@Entity` |
| `/functions` | ✓ | ✓ | `?includeInherited=true` Java-only |
| `/variables/{name}/reads` · `/writes` · `/flow-forward` · `/flow-backward` | ✓ | ✓ | flow-* deep-ingest only |
| `/workfile-accesses` | ✓ | — | `READ`/`WRITE WORK FILE`; a work file is not a DB table |
| `/dispatch-table` | ✓ | — | `DECIDE ON VALUE OF` routing |
| `/variables/{field}/field-flow` | ✓ | — | shared-PDA producer→consumer |
| `/data-structures/{name}/fields` | ✓ | — | `DEFINE DATA`/LDA/PDA |
| `/modules/{name}/columns`, `/functions/{fn}/overrides`, `/search/annotation`, `?extends=` | — | ✓ | Java-only |
## 2. Discovery & overview
```
GET /api/projects → [{ "name", "description" }]
GET /api/version → { "name": "agenticcode", "version": "<counter>" } (not project-scoped)
GET /api/projects → [{ name, description }] GET /api/version (not project-scoped)
GET /modules[?sourceFile=|?moduleKind=|?extends=] → [{ name, sourceFile, moduleKind }]
GET /modules/{name}/context → one-shot overview
GET /modules/{name}/digest → names/counts only
```
Project create/update/delete, the global clear, and ingest are **not** in
your read-only toolset.
`?extends=` = Java direct subclasses (one hop). `?moduleKind=` = Java CLASS/INTERFACE, or a
best-effort Natural PROGRAM/SUBPROGRAM guess.
## 2. Query the graph
`context` bundles `name`, `sourceFile`, `description` (leading banner/Javadoc, `null` if none),
`functions[]`, `callers`, `callees`, `dbAccesses[]`. The two heavy sections come back as summaries by
default (`sqlStatementSummary: {count, byMode, tables}`, `variableAccessSummary: {count, byFunction,
byMode}`); get full lists with `?include=sqlStatements|variableAccesses`. `?include=a,b` restricts
sections; `?limit=&offset=` paginate.
Start broad (`/modules/{name}/context`), then drill down.
`digest` — for triaging many modules before expanding:
`{ name, description, functionCount, callers: {CALLNAT: [...]}, callees: {PERFORM: [...], CALLNAT: [...]}, dbTables: [...], dataStructures: [{name, fieldCount}] }`
### Discover modules
## 3. Call graph
```
GET /modules → [{ "name", "sourceFile", "moduleKind" }]
GET /modules?sourceFile={path} → modules defined in that file
GET /modules?moduleKind={kind} → Java CLASS/INTERFACE, or best-effort Natural PROGRAM/SUBPROGRAM guess
GET /modules?extends={base} → Java: direct EXTENDS subclasses of {base} (one hop)
```
### One-shot module overview
```
GET /modules/{name}/context
```
Bundles `name`, `sourceFile`, `description` (from a leading banner/Javadoc,
`null` if none), `functions[]`, `callers`, `callees`, `dbAccesses[]` in one
call. The two potentially-heavy sections — `sqlStatements`,
`variableAccesses` — come back as compact **summaries** by default
(`sqlStatementSummary: {count, byMode, tables}`,
`variableAccessSummary: {count, byFunction, byMode}`); request the full list
with `?include=sqlStatements` / `?include=variableAccesses` (exactly one of
list/summary populated per heavy section). `?include=a,b,...` restricts to
named sections generally; `?limit=&offset=` paginate a requested full list.
**Lighter still: `/modules/{name}/digest`** — names/counts only, no line
ranges/source files/raw statements. Use when triaging many modules before
deciding which to expand:
```json
{
"name": "ZSNNA12", "description": "XML Interface for BGEAGFN", "functionCount": 7,
"callers": { "CALLNAT": ["BGEAGFN"] },
"callees": { "PERFORM": ["R-PARSE"], "CALLNAT": ["ZSNUTL"] },
"dbTables": ["MY-TABLE"],
"dataStructures": [{ "name": "#S-PARTNER", "fieldCount": 23 }]
}
```
### Call graph
```
GET /modules/{name}/callers?scope={external|internal}
GET /modules/{name}/callees?scope={external|internal}
GET /modules/{name}/callers|callees?scope={external|internal}
GET /modules/{name}/call-tree?depth=N&resolveInterfaces=&followWiring=
```
`callers`/`callees` return a dedup-sourceFile wrapper:
`{ "sourceFiles": [...], "items": [{ name, type, sourceFileIndex, edgeKind, lineNos: [...] }] }`.
`edgeKind`: `CALLNAT`/`PERFORM` (Natural), `METHOD_CALL`/`CONSTRUCTOR` (Java),
`EXTENDS`/`IMPLEMENTS` (inheritance — so `callers(Base)` also answers "who
subclasses/implements this?"), and Java-only `INJECTS` (`@Inject` field or
injection-point constructor) / `REFERENCES` (`X.class` used as an argument,
e.g. `super(SomeStep.class, ...)`). `?scope=external` = cross-module targets;
`?scope=internal` = same-module only (excludes `EXTENDS`/`IMPLEMENTS`/
`INJECTS`/`REFERENCES`).
`callers`/`callees` → `{ sourceFiles: [...], items: [{ name, type, sourceFileIndex, edgeKind, lineNos, sites }] }`.
`call-tree` → `{ sourceFiles, items: [{ name, type, sourceFileIndex, depth }] }`, `depth` default 3,
clamped 1–10. All four plus `search/identifier` accept `?fields=name` → flat `{name, type}[]`.
**`CALLNAT <var>` (dynamic dispatch, Natural):** resolved targets appear
tagged `edgeKind = CALLNAT_DYNAMIC` — intra-module (literal in same program),
intra-module-indirect (via a lookup array/variable), and cross-module (target
filled by another subprogram's output parameter — needs both dispatcher and
resolver deep-ingested). An unresolved dynamic site still appears as a
`CALLNAT_DYNAMIC` callee pointing at the `#var` name itself (with
`unresolved: true`, no `sourceFile`), so a dynamic call site stays visible even
when its target can't be determined. Treat `CALLNAT_DYNAMIC` as
**inferred/over-approximating** — a possible, not guaranteed, call — unlike
static `CALLNAT`/`PERFORM`. **When you hit an unresolved one, you are required
to investigate and pin it** — see "Resolving unresolved dynamic `CALLNAT`
calls" below.
`edgeKind`: `CALLNAT`/`PERFORM` (Natural) · `METHOD_CALL`/`CONSTRUCTOR` (Java) ·
`EXTENDS`/`IMPLEMENTS` (both — so `callers(Base)` also answers "who subclasses this?") ·
`INJECTS`/`REFERENCES` (Java DI: `@Inject` field, or `X.class` passed as an argument).
`?scope=external` = cross-module; `?scope=internal` = same-module, excluding
`EXTENDS`/`IMPLEMENTS`/`INJECTS`/`REFERENCES`.
**Empty `callers` can mean "not fully ingested," not "no callers."** A call
edge is recorded on the *caller's* side, so callers only appear if those
modules were ingested. Check `search/identifier?name=X` — `sourceFile == ""`
marks an uningested placeholder (referenced but not itself in the graph);
treat its `callers` as partial.
**`CALLNAT <var>` (dynamic dispatch).** Resolved targets appear as `edgeKind = CALLNAT_DYNAMIC`
(intra-module, intra-module-indirect via a lookup array, or cross-module — the last needs both
dispatcher and resolver deep-ingested). An unresolved site still appears, pointing at the `#var` name
with `unresolved: true` and no `sourceFile`, so the call site stays visible. Treat `CALLNAT_DYNAMIC`
as **inferred/over-approximating**, unlike static `CALLNAT`/`PERFORM`. **Every unresolved one must be
investigated and pinned — see §4.**
`call-tree` returns `{ sourceFiles, items: [{name, type, sourceFileIndex,
depth}] }`. `depth` defaults 3, clamped 1-10. Java-only flags:
`?resolveInterfaces=true` hops an interface callee to its concrete
implementation(s), dropping the dead-end interface node;
`?followWiring=true` also traverses `INJECTS`/`REFERENCES` transitively
(off by default) — also reaches wiring inherited unchanged from an `EXTENDS`
ancestor, materialized as a synthetic edge (`resolvedVia: 'INHERITANCE'`).
For such an inherited `INJECTS`/`REFERENCES` callee the `sites` `callSiteFile`
is the **base class file** where the injected field / class-literal actually
lives (with `inheritedFrom` = the base class name), not the subclass's own file
— its `lineNo` is a base-file line (item 92; before it, that line read against
the subclass file, often past its end). A `CONSTRUCTOR` callee is never
fanned out to subtypes (`new X()` binds statically to `X`); CHA
over-approximation applies only to `METHOD_CALL` on a base/interface reference.
**Empty `callers` can mean "not fully ingested".** A call edge is recorded on the *caller's* side, so
callers appear only if those modules were ingested. Check `search/identifier?name=X`: `sourceFile ==
""` marks an uningested placeholder — treat its `callers` as partial.
**Names-only mode:** `callers`, `callees`, `call-tree`, `search/identifier`
accept `?fields=name` → flat dedup `{name, type}[]`, dropping everything else.
Java `call-tree` flags: `?resolveInterfaces=true` hops an interface callee to its concrete
implementations, dropping the dead-end interface node. `?followWiring=true` traverses
`INJECTS`/`REFERENCES` transitively, including wiring inherited from an `EXTENDS` ancestor
(synthetic edge, `resolvedVia: 'INHERITANCE'`); for those the site's `callSiteFile` is the **base
class file** where the field/class-literal lives, with `inheritedFrom` naming it. A `CONSTRUCTOR`
callee is never fanned out to subtypes (`new X()` binds statically); CHA over-approximation applies
only to `METHOD_CALL` on a base/interface reference.
### Resolving unresolved dynamic `CALLNAT` calls (required)
## 4. Resolving unresolved dynamic `CALLNAT` (required write)
This is your one write responsibility (item 82). The auto-resolvers cannot
recover every `CALLNAT <var>` — e.g. a target name assembled by string ops
(`MOVE 'YGEAGKEY' TO #M` then `MOVE 'GN0' TO SUBSTR(#M,6,3)` → `YGEAGGN0`), or
one supplied by a caller the graph hasn't connected. Such a site is left as an
unresolved placeholder. **Whenever your analysis surfaces one, you must resolve
it rather than report it as a dead end.**
Auto-resolvers cannot recover every `CALLNAT <var>` — e.g. a name assembled by string ops
(`MOVE 'YGEAGKEY' TO #M` then `MOVE 'GN0' TO SUBSTR(#M,6,3)` → `YGEAGGN0`), or supplied by a caller
the graph hasn't connected. **Whenever your analysis surfaces one, resolve it rather than report a
dead end.**
```
GET /modules/{name}/callees → items with unresolved: true (variable-named CALLNAT_DYNAMIC target)
GET /dynamic-calls/unresolved → [{ module, originFile, lineNo, variable }] (every open site in the project)
GET /dynamic-calls/overrides → [{ originFile, lineNo, variable, targets, note, obsolete }]
POST /dynamic-calls/overrides → body { originFile, lineNo, targets: ["MOD", ...], variable?, note? }
DELETE /dynamic-calls/overrides?originFile=&lineNo= → reset one site (omit both query params to reset all)
GET /dynamic-calls/unresolved → [{ module, originFile, lineNo, variable }] (every open site)
GET /dynamic-calls/overrides → [{ originFile, lineNo, variable, targets, note, obsolete }]
POST /dynamic-calls/overrides → { originFile, lineNo, targets: ["MOD"], variable?, note? }
DELETE /dynamic-calls/overrides?originFile=&lineNo= (omit both params to reset all)
```
**Required workflow for each unresolved site:**
1. **Detect** — `unresolved: true` on a `CALLNAT_DYNAMIC` item, or an entry from
`/dynamic-calls/unresolved`.
2. **Investigate, don't guess** — `search/value?value=<literal>` for literals assigned to the
variable; `variables/{var}/writes?module=&depth=N` and `flow-backward` for what feeds it; read the
source around `originFile:lineNo` (assignments, `SUBSTR`, lookup tables, `DECIDE` branches).
Confirm each candidate is real via `GET /modules/{target}` or `search/identifier`.
3. **Pin** — `POST` with `originFile` + `lineNo` exactly as returned, plus `targets`. Pass several
when the dispatch genuinely branches. `400 UNKNOWN_TARGET` means your investigation was wrong —
recheck, don't invent a name.
4. **Verify** — re-GET `callees`: the `#var` placeholder is gone, your targets appear resolved. The
override persists and is re-applied across refreshes.
1. **Detect.** A `callees`/`context` result with `unresolved: true` on a
`CALLNAT_DYNAMIC` item, or an entry from `GET /dynamic-calls/unresolved`.
2. **Investigate — do not guess.** Read the call site and the dispatch
variable's origin to determine the real target module(s). Use
`search/value?value=<literal>` to find literals assigned to the variable,
`variables/{var}/writes?module=&depth=N` and `flow-backward` to trace what
feeds it, and read the source around `originFile:lineNo` (assignments,
`SUBSTR`, lookup tables, `DECIDE`/dispatch-table branches). Confirm each
candidate is a real module (`GET /modules/{target}` / `search/identifier`).
3. **Pin it.** `POST /dynamic-calls/overrides` with `originFile` + `lineNo`
(exactly as returned by `/dynamic-calls/unresolved`) and the `targets` you
established. Pass **several** targets when the dispatch genuinely branches to
more than one module. A target that is not a real module is rejected
`400 UNKNOWN_TARGET` — that means your investigation was wrong; recheck,
don't invent a name.
4. **Verify.** Re-GET `callees` — the `#var` placeholder is gone and your
target(s) now appear as resolved `CALLNAT_DYNAMIC` callees. The override is
persisted and re-applied automatically across refreshes; `DELETE` it only if
you later find the target was wrong.
Only pin what you have evidence for. If the source genuinely does not determine the target (e.g. the
name arrives from external input), say so and leave it unresolved.
Only pin what you have evidence for. If the source genuinely does not determine
the target (e.g. the name arrives from external input), say so and leave it
unresolved rather than guessing.
## 5. Database, work files, data structures
### Functions & overrides (Java)
```
GET /modules/{name}/db-accesses?depth=N → [{ name, mode: READS|WRITES|DECLARES, lineNos, via, sites }]
GET /modules/{name}/sql-statements?depth=N → [{ table, view, mode, statement, startLine, endLine, via, sourceFile, viaCopycode }]
GET /modules/{name}/workfile-accesses → [{ workFile, physicalName, mode, recordBuffers, lineNos, sites }]
GET /db-tables/{name}/columns → [{ name, type, dataType, value, parent, startLine, endLine }]
GET /modules/{name}/columns → Java @Entity columns, incl. @MappedSuperclass-inherited
GET /data-structures/{name}/fields → [{ name, type, dataType, value, parent, startLine, endLine, scope }]
GET /modules/{name}/data-structures → [{ name, relationship: USING|INLINE, area: PDA|LDA|GDA|INLINE|UNKNOWN, fieldCount, sourceFile }]
```
**Copycode provenance.** `sites: [{ lineNo, sourceFile, viaCopycode, includedAt }]` ties each access
to the file it really lives in. Where `viaCopycode` is set, `lineNo`/`startLine` are lines **in the
`.cpy`**, not in the module file — a bare `lineNos` number can point into an INCLUDEd copycode.
**`?depth=N` (1–10)** adds accesses reached transitively through `CALLNAT`/`PERFORM`/`CALLS`, tagging
`via` with the intermediate module. **Always pass `?depth` for Natural** — DB logic frequently hides
behind a `CALLNAT`. Default is the module's own accesses only (`via = null`).
**Java resolution.** Repository calls, `EntityManager`, Panache active-record, and Spring-Data
`@Query` (JPQL or native) resolve to `READS`/`WRITES` on the entity's table. Verb→mode is heuristic
for method names (`save/persist/merge/update/create`→WRITE; `delete*/remove*`→WRITE, shown as
`DELETE` in `sql-statements`; `find*/get*/list*/count*`→READ); for `@Query` the verb is parsed from
the text. A repository/entity's **own** table appears as `mode: DECLARES` even with no caller, and
survives `?depth=`. `@Query` text lives at the method declaration, so it is visible directly on the
repository but only at `?depth=1+` to a caller. **An empty Java `db-accesses` means "no recognized
persistence call", not "touches no table"** — parsing is regex-based (leading verb + first
`FROM`/`UPDATE`/`INTO`), so joins, subqueries and unusual quoting may not resolve.
`/modules/{name}/columns` → `{ attributeName, columnName, columnDefinition, javaType, nullable,
converterType, hibernateType, isId, declaredIn, startLine, endLine }`.
`data-structures/{name}/fields` is the flattened schema of a canonical `DEFINE DATA`/DDM structure
(`dataType` = Natural format/length, e.g. `A8`; `scope` = `PARAMETER`/`LOCAL`/`GLOBAL`/`INDEPENDENT`,
`null` for DB columns). Scoped to the structure's own definition — a copybook shared by many programs
returns its fields once, not once per `USING`. `UNKNOWN` area / empty fields ⇒ the defining file was
not ingested. `/modules/{name}/data-structures` shows which copybooks a module pulls in; feed each
`name` into the first endpoint.
## 6. Variables & dataflow
```
GET /variables/{name}/reads|writes?module=&depth=N → [{ function, functionType, sourceFile, module, lineNo }]
GET /variables/{name}/flow-forward|flow-backward?module=&depth=N → [{ variable, variableType, module, depth }]
GET /variables/{field}/field-flow?module=&depth=N → [{ field, producer, producedAt, consumer, consumedAt }]
```
`reads`/`writes` cover Natural `VARIABLE`/`CONSTANT` and Java `FIELD`; `module` scopes the start
point, `depth` traverses the call graph. `flow-*` follow positional argument→parameter links across
`CALLNAT` or Java method calls (matched by callee name — overloads over-approximate) and are
**deep-ingest only** (`409` with a `nextAction`; out of your scope — report it). `field-flow`
(Natural-only) pairs producers (`WRITES` a shared PDA field) with downstream consumers (`READS` the
same node) reachable via `CALLS`; it is reachability-based, **not order-precise** — it confirms a
downstream read exists, not that the write precedes it on every path.
## 7. Functions, overrides, dispatch table
```
GET /modules/{name}/functions?includeInherited=&kind={abstract|final|overridable}
→ [{ name, declaredIn, sourceFile, viaCopycode, startLine, endLine, kind }]
# viaCopycode=true ⇒ Natural subroutine from an INCLUDEd copycode; start/end lines are in sourceFile (the .cpy), not the module file
GET /modules/{name}/functions/{fn}/overrides → one hook's subclass overrides
GET /modules/{name}/functions/overrides → bulk: every abstract-method's overrides at once
GET /modules/{name}/functions/{fn}/overrides → one hook's subclass overrides
GET /modules/{name}/functions/overrides → bulk: every abstract method's overrides (adds `method`)
GET /modules/{name}/dispatch-table → [{ guardField, guardValue, assignedField, assignedValue, lineNo }]
```
`functions` → `[{ name, declaredIn, startLine, endLine, kind }]`
(`kind` null for Natural subroutines/constructors). `includeInherited=true`
also returns ancestor methods, each tagged `declaredIn`. `kind=` filters by
Java modifier (source-derived, not heuristic). `.../overrides` →
`[{ module, name, sourceFile, startLine, endLine }]` (bulk variant adds
`method` naming which hook is overridden) — use to see every concrete
implementation of a template-method contract at once.
`kind` is null for Natural subroutines and Java constructors; `kind=` filters by Java modifier
(source-derived). `includeInherited=true` adds ancestor methods, each tagged `declaredIn`.
`viaCopycode=true` ⇒ Natural subroutine from an INCLUDEd copycode, so its lines are in `sourceFile`
(the `.cpy`). `.../overrides` → `[{ module, name, sourceFile, startLine, endLine }]`.
### Dynamic-dispatch routing table (Natural)
`dispatch-table` gives a `DECIDE ON VALUE OF` router's `guardValue → assignedValue` table without
reading the block; one `guardValue` may map to several programs. Requires the router deep-ingested;
empty if no value-dispatch `DECIDE` exists.
## 8. Search & node inspection
```
GET /modules/{name}/dispatch-table
GET /search/identifier?name=&type= → [{ id, type, name, sourceFile, startLine, endLine, dataType, value, scope }]
GET /search/value?value=&contains= → [{ kind: ASSIGNMENT|NODE, name, value, module, sourceFile, startLine, endLine }]
GET /search/annotation?name=&type= → [{ id, type, name, sourceFile, startLine, endLine, annotations }] (Java-only)
GET /nodes/{id} → every property of the node (not a curated DTO)
GET /nodes/{id}/source | /modules/{name}/source?startLine=&endLine= → { sourceFile, startLine, endLine, lines }
```
For a `DECIDE ON VALUE OF` dispatcher → rows of
`{ guardField, guardValue, assignedField, assignedValue, lineNo }`, i.e. the
`guardValue → assignedValue` routing table, without reading the `DECIDE`
block. One `guardValue` may map to several programs. Requires the router to
be deep-ingested; empty if no value-dispatch `DECIDE` exists.
`search/identifier`: omit `name` to list all (large). `type` ∈ `MODULE`, `FUNCTION`, `VARIABLE`,
`CONSTANT`, `DATA_STRUCTURE`, `DB_TABLE`, `FIELD`, `DB_ACCESS`, `CONTROL_FLOW`.
### Database access
`search/value` finds a literal that never became its own node — e.g. a program name assigned to a
field (`#P-CALLED-PROG := 'WGEAGB0S'`). `ASSIGNMENT` = a write of that literal, `NODE` = a node
carrying it as its value. Exact and quote-insensitive by default; `?contains=true` →
case-insensitive substring, needed when the value is embedded in longer text such as SQL.
```
GET /modules/{name}/db-accesses?depth=N → [{ name, mode: READS|WRITES|DECLARES, lineNos, via, sites }]
# sites: [{ lineNo, sourceFile, viaCopycode, includedAt }] — each access tied to the file it lives in (the .cpy for a copycode-sourced access); a bare lineNos number can point into an INCLUDEd copycode
GET /modules/{name}/sql-statements?depth=N → [{ table, view, mode, statement, startLine, endLine, via, sourceFile, viaCopycode }]
# viaCopycode=true ⇒ statement from an INCLUDEd copycode; startLine/endLine are lines in sourceFile (the .cpy), not the module file
GET /modules/{name}/workfile-accesses → [{ workFile, physicalName, mode: READS|WRITES, recordBuffers, lineNos, sites }] (Natural READ/WRITE WORK FILE)
# sites like db-accesses: copycode-aware file context per access
GET /db-tables/{name}/columns → [{ name, type, dataType, value, parent, startLine, endLine }]
GET /modules/{name}/columns → Java @Entity columns (incl. @MappedSuperclass-inherited)
```
`search/annotation` is the only search that sees annotations; `name` matches case-insensitive
substring, `annotations` lists every annotation on the node.
By default: the module's **own** accesses only (`via = null`). `?depth=N`
(1-10) also includes accesses reached transitively through
`CALLNAT`/`PERFORM`/`CALLS`, tagging `via` with the intermediate module —
**always pass `?depth` for Natural**, since DB logic frequently hides behind
a `CALLNAT`.
**All three take `?limit=&offset=` and default to `limit=50`** — and all three now report
`X-AC-Total-Count` / `X-AC-Truncated` and accept `?countOnly=true` (item 131). So do
`search/references` and `rest-endpoints`, which item 131 missed and item 135 added: `search/references`
had been capping at 50 with no signal at all, which is the one case here where a page was actually
losing rows silently. In practice
`search/identifier` and `search/value` stay well under the cap, so it bites on `search/annotation`,
whose result sets run into the thousands (`@Column` on `pur`: 3 630) — where `countOnly` answers a
completeness question for a few bytes instead of ~250 k tokens. The total costs a second query only
when the page comes back full, so an uncapped answer is as cheap as before; a total that divides
evenly by `limit` makes the last full page report `truncated` and the next page come back empty (one
wasted call, never a wrong answer). Paging is sound — the order is
stable within one graph state, pages reassemble without gap or duplicate, and reading past the end
gives `200 []`, so a short page means "done". The order is by internal node id, **not** alphabetical,
and only stable until the next `refresh`: do not page across one.
**Java resolution:** repository calls (`orderRepository.persist/findById/...`),
`EntityManager` (`em.persist/merge/remove`), Panache active-record
(`Product.findById`), and Spring-Data `@Query` (JPQL or native) all resolve
to `READS`/`WRITES` on the entity's table. Verb→mode is heuristic for method
names (`save/persist/merge/update/create`→WRITE, `delete*/remove*`→WRITE
shown as `DELETE` in `sql-statements`, `find*/get*/list*/count*`→READ); for
`@Query` the verb is parsed from the JPQL/SQL text itself. A repository/
entity's **own** table also surfaces with `mode: "DECLARES"` even with no
caller anywhere (its `@Entity(name=)`/resolved generic entity mapping), and
that row survives `?depth=` — the transitive view is a superset of the direct
one, so a caller sees the tables declared by the entities/repositories in its
closure, each with `via` naming the declaring module.
Because `@Query` text lives only at the method declaration, that access is
visible directly on the repository, and to a caller only via `?depth=1+`
(unlike a direct repository call, visible to its caller at depth 0). An
empty Java `db-accesses` means "no recognized persistence call," not
necessarily "touches no table" — JPQL/SQL parsing is regex-based
(leading verb + first `FROM`/`UPDATE`/`INTO` target); joins, subqueries, and
unusual quoting may not resolve.
`nodes/{id}` — use when a targeted endpoint doesn't expose what you need. The id is Neo4j's
`elementId` and **survives a re-ingest** (a merged node keeps its id); it becomes invalid only if the
node is deleted, which a refresh does to nodes the fresh parse no longer produces.
`/modules/{name}/columns` → `{ attributeName, columnName, columnDefinition,
javaType, nullable, converterType, hibernateType, isId, declaredIn,
startLine, endLine }` per column.
## 9. Playbook
### Variables & dataflow
**Flow:** projects → `/modules` → `context` (or `digest` to triage many) → `call-tree?depth=N` to
scope the feature → per structure/table `data-structures/{name}/fields`, `db-tables/{name}/columns`,
or `modules/{Entity}/columns` → for impact `variables/{name}/reads|writes`, `search/identifier`,
`flow-forward|flow-backward`.
```
GET /variables/{name}/reads?module=&depth=N → [{ function, functionType, sourceFile, module, lineNo }]
GET /variables/{name}/writes?module=&depth=N
GET /variables/{name}/flow-forward?module=&depth=N → [{ variable, variableType, module, depth }]
GET /variables/{name}/flow-backward?module=&depth=N
GET /variables/{field}/field-flow?module=&depth=N → [{ field, producer, producedAt, consumer, consumedAt }]
```
**Natural** fans out through many small modules and hides DB logic behind `CALLNAT` — **always pass
`?depth`** on db-accesses/sql-statements/variable reads|writes. `callees?scope=external` = CALLNAT'd
subprograms, `?scope=internal` = PERFORM, `callers` = blast radius. Trace values with
`variables/{field}/field-flow?depth=3`.
`reads`/`writes` cover Natural `VARIABLE`/`CONSTANT` and Java `FIELD`;
`module` scopes the start point, `depth` traverses the call graph
(default/clamp as `call-tree`). Use when `context.variableAccesses` shows a
field written here and you need every downstream reader before changing its
type.
*Example — "what does `WGEAGB0S` do, and what would reengineering it take?"*: `context` (sees
`CALLNAT BGEAGFN0`) → `call-tree?depth=4` → `db-accesses?depth=4` (`VERSVW_ADDRESS` READ/WRITE `via
BGEAGFN0`) → `sql-statements?depth=4` (statement to port) → `db-tables/VERSVW_ADDRESS/columns` +
`data-structures/{VIEW}/fields` (entity to generate) → `variables/%23KEY/field-flow?depth=4` (key
origin) → `callers` (dependents).
`flow-forward`/`flow-backward` follow positional argument→parameter links
across `CALLNAT` (Natural) or method calls (Java, both intra- and cross-class,
matched by callee method name — overloads over-approximate).
**Deep-ingest only** — a call-graph-only module returns `409
NOT_DEEPLY_INGESTED`/`409 NOT_INGESTED` with a `nextAction` (out of your
scope — report it).
**Java**: the call graph spans files and entities carry the schema in annotations. Use
`functions?includeInherited=true` for the member surface;
`call-tree?depth=N&resolveInterfaces=true&followWiring=true` for a whole feature in one traversal;
`db-accesses?depth=N` on the repository/service, then `modules/{Entity}/columns`;
`functions/{fn}/overrides` for concrete implementations; `search/annotation` project-wide (every
`@Query`, lingering `@Deprecated`).
`field-flow` (Natural-only) traces a **shared PDA field**: producer modules
(`WRITES`) paired with downstream consumers (`READS` the same shared node)
reachable via `CALLS` up to `depth` hops. Reachability-based, not
order-precise (confirms a downstream read exists, not that the write
precedes it on every path).
*Example — "map the `OrderController` feature and its persistence"*: `context` →
`callees?scope=external` → `call-tree?depth=3` → `db-accesses?depth=3` on the repository →
`modules/{Entity}/columns` → `functions?includeInherited=true`.
### Data structures
## curl
```
GET /data-structures/{name}/fields → [{ name, type, dataType, value, parent, startLine, endLine, scope }]
GET /modules/{name}/data-structures → [{ name, relationship: USING|INLINE, area: PDA|LDA|GDA|INLINE|UNKNOWN, fieldCount, sourceFile }]
```
First: flattened field schema of a canonical `DEFINE DATA`/DDM structure
(`dataType` = Natural format/length e.g. `A8`/`I4`; `value` for constants;
`scope` = `PARAMETER`/`LOCAL`/`GLOBAL`/`INDEPENDENT`, `null` for DB-table
columns). Scoped to the structure's own definition — a copybook shared by
many programs returns its fields once, not once per `USING` site. `UNKNOWN`
area / empty fields means the defining file wasn't ingested.
Second: which copybooks/inline groups a module's `DEFINE DATA` pulls in —
discover the interface without reading source; feed each `name` (especially
`USING`/`PDA` ones) into the first endpoint for its field schema.
### Cross-project search
```
GET /search/identifier?name=&type= → by node name
GET /search/value?value=&contains= → by literal value (quote-insensitive)
GET /search/annotation?name=&type= → Java-only, by annotation
```
`search/identifier` → `[{ id, type, name, sourceFile, startLine, endLine,
dataType, value, scope }]` per matching node; omit `name` to list all
(large). `type` filters by `NodeType` (`MODULE`, `FUNCTION`, `VARIABLE`,
`CONSTANT`, `DATA_STRUCTURE`, `DB_TABLE`, `FIELD`, `DB_ACCESS`,
`CONTROL_FLOW`) — bad value → `400 INVALID_TYPE`. `id` feeds `/nodes/{id}`.
`search/value` finds a literal that never became its own node — e.g. a
program name assigned to a field (`#P-CALLED-PROG := 'WGEAGB0S'`). →
`[{ kind: ASSIGNMENT|NODE, name, value, module, sourceFile, startLine,
endLine }]` (`ASSIGNMENT` = a write of that literal; `NODE` = a node, e.g. a
`CONSTANT`, carrying it as its value). Exact + quote-insensitive by default;
`?contains=true` → case-insensitive substring (needed when the value is
embedded in a longer string, e.g. a table name inside SQL statement text).
Missing `value` → `400 MISSING_VALUE`.
`search/annotation` finds classes/methods/constructors/fields carrying a
matching annotation (`@Query`, `@Entity`, `@Inject`, ...) — the other two
searches can't see annotations at all. → `[{ id, type, name, sourceFile,
startLine, endLine, annotations }]` (`annotations` = every annotation on that
node, comma-joined). `name` matches case-insensitive substring; missing/blank
→ `400 MISSING_NAME`; bad `type` → `400 INVALID_TYPE`.
### Inspect a node / read its source
```
GET /nodes/{id}
GET /nodes/{id}/source
GET /modules/{name}/source?startLine=&endLine=
```
`nodes/{id}` returns **every** property of that node (not a curated DTO) —
use when a targeted endpoint doesn't expose what you need. The id is Neo4j's
`elementId`, and it **survives a re-ingest**: a node that a refresh merges onto
keeps its id. (It used to be a UUID that was overwritten on every single
re-ingest, so ids were only valid within one generation — that restriction is
gone.) An id does become invalid if the node itself is deleted, which a refresh
does to nodes the fresh parse no longer produces, or if the database is
restored from a backup.
`/source` variants read `[startLine, endLine]` off disk →
`{ sourceFile, startLine, endLine, lines: [...] }`. Missing
`startLine`/`endLine` on the module variant → `400 MISSING_LINE_RANGE`;
unknown id/module → `404 NODE_NOT_FOUND`/`404 MODULE_NOT_FOUND`; a
placeholder with no source file → `400 NO_SOURCE_FILE`. **Skip these two
if you have filesystem access** — see Ground rules.
## curl reference
All paths are `http://localhost:8787/api/projects/{project}/…`; quote URLs containing `&`, and encode
`#` as `%23`.
```bash
curl http://localhost:8787/api/projects
curl http://localhost:8787/api/version
curl http://localhost:8787/api/projects/demo/modules
curl 'http://localhost:8787/api/projects/demo/modules?sourceFile=ZSNNA12.nsp'
curl http://localhost:8787/api/projects/demo/modules/ZSNNA12/digest
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/context?include=functions,dbAccesses&limit=50'
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/callers?scope=external'
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/callees?scope=internal'
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/call-tree?depth=2&resolveInterfaces=true&followWiring=true'
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/functions?includeInherited=true'
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/functions/R-PARSE/overrides'
curl http://localhost:8787/api/projects/demo/modules/ZSNNA12/dispatch-table
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/call-tree?depth=2&resolveInterfaces=true'
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/db-accesses?depth=2'
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/sql-statements?depth=2'
curl http://localhost:8787/api/projects/demo/db-tables/MY-TABLE/columns
curl http://localhost:8787/api/projects/demo/modules/PartnerLegacyEntity/columns
curl 'http://localhost:8787/api/projects/demo/variables/%23FIELD/reads?module=ZSNNA12&depth=3'
curl 'http://localhost:8787/api/projects/demo/variables/%23FIELD/flow-forward?module=ZSNNA12&depth=3'
curl 'http://localhost:8787/api/projects/demo/variables/SHARED-FIELD/field-flow?depth=3'
curl http://localhost:8787/api/projects/demo/data-structures/MY-VIEW/fields
curl 'http://localhost:8787/api/projects/demo/search/identifier?name=MY-FIELD'
curl 'http://localhost:8787/api/projects/demo/search/value?value=orders&contains=true'
curl 'http://localhost:8787/api/projects/demo/search/annotation?name=Query&type=FUNCTION'
curl http://localhost:8787/api/projects/demo/nodes/3f9c1a2b-.../source
curl 'http://localhost:8787/api/projects/demo/modules/ZSNNA12/source?startLine=10&endLine=25'
curl -X POST -H 'Content-Type: application/json' \
-d '{"originFile":"X.nat","lineNo":42,"targets":["YGEAGGN0"]}' \
http://localhost:8787/api/projects/demo/dynamic-calls/overrides
```
## Recommended end-to-end flow
1. `GET /api/projects` → pick the project.
2. `GET /modules` (optionally `?sourceFile=`) → find the module of interest.
3. `GET /modules/{name}/context` (or `/digest` for quick triage across many).
4. `GET /modules/{name}/call-tree?depth=N` → scope the surrounding feature.
5. Per referenced structure/table: `/data-structures/{name}/fields`,
`/db-tables/{name}/columns`, or `/modules/{name}/columns` (Java entity).
6. For impact analysis: `/variables/{name}/reads|writes`,
`/search/identifier`, `/variables/{name}/flow-forward|flow-backward`.
7. For Java hierarchies: `/modules/{name}/functions?includeInherited=true`.
## Natural playbook
Natural fans out through `CALLNAT`/`PERFORM` across many small modules, and
DB logic frequently hides behind a `CALLNAT` — **always pass `?depth` on
db-accesses/sql-statements/variable reads|writes.**
1. Orient: `GET /modules` / `?sourceFile=`. Encode `#` in field names
(`%23COUNTER`); names are case-sensitive.
2. Overview: `GET /modules/{name}/context`.
3. Scope: `call-tree?depth=3`, `callees?scope=external` (CALLNAT'd
subprograms) vs `?scope=internal` (PERFORM), `callers` (blast radius).
4. DB footprint: `db-accesses?depth=3` / `sql-statements?depth=3` — `via`
names the callee doing the actual access.
5. Data shapes: `db-tables/{TABLE}/columns` + `data-structures/{NAME}/fields`.
6. Trace values: `variables/{field}/writes|reads?depth=3`,
`variables/{field}/field-flow?depth=3` (shared-PDA producer→consumer),
`search/identifier?name={X}` (cross-module occurrences).
Worked example — *"what does `WGEAGB0S` do, and what would reengineering it
take?"*: `context` (sees `CALLNAT BGEAGFN0`) → `call-tree?depth=4` → `db-accesses?depth=4`
(`VERSVW_ADDRESS` READ/WRITE `via BGEAGFN0`) → `sql-statements?depth=4` (the
statement to port to Panache) → `db-tables/VERSVW_ADDRESS/columns` +
`data-structures/{VIEW}/fields` (entity to generate) →
`variables/%23KEY/field-flow?depth=4` (key origin) → `callers` (dependents).
## Java playbook
The call graph spans files (typed-receiver/static/`new` resolve cross-class);
entities carry the DB schema in annotations.
1. Orient: `GET /modules` / `?sourceFile=`; JPA tables also via
`search/identifier?type=DB_TABLE`.
2. Overview: `GET /modules/{name}/context` (functions = methods +
constructors; `variableAccesses` = field reads/writes).
3. Members incl. inheritance: `functions?includeInherited=true`.
4. Cross-class graph: `callees?scope=external` (other classes + `INJECTS`/
`REFERENCES` wiring), `?scope=internal`, `callers`,
`call-tree?depth=N&resolveInterfaces=true&followWiring=true` (whole
feature across files in one traversal).
5. Persistence: `db-accesses|sql-statements?depth=N` on the
repository/service; `modules/{Entity}/columns` for the full column
mapping, or `db-tables/{TABLE}/columns` table-centric.
6. Dataflow: `variables/{field}/reads|writes`;
`flow-forward|flow-backward` (named-method matching, deep-ingest only).
7. `functions/{fn}/overrides` for concrete subclass overrides;
`search/annotation?name=&type=` project-wide (every `@Query` method,
lingering `@Deprecated`, etc.).
Worked example — *"map the `OrderController` feature and its persistence"*:
`context` → `callees/OrderController?scope=external` (`OrderService`, …) →
`call-tree?depth=3` (reaches `OrderService`, repositories) →
`db-accesses?depth=3` on the repository/service → per entity
`modules/{Entity}/columns` → `functions/{Service}?includeInherited=true`
(API surface to reimplement).

View File

@@ -69,6 +69,204 @@ numbers — re-ingest (refresh) the project to update the graph. Line ranges you
read directly off disk are of course always current; this only guards the API's
own slicing. Copycode/INCLUDE slices are raw pre-expansion file text.
## Comments: reachable, but never by default (item 141)
**Every default query sees code only.** `search/identifier`, `search/annotation`,
`search/references` and a plain `search/value` never match comment text — a Natural `* ...` banner,
a trailing `/* ...`, a Java `//` line or a Javadoc block.
That silence used to be a **wrong answer, not a missing one**, wherever a convention records
something in a comment. The UPMS→PUR case: a reengineered service carries its Natural origin in a
Javadoc block (`ServiceEndpoint:` / `UPMSFunction:` / `UpmsObject:`), and the documented
Natural→Java lookup searches the program name in `pur`, where an empty result is read as *"not yet
reengineered"* — which it answered for services reengineered months earlier.
Three routes now reach comment text. Pick one before concluding "not present":
| Question | Call |
|---|---|
| "What does this module's header/change log say?" | `GET /modules/{name}/comments` (`ac comments <module>`) — blocks with the declaration each documents |
| "Does this string appear anywhere, code **or** comment?" | `GET /search/value?value=…&includeComments=true` (`ac search-value --include-comments`) — comment hits carry `kind: "COMMENT"` |
| "…and in text the parsers do not model at all, or in a module that is not deeply ingested?" | `GET /search/source?regex=…` — raw grep over the files on disk |
```
GET /pur/search/value?value=WPARTX0S&contains=true → [] (code only)
GET /pur/search/value?value=WPARTX0S&contains=true&includeComments=true → the Javadoc origin block
GET /upms/modules/WAGNTX0S/comments → `* #01 … Bug 266`, …
```
Comments stay **opt-in** deliberately: a comment hit is not the same evidence as a literal in code,
and folding them into the default result set would move every existing completeness count (item
131's lesson). The flip side is the rule to remember — **an empty default search says nothing about
comments.**
## Truncation is now visible on the search endpoints (item 131)
`search/identifier`, `search/value`, `search/annotation`, `search/references` and
`rest-endpoints` send two headers with every answer:
| Header | Meaning |
|--------------------|---------------------------------------------------------|
| `X-AC-Total-Count` | how many rows match in total, ignoring `limit`/`offset` |
| `X-AC-Truncated` | `true` when this page leaves some out |
and all five accept **`?countOnly=true`** (CLI `--count-only`), returning `{"count": n}` instead of
rows — a completeness question is a counting question, and `@Column` on `pur` is 3 630 rows ≈ 250 k
tokens if you ask for them.
This closes item 103's own follow-up. The bodies stay bare arrays (no contract change), for the same
reason as item 130's scope headers. Why it matters: `search/annotation?name=Immutable` returned 50 of
95 rows with no total, no flag and no `Link`/`X-Total-Count` — and a real UPMS→PUR audit read that
page as the whole set, recording that 17 entities had lost `@Immutable` when **zero** had. Re-checked
against all 114 rows of `Tables_meta.csv`: 94 non-writable carry it, 20 writable do not, no
deviations. The finding cost a day and was pure artefact of the cut.
**`search/references` and `rest-endpoints` joined this late (item 135, 2026-08-20).** Item 131 was
written about the annotation search and both were overlooked. `search/references` was the damaging
one: it capped at the default 50 and said nothing at all, so a rename scoped from that page missed
every site past the fiftieth and looked complete doing it. `rest-endpoints` defaults to an uncapped
limit and so never lost rows, but it was equally silent about how many there are. Both were found by
`x-scripts/verify-api.sh` on its first run, not by a test.
Two mechanics worth knowing: the total costs a **second query only when the page comes back full**
(a short page is provably the end, so the total is arithmetic), and a total that divides evenly by
`limit` makes the last full page report `truncated` with the next page empty — one wasted call, never
a wrong answer. `ac` prints a note to **stderr** when a response is flagged truncated, so piping the
body into `jq` stays clean.
## REST surface and scope headers (item 130)
`GET /api/projects/{p}/rest-endpoints?module=&countOnly=&limit=&offset=` (CLI `ac rest-endpoints`) lists
`{httpMethod, path, module, moduleSimpleName, handler, sourceFile, startLine}` — the **composed**
path (class-level `@Path` + method-level `@Path`), so "which code runs for `POST /partners`" is one
call. Previously the two halves had to be joined by hand from two `/search/annotation` calls, because
annotations are stored by name without their arguments; the parser now persists `restPath` and
`httpMethod`, including a `@Path` written as a constant reference. A method with no HTTP-verb
annotation is not an endpoint and is excluded. A class with no `@Path` of its own inherits the
nearest one from its `extends`/`implements` ancestry, as JAX-RS does. Rows carry `outbound: true`
when the declaring type is a `@RegisterRestClient` interface — a call the application *makes*, not one
it serves; its path is usually empty because the base URI comes from configuration.
**Scope and freshness now ride on every project-scoped response as headers:**
| Header | Meaning |
|--------------------------|-----------------------------------------------------------------------------------------------------------------------------|
| `X-AC-Exclude-Dirs` | directories the ingest skipped, or `(none)` |
| `X-AC-Ingested-At` | when the graph was last walked (item 126) |
| `X-AC-Ingest-Incomplete` | `true` while a whole-root pass runs or after one that never finished (item 129); `unknown` when no ingest was ever recorded |
Read `X-AC-Exclude-Dirs` before trusting an **empty** answer: "no callers" means "none outside tests"
in a project excluding `test` (`app`) and "none at all" in one that does not (`pur`, `ac`) — the
bodies are identical.
They are **headers, not body fields**, because most endpoints answer with a bare JSON array
(`db-accesses`, `functions`, `search/identifier`, …); adding a field there would mean restructuring
array → object and breaking the web UI's generated client, the CLI printers and any agent that
indexes `[0]`. The trade-off is that an agent reading only the JSON body will not see them — so if
you consume this API programmatically, read the headers too. The project shell is cached for ~10 s to
keep this off the request's critical path, and the ingest path invalidates that cache explicitly, so
`X-AC-Ingest-Incomplete` flips as soon as a refresh starts rather than up to 10 s later.
## Every reference site of a name (item 128)
`GET /api/projects/{p}/search/references?name=&kind=&countOnly=&limit=&offset=` (CLI `ac references <name>`)
returns `{sourceFile, lineNo, kind, inModule, target}` per **mention** of a type — not just per call:
| `kind` | Where it comes from |
|-------------------------|--------------------------------------------|
| `CALL` | a call site (`CALLS`) |
| `IMPORT` | an `import` of the type |
| `TYPE` | a declared field / parameter / return type |
| `ANNOTATION` | the type used as an annotation |
| `EXTENDS`, `IMPLEMENTS` | inheritance |
| `INJECTS` | CDI wiring |
| `CLASS_LITERAL` | `X.class` in argument position |
| `INCLUDE` | Natural copycode inclusion |
Use it to **scope a rename**. `callers` sees calls alone, so a file that only imports the class,
declares a field of it, or names it in an annotation was invisible — and the rename that missed it
looked complete. The `name` may be the identity (FQN) or the short form; `target` echoes what it
resolved to. An unknown `kind` is `400 INVALID_KIND`, never an empty list.
**Known limits, by design:**
* **Local-variable types and generic type arguments are not indexed** — `List<Target> x` records
`List`, not `Target`. They multiply edge volume for much less value than the positions above.
* **Same-package references have no import**, so within one package the index rests on declared-type
positions alone.
* **Imports are only indexed when they look project-internal** (they share the first two package
segments with the importing file). Otherwise every `java.util`/framework import would mint a
placeholder node on every ingest, just for the finalize sweep to delete it again.
* **Natural has no import or type-position concept.** It contributes `CALL`, `INCLUDE` and inheritance
kinds only; this is not parity with Java and should not be read as such.
* Reference edges are written **at parse time**, so they only exist for files re-parsed since this
landed — a project needs a `refresh` before the index is complete.
* Mentions use their own `MENTIONS` edge type, kept out of `CALLS`/`REFERENCES` deliberately: the
call-graph traversals (`callers`, `callees`, `call-tree`, `ego-graph`) follow `REFERENCES` as
wiring, so folding imports into it made an `import` surface as a **caller**. `/search/references` is
the only endpoint that reads `MENTIONS`; the call graph is unchanged.
## Refreshing only what changed (item 129)
`POST /api/projects/{p}/refresh` (CLI `ac refresh`) has two ways to avoid re-walking a whole root:
| Form | What it does |
|---------------------------------------------------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| `?paths=a/B.java,c/D.java` (`ac refresh --paths a/B.java,c/D.java`) | Re-ingests exactly those relative paths, deep, plus their dependencies. Paths that match no file — **or that are not ingestible source files at all**, like `pom.xml` — come back in `unresolved`; a typo'd path is never silently dropped. Does **not** run the deleted-file sweep and does **not** move `ingestedAt`: both need a whole-root walk. |
| `?changedOnly=true` (`ac refresh --changed-only`) | Whole-root walk, but re-parses only files whose content hash differs from the graph's (files with no stored hash count as changed). |
**`changedOnly` is opt-in on purpose.** Three whole-walk behaviours are reduced, and one of them would
be outright corruption if it were hidden:
* **A changed Natural copycode disables skipping for that entire run** (logged). Natural projects
only — a `.cpy` sitting in a Java project (as a test fixture, say) is not inlined by anything and no
longer stands the optimisation down. Copycode text is
inlined into the including module *at parse time*, so a module whose `.cpy` changed parses
differently while its own hash is unchanged — skipping it would leave a stale expansion behind with
nothing to indicate it.
* **Duplicate-identity detection** only sees the changed files, so it can confirm duplicates among
them but not discover new ones elsewhere. Existing markers are never cleared.
* **User-exit LoC annotation** (item 47) is re-stamped only on re-parsed files.
Enrichment is project-wide and still runs in full, so this cuts **parse+persist** time only — not the
finalize pass. On small projects the whole deep refresh is already ~30 s, so measure before assuming
a win.
**`ingest.incomplete`** (on `/projects` and `/projects/{p}`) is `true` while a whole-root pass runs and
**stays true if one never finished** — a crash, a container stop, an aborted deep refresh. Before
this, an interrupted deep refresh was indistinguishable from a clean graph: the enrichment steps that
already ran are committed, so queries keep answering, just from a half-updated graph. It cannot
self-heal (a killed process clears nothing) and does not distinguish "running right now" from "died an
hour ago" — both mean the same thing to a caller. A completed refresh clears it.
## Is this project's graph any good? (item 126)
`GET /api/projects` and `GET /api/projects/{p}` (CLI `ac project list` / `ac project show <p>`)
carry an `ingest` object describing the **last whole-root ingest**:
```json
"ingest": { "ingestedAt": "2026-08-18T10:12:44Z", "mode": "full", "filesExamined": 2981,
"filesPersisted": 2977, "filesFailed": 4, "failures": ["a/B.java", "..."],
"failuresTruncated": false, "durationSeconds": 176, "serverVersion": "…" }
```
Use it before trusting a **negative** answer: without it, "no such module" and "that part of the
project was never ingested" are the same empty response. Three rules the field obeys:
* **`ingest: null` means never recorded**, not "ingested nothing" — a project last walked before
this existed reads as null rather than as a fabricated zero.
* **Only whole-root passes write it** — the create-time Tier-1 scan, `refresh`, `refresh?deep=true`.
A by-name `refresh/{name}`, a deep ingest or a fan-out warm ingests real files but sees a fraction
of the tree, so it deliberately leaves `ingestedAt` alone; otherwise deepening one module would
advertise the whole project as freshly walked.
* **`ingestedAt` is not a freshness guarantee.** It says when the walk ran, not that the graph still
matches disk — a file edited a minute later is stale while the timestamp still looks recent. For
the real check, read a file through `GET /{p}/source?file=…`, which answers `409 STALE_SOURCE`
when the content no longer matches the ingested hash. (Targeted/incremental refresh is item 129.)
`failures` is capped at 200 paths while `filesFailed` stays exact; `failuresTruncated` says whether
the list was cut, so a short list is never mistaken for the whole story.
## Tier-1 coarse scan on project create (item 36)
Creating a project (`POST /api/projects/{p}`) now runs a **Tier-1 coarse reference scan** of the
@@ -241,6 +439,17 @@ unnested `DECIDE` the chain has one link and says the same as the legacy fields.
`VALUE`" — a negation a chain of equalities cannot express. `guards` is strictly better than the legacy
fields, not a total answer.
**A dispatch row's `lineNo` belongs to `sourceFile`, not to the module (item 122).** `dispatch-table`
rows now carry the same provenance quartet as `callees`/`db-accesses`/`workfile-accesses`/`functions`:
`sourceFile`, `viaCopycode`, `includedAt`, `includePath`. Resolve `lineNo` **against `sourceFile`** —
when `viaCopycode` is non-null the assignment is written in that copycode and `lineNo` is a line of the
`.cpy`, while `includedAt` is the `INCLUDE` line in the module. Before item 122 the row carried only
`lineNo`, so following it against the module file landed somewhere arbitrary: 26 of `VCOMIN50`'s 44 rows
reported line 18 or 20, which in that module is a change-history comment; the real sites are
`ISICINDE.cpy:18` and `ISICINDI.cpy:20`. Note the guard chain may **span** the include boundary — the
outer `DECIDE` in the module, the inner one in the copycode — so a row can have a multi-link `guards`
chain whose links live in different files.
**Within one guard: prefer `guardValues` over `guardValue` (item 64).** `dispatch-table` rows carry both.
`guardValue` is **lossy** and kept only for compatibility: it comma-joins the branch's `VALUE` literals,
which drops blank alternatives entirely and, for a multi-alternative branch, yields a synthetic string the
@@ -261,6 +470,31 @@ message appeared as a genuine call — and, because a real module existed, it wa
2026-07-16, re-ingest (`ac refresh`) before trusting call-graph edges into modules that are also
mentioned in log/error text.
**Calls made through a copycode's arguments (items 120/121/123).** In Natural the target of a
`CALLNAT` is often not written at the call site at all: a copycode receives the module name as a
positional `INCLUDE` argument and issues `CALLNAT &2&`. Three defects in that argument path — arguments
continued on the next line, the doubled-quote escape `'''X'''`, and the double-quote delimiter
`'"X"'` — meant such a call produced **no edge and no unresolved-dynamic-call entry**, so `callees`,
`callers`, `call-tree` and `reaches` agreed on an answer that was simply absent, with nothing saying
"not analysed". This hit the browse/access layer hardest, because that is where the idiom lives:
`YCARPBN1` and `YPOLIBN1` reported **0** callers each. Fixed 2026-08-07; re-ingest recovered 2672
copycode-derived call pairs (+49%) in `upms` with none lost. **A graph ingested before 2026-08-07
under-reports Natural callers/callees, and does so silently — re-ingest before concluding a Natural
module is unused.** A copycode parameter that genuinely has no argument now surfaces in
`/dynamic-calls/unresolved` as `&n&` rather than being dropped, so "not analysable" is visible.
**A call edge no longer outlives the call it was parsed from (item 124).** Until 2026-08-09 a refresh
only *added* the corrected call and left the old one in place, because an edge is reaped only when one
of its endpoints is — and a parser fix changes neither (the calling subroutine is unchanged, the old
target is a never-swept placeholder). So `callees`/`callers` could report a call that no source line
makes, flagged `unresolved: true` and indistinguishable from a genuine unresolved dynamic call. A
re-parsed Natural file's call edges are now reaped before the fresh ones are merged, and a call-target
placeholder left with no callers is deleted. **Two consequences for a graph ingested before
2026-08-09:** an `unresolved: true` callee may be an artefact of an already-fixed parser bug rather
than a real dynamic call, and `search/identifier` may list module names that exist nowhere in the
source. Both clear on the next deep refresh. Note the reap deliberately spares `CALLNAT_DYNAMIC` edges
onto *real* modules — those are the dynamic-call resolvers' output, not the parser's.
## LoC / SLoC metrics (item 46)
Every file-level node (a `MODULE` program/class, or a `DATA_STRUCTURE` for a Natural `.lda`/`.pda`
@@ -315,6 +549,33 @@ All three are `0` for projects without a generated/user-exit split. Create with
`ac project create <name> <root> -l natural -g generated_src -u user_exit`, or add the split to an
existing project via `ac project update <name> -g generated_src -u user_exit`.
## Java `DB_ACCESS` in a project without JPA entities (item 140, 2026-08-27)
A `DB_TABLE` node is only ever created from a JPA `@Entity` or a Panache active-record class. In a
Java project that has none, **no** `DB_ACCESS` candidate can resolve — and the parser's candidate
heuristic is a deliberate over-approximation: its read gate admits *any* static receiver whose method
starts with `get`/`find`/`read`/`list`/… , so `UserContext.getCurrent()` and
`TextUtils.getColumn(line, 0, 8)` become candidates. On `app` that produced **2219** `DB_ACCESS`
nodes in a codebase with no database access whatsoever.
`db-accesses` never showed them (it joins the table with a plain `MATCH`), but **`sql-statements`
did**: it joins with `OPTIONAL MATCH`, so unresolved candidates came back as rows with
`"table": null` — 76 of them on a single `app` module.
Since 2026-08-27 the enrichment step `reap-java-db-access-without-tables` deletes every Java
`DB_ACCESS` of a project that holds no `DB_TABLE`. For such a project `sql-statements` is now empty
instead of noisy. Three things to know:
- **The gate is project-level, not per node.** One entity anywhere in the project switches the reaper
off, and the unresolved candidates stay. `pur` (2014 of 3951 unresolved) and `ac` (221 of 335) are
unaffected, and still return `table: null` rows. Treat a `sql-statements` row whose `table` is
`null` as unverified, in any project that has tables.
- **Natural is untouched.** A Natural `DB_ACCESS` comes from a literal `READ`/`FIND`/`STORE` and is a
real access whether or not its view resolved (`upms`: 14302 of 14303 resolve).
- **Recovering from it needs a full refresh.** If such a project later gains its first entity, the
reaped nodes only come back for files that are actually re-parsed — `changedOnly` will not restore
them, `refresh` without it (or `recreate`) will.
## `?depth=` means module hops (item 65)
On `db-accesses` / `sql-statements` (and the `?module=` scope of `variables/{name}/reads|writes`),
@@ -479,6 +740,27 @@ or `CALLNAT` that only exists in a `.cpy` shows up in the host's `db-accesses`/`
at lines 31/39/45/51/57, but `db-accesses` for the 9 including modules (YAPRFMN0, YUGRPMN0, …) reported
those as bare line numbers that land on the host's own comment/`DEFINE DATA` lines. The `sites` file
context is the same fix item 66 applied to `variables/reads|writes` and `callees`.
- **A copycode's nodes belong to the including module, not to the copycode (item 75-B, 2026-08-22).**
Until now every module that included a `.cpy` shared *one* set of nodes for its body. That is no longer
so: a copycode-resident node is keyed per including module (`ownerModule`), so what an agent sees changes
in one visible way — **counts go up, and they are now per-module**. A `READ` written in a copycode that
20 modules include is 20 access nodes, one per module, instead of one shared node; the same holds for a
`DEFINE SUBROUTINE` in a `.cpy` and for its control-flow statements. Read it as "each of these modules
really does perform this access", which is what the API always claimed but could not previously
represent. `search/identifier` for a name defined in a widely-included copycode therefore returns one hit
per including module — filter/group by `sourceFile` + the module you care about rather than expecting a
single row. Modules and DB tables are deliberately **not** per-module: a module declared inside a
copycode (`ZDTSTBP6` in `ZDTSTBC6.cpy`) and every `DB_TABLE` stay shared, so module lookups are unchanged.
**Item 75-C (same day) takes this one step further: identity is per *expansion site*, not per module.**
A copycode included several times by the same module (`JX0031N0.nat` includes `YFRAMBC0` 16 times) now
yields one set of nodes *per include site*, keyed by `includePath`. So counts rise again for those
modules, and — the point of the change — a copycode that opens a block it does not close no longer
collects every site's nesting into one node. `includedAt` alone does **not** identify a site (item 104
makes it the host's INCLUDE line at every nesting level, and `VPARTC02.cpy` includes `L4NLOGIC` 136 times
behind a single host line); use `includePath` when you need to tell two expansions apart.
Measured on `upms` after the recreate: copycode-resident nodes 27,551 -> 51,895 (project total +5.0%),
spread over 19,565 distinct owners. Endpoint latency on the heaviest module (`JX0030N0.nat`, 91 include
sites) is unaffected: `digest` 0.83 s, `context` 0.25 s, `graph` 0.19 s.
- **The same line number can legitimately appear twice (item 69).** A host statement on line 10 and a
copycode statement on line 10 are two different statements, and both are returned — as separate entries
differing only in their file. Until item 69 the graph could not hold both: an edge was identified by
@@ -637,35 +919,75 @@ origins (`http://localhost:5173`, `http://localhost:4173`) — extend the
## Endpoint quick reference
| Endpoint | Use for |
|----------------------------------------------------------------------------------------------------|---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| `GET /modules?sourceFile=&moduleKind=&extends=` | List/filter modules; map a source file to its module name(s). Each row carries `loc`/`sloc` (item 46) and `ingestStatus`/`ingestDepth` (item 50) for status badges without a per-module round trip |
| `GET /loc?language=&sourceFile=` | Per-language LoC/SLoC rollup (fileCount/loc/sloc) + project total; each file counted once (item 46). For a generated/user_exit project also `userExitLoc`/`userExitSloc` + `generatedExclusiveLoc`/`generatedExclusiveSloc` (item 47) |
| `GET /modules/{name}/digest` | Tiny triage view before deciding which modules to expand |
| `GET /modules/{name}/context` | One-shot overview: functions, callers, callees, DB accesses, SQL/variable summaries (`?include=` for full lists) |
| `GET /modules/{name}/callers` \| `/callees` | Direct callers/callees incl. `EXTENDS`/`IMPLEMENTS`/`INJECTS`/`REFERENCES`. `callers` `scope`: **`external` (default)** = modules that call this one (CALLNAT/inheritance), **rolled up to the calling MODULE**: a call made from inside a subroutine/method is attributed to its owning module (never the calling `FUNCTION` node), and repeated call sites from one caller collapse to a single row whose `sites` list every line — symmetric with how `callees` anchors its source side. `internal` = the module's own subroutines' `PERFORM` wiring (function-level). The default is external-only, module-typed only, and never lists the module as its own caller (no `MODULE→MODULE` self-loop); use `scope=internal` or `/functions/{fn}/callers` for intra-module / function-level wiring. `callees` is unchanged (default lists both external CALLNAT and internal PERFORM targets) |
| `GET /modules/{name}/functions/{function}/callers` | **FUNCTION-level callers** (item 52): who `PERFORM`s (Natural) or calls (Java cross-class) a specific subroutine/method, with call-site `lineNos`. Finer-grained than the module-level `/callers` (which is module→module). Same `CallRefResponse` shape. CLI `ac function-callers <module> <function>` |
| `GET /modules/{name}/call-tree?depth=` | Transitive call graph to scope a feature
| `GET /modules/{name}/reaches?target=A,B,C&direction=up\|down&depth=` | **Item 110 — "can A reach B, and how?"** Returns `{reachable, paths, truncated}` with one witness route per reached target (module names, source→target). `direction=down` (default): paths from this module to each target. `up`: paths from each target to this module. The counterpart to `call-tree`, which only walks downward and returns a closure without routes — one audit hand-rolled this as ~100 `/callers` requests. **`reachable: false` means "no path over known edges", not "no path"**: the traversal runs on resolved module calls, so a route through an unresolved dynamic `CALLNAT` (item 82) is invisible. Bounded by `depth` (item 75: the call graph has cycles). CLI `ac reaches <module> --target A,B --direction up` |
| `GET /duplicates` | **Item 114 — identities skipped at ingest** because they exist in more than one file (`{name, kind, paths}`, paths relative to the project root). These are *not* in `/modules`; asking for one by name gives `409 DUPLICATE_IDENTITY`. Their own calls are absent from the graph, so caller lists elsewhere can be short. CLI `ac duplicates` | |
| `GET /dynamic-calls/unresolved` \| `/overrides` · `POST`/`DELETE /overrides` | **Manual dynamic-`CALLNAT` overrides (item 82).** `unresolved` lists open `CALLNAT <var>` sites `{module, originFile, lineNo, variable}`; `POST /overrides {originFile, lineNo, targets[], variable?, note?}` pins a site to real module(s) (applied at once, persisted across refreshes, `400 UNKNOWN_TARGET` for a non-module); `DELETE /overrides?originFile=&lineNo=` resets one site (omit both = all) and restores the placeholder inline; `GET /overrides` lists them with an `obsolete` flag. CLI `ac dynamic-calls unresolved\|overrides\|set\|reset` |
| `GET /modules/{name}/graph?direction=&depth=&limit=` | Ego graph (item 49): bounded module-level call neighbourhood as **nodes + edges** (unlike call-tree). `direction` = `out`/`in`/`both`; `limit` caps nodes (BFS order) and sets `truncated`; unresolved targets carry `unresolved=true` + empty `sourceFile`. CLI `ac ego-graph` |
| `GET /modules/{name}/db-accesses` \| `/sql-statements` | DB tables + mode, raw statement text (pass `?depth=` for Natural). **`db-accesses`/`workfile-accesses` return every row when no `limit` is given (item 103)** — they used to default to 50, and since the response is a bare array with no total and no `truncated` flag the cut was invisible: `WGEAGB0S?depth=10` returned 50 of 64 rows and hid 7 tables outright. An explicit `limit` is still honoured exactly. `db-accesses` items carry **`sites: [{lineNo, sourceFile, viaCopycode, includedAt}]`** (+ kept `lineNos`); `sql-statements` items carry **`sourceFile`** + **`viaCopycode`** — so a copycode-sourced access (e.g. `SELECT … FROM SYSIBM-SYSDUMMY1` in `USIX043C.cpy`) reports the `.cpy` line, not a bare number that reads as a host-file line |
| `GET /modules/{name}/workfile-accesses` | Natural **work files** (sequential/flat-file I/O — `READ`/`WRITE WORK FILE n`), the work-file analogue of `db-accesses` (item 84): `[{workFile, physicalName, mode: READS\|WRITES, recordBuffers, lineNos, sites}]`, aggregated per work-file number + mode. `sites: [{lineNo, sourceFile, viaCopycode, includedAt}]` gives each access its file context (copycode-aware), like `db-accesses`. `physicalName` comes from a `DEFINE WORK FILE n '<name>'`, else `null`. **Kept separate from `db-accesses`** — a work file is not an ADABAS/SQL table (fixes a former bug where `READ WORK FILE` created a phantom `DB_TABLE 'WORK'`). CLI `ac workfile-accesses <module>` |
| `GET /modules/{name}/data-structures` | Which copybooks/inline groups a module uses. A `USING <member>` binds by **member (file) name**, never by a level-1 record inside the file (item 100) — before that, `WGEAGB0S USING W-WIF-A2` reported `old/W-WIF-A7.pda` (whose level-1 record is a copy-pasted `1W-WIF-A2`), and a data area with several level-1 records and none named after the member (`VLAYERLA.lda`, `USIX020L.lda`) resolved to nothing at all (`sourceFile: null`, `area: UNKNOWN`, `fieldCount: 0`) although the file was ingested. One row per resolved definition, `(name, sourceFile)` (item 102) — never one row blending an arbitrary file with another definition's `fieldCount` |
| `GET /modules/{name}/payload` | Natural XML wire-payload contract: `{tag, field, direction, source, lineNo, sourceFile}` — static `ADD-XML-LINE` idiom (`source=IDIOM`, item 45) or derived from the wrapper's interface PDA (`source=PDA`, item 46b). `sourceFile` is the file `lineNo` refers to (module for IDIOM, PDA for PDA) |
| `GET /modules/{name}/dispatch-table` | Natural `DECIDE ON VALUE OF` routing table |
| `GET /modules/{name}/functions?kind=` \| `/functions/{fn}/overrides` \| `/functions/overrides` | Method list, modifier filter (Java), subclass overrides (single/bulk). Each item carries **`sourceFile`** + **`viaCopycode`** (item 84): a Natural subroutine pulled in via `INCLUDE` reports the **copycode** file and `viaCopycode:true`, so its `startLine`/`endLine` are read as offsets into that copycode — **not** into the including module's own file (which is shorter). `viaCopycode:false` = declared inline. Always `false` for Java |
| `GET /data-structures/{name}/fields` \| `/db-tables/{name}/columns` \| `/modules/{name}/columns` | Field/column schemas for DTO/entity generation. Every field carries **`sourceFile`** (item 101). When a structure name resolves to several definitions (42 level-1 names recur across `upms` data areas), the **member root** — the definition whose file basename equals the name, i.e. what a `USING <member>` binds to — wins; **`?sourceFile=`** pins a specific one. Before item 101 the definitions were silently unioned: `W-WIF-A2` returned 15 fields, the merge of `W-WIF-A2.pda` (5) and `W-WIF-A7.pda` (10), a layout that exists nowhere |
| `GET /variables/{name}/reads` \| `/writes` \| `/flow-forward` \| `/flow-backward` \| `/field-flow` | Impact analysis and dataflow tracing |
| `GET /search/identifier` \| `/search/value` \| `/search/annotation` | Cross-project lookup by name / literal value / annotation. `search/identifier` matches the **exact** declared name but is **sigil-insensitive**: a leading Natural sigil (`#` user, `&` AIV, `+` GDA) is ignored on both sides, so `name=K-OUT-MAX` finds the declared `#K-OUT-MAX` (and vice-versa). Optional **scope** filters `sourceFile=<relpath>` and `module=<name>` (item 53) narrow the match to one file / one module — use them to pinpoint a module-local declaration when a name recurs across dozens of modules (the result is otherwise paginated and the local one may fall off the page). To keep the **full cross-project list** yet still guarantee a given module's own declaration is on the first page, pass `priorityModule=<name>` instead of `module=`: it does not filter, but pins that module's matches to the front (ahead of the otherwise `sourceFile`-ordered rest) so they survive the `limit`. This is what the web UI's click-to-identify sends for the open module. CLI `ac search-identifier --module --priority-module --source-file --type` accept the same filters. **Latency (item 105):** a lookup whose hits lie in a Natural data area used to take 60-75 s — every fan-out query deep-ingested the surfaced `.lda`/`.pda`, which can never reach `FULL` (a data area yields no `MODULE` node), so it was re-warmed on every call and each warm dragged a whole-project finalize behind it. Data areas are now excluded from the fan-out warm; they have no deep tier to gain |
| `GET /search/source?regex=&limit=&ignoreCase=` (`ac search-source`) | Regex **grep over module source text** (item 54): `{module, sourceFile, lineNo, line}` hits + `truncated`. Case-insensitive by default. Complements `/search/identifier` (declared names) — use for code patterns (statements, table names, literals) |
| `GET /nodes/{id}` | Every property of one node (when a curated DTO is missing something) |
| `GET /nodes/{id}/source` \| `/modules/{name}/source` \| `/source?file=` | Source text — **only when you have no other access to the source** (you always do in this repo, see "Reading source in this repo" above). `/modules/{name}/source` returns the **whole file** when the line range is omitted (M1), or a `[startLine,endLine]` slice when both are given. `/source?file=<relpath>` (CLI `ac file-source`) serves a file by **relative path** rather than module name — for files that aren't standalone modules, e.g. a Natural data area (PDA/LDA) USING'd by a module, whose field line numbers refer to that file. Same whole-file/range + stale-source semantics; the client-supplied path is rejected (`400 INVALID_SOURCE_FILE`) if it escapes the project root |
| Endpoint | Use for |
|----------------------------------------------------------------------------------------------------|----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
| `GET /modules?sourceFile=&moduleKind=&extends=` | List/filter modules; map a source file to its module name(s). Each row carries `loc`/`sloc` (item 46) and `ingestStatus`/`ingestDepth` (item 50) for status badges without a per-module round trip |
| `GET /loc?language=&sourceFile=` | Per-language LoC/SLoC rollup (fileCount/loc/sloc) + project total; each file counted once (item 46). For a generated/user_exit project also `userExitLoc`/`userExitSloc` + `generatedExclusiveLoc`/`generatedExclusiveSloc` (item 47) |
| `GET /modules/{name}/digest` | Tiny triage view before deciding which modules to expand |
| `GET /modules/{name}/context` | One-shot overview: functions, callers, callees, DB accesses, SQL/variable summaries (`?include=` for full lists) |
| `GET /modules/{name}/callers` \| `/callees` | Direct callers/callees incl. `EXTENDS`/`IMPLEMENTS`/`INJECTS`/`REFERENCES`. `callers` `scope`: **`external` (default)** = modules that call this one (CALLNAT/inheritance), **rolled up to the calling MODULE**: a call made from inside a subroutine/method is attributed to its owning module (never the calling `FUNCTION` node), and repeated call sites from one caller collapse to a single row whose `sites` list every line — symmetric with how `callees` anchors its source side. `internal` = the module's own subroutines' `PERFORM` wiring (function-level). The default is external-only, module-typed only, and never lists the module as its own caller (no `MODULE→MODULE` self-loop); use `scope=internal` or `/functions/{fn}/callers` for intra-module / function-level wiring. `callees` is unchanged (default lists both external CALLNAT and internal PERFORM targets) |
| `GET /modules/{name}/functions/{function}/callers` | **FUNCTION-level callers** (item 52): who `PERFORM`s (Natural) or calls (Java cross-class) a specific subroutine/method, with call-site `lineNos`. Finer-grained than the module-level `/callers` (which is module→module). Same `CallRefResponse` shape. CLI `ac function-callers <module> <function>` |
| `GET /modules/{name}/call-tree?depth=` | Transitive call graph to scope a feature
| `GET /modules/{name}/reaches?target=A,B,C&direction=up\|down&depth=` | **Item 110 — "can A reach B, and how?"** Returns `{reachable, paths, truncated}` with one witness route per reached target (module names, source→target). `direction=down` (default): paths from this module to each target. `up`: paths from each target to this module. The counterpart to `call-tree`, which only walks downward and returns a closure without routes — one audit hand-rolled this as ~100 `/callers` requests. **`reachable: false` means "no path over known edges", not "no path"**: the traversal runs on resolved module calls, so a route through an unresolved dynamic `CALLNAT` (item 82) is invisible. Bounded by `depth` (item 75: the call graph has cycles). CLI `ac reaches <module> --target A,B --direction up` |
| `GET /duplicates` | **Item 114 — identities skipped at ingest** because they exist in more than one file (`{name, kind, paths}`, paths relative to the project root). These are *not* in `/modules`; asking for one by name gives `409 DUPLICATE_IDENTITY`. Their own calls are absent from the graph, so caller lists elsewhere can be short. CLI `ac duplicates` | |
| `GET /dynamic-calls/unresolved` \| `/overrides` · `POST`/`DELETE /overrides` | **Manual dynamic-`CALLNAT` overrides (item 82).** `unresolved` lists open `CALLNAT <var>` sites `{module, originFile, lineNo, variable}`; `POST /overrides {originFile, lineNo, targets[], variable?, note?}` pins a site to real module(s) (applied at once, persisted across refreshes, `400 UNKNOWN_TARGET` for a non-module); `DELETE /overrides?originFile=&lineNo=` resets one site (omit both = all) and restores the placeholder inline; `GET /overrides` lists them with an `obsolete` flag. CLI `ac dynamic-calls unresolved\|overrides\|set\|reset` |
| `GET /modules/{name}/graph?direction=&depth=&limit=` | Ego graph (item 49): bounded module-level call neighbourhood as **nodes + edges** (unlike call-tree). `direction` = `out`/`in`/`both`; `limit` caps nodes (BFS order) and sets `truncated`; unresolved targets carry `unresolved=true` + empty `sourceFile`. CLI `ac ego-graph` |
| `GET /modules/{name}/db-accesses` \| `/sql-statements` | DB tables + mode, raw statement text (pass `?depth=` for Natural). **`db-accesses`/`workfile-accesses` return every row when no `limit` is given (item 103)** — they used to default to 50, and since the response is a bare array with no total and no `truncated` flag the cut was invisible: `WGEAGB0S?depth=10` returned 50 of 64 rows and hid 7 tables outright. An explicit `limit` is still honoured exactly. `db-accesses` items carry **`sites: [{lineNo, sourceFile, viaCopycode, includedAt}]`** (+ kept `lineNos`); `sql-statements` items carry **`sourceFile`** + **`viaCopycode`** — so a copycode-sourced access (e.g. `SELECT … FROM SYSIBM-SYSDUMMY1` in `USIX043C.cpy`) reports the `.cpy` line, not a bare number that reads as a host-file line |
| `GET /modules/{name}/workfile-accesses` | Natural **work files** (sequential/flat-file I/O — `READ`/`WRITE WORK FILE n`), the work-file analogue of `db-accesses` (item 84): `[{workFile, physicalName, mode: READS\|WRITES, recordBuffers, lineNos, sites}]`, aggregated per work-file number + mode. `sites: [{lineNo, sourceFile, viaCopycode, includedAt}]` gives each access its file context (copycode-aware), like `db-accesses`. `physicalName` comes from a `DEFINE WORK FILE n '<name>'`, else `null`. **Kept separate from `db-accesses`** — a work file is not an ADABAS/SQL table (fixes a former bug where `READ WORK FILE` created a phantom `DB_TABLE 'WORK'`). CLI `ac workfile-accesses <module>` |
| `GET /modules/{name}/data-structures` | Which copybooks/inline groups a module uses. A `USING <member>` binds by **member (file) name**, never by a level-1 record inside the file (item 100) — before that, `WGEAGB0S USING W-WIF-A2` reported `old/W-WIF-A7.pda` (whose level-1 record is a copy-pasted `1W-WIF-A2`), and a data area with several level-1 records and none named after the member (`VLAYERLA.lda`, `USIX020L.lda`) resolved to nothing at all (`sourceFile: null`, `area: UNKNOWN`, `fieldCount: 0`) although the file was ingested. One row per resolved definition, `(name, sourceFile)` (item 102) — never one row blending an arbitrary file with another definition's `fieldCount` |
| `GET /modules/{name}/payload` | Natural XML wire-payload contract: `{tag, field, direction, source, lineNo, sourceFile}` — static `ADD-XML-LINE` idiom (`source=IDIOM`, item 45) or derived from the wrapper's interface PDA (`source=PDA`, item 46b). `sourceFile` is the file `lineNo` refers to (module for IDIOM, PDA for PDA) |
| `GET /modules/{name}/comments?kind=&limit=&offset=` (`ac comments`) | **Item 141:** the module's **comment blocks** — `{text, kind, sourceFile, startLine, endLine, target, targetType, truncated}`, one row per contiguous block, ordered by line. `target`/`targetType` name the declaration the block documents: the declaration immediately below it, else the one enclosing it (so a file header banner documents the `MODULE`, a `/*` comment on a field's own line documents that field). `kind` is `JAVADOC`\|`LINE`\|`BLOCK` (Java) or `NATURAL_BANNER`\|`NATURAL_INLINE`\|`SAG` (Natural); `?kind=` filters to one. **`SAG` is excluded by default** — `**SAG` directives are generator metadata, not human notes, and would otherwise be most of the answer for every generated Natural module. Text is cut at 4 000 chars (`truncated:true`); read the file for the rest. Natural copycode comments belong to the **copycode's own module**, not to each includer. **Deep-gated:** comments come from the full parse, not the Tier-1 coarse scan, so the module is deep-ingested on demand and a still-shallow module answers `409 NOT_DEEPLY_INGESTED` rather than a misleading `[]` |
| `GET /modules/{name}/dispatch-table` | Natural `DECIDE ON VALUE OF` routing table |
| `GET /modules/{name}/functions?kind=` \| `/functions/{fn}/overrides` \| `/functions/overrides` | Method list, modifier filter (Java), subclass overrides (single/bulk). Each item carries **`sourceFile`** + **`viaCopycode`** (item 84): a Natural subroutine pulled in via `INCLUDE` reports the **copycode** file and `viaCopycode:true`, so its `startLine`/`endLine` are read as offsets into that copycode — **not** into the including module's own file (which is shorter). `viaCopycode:false` = declared inline. Always `false` for Java |
| `GET /data-structures/{name}/fields` \| `/db-tables/{name}/columns` \| `/modules/{name}/columns` | Field/column schemas for DTO/entity generation. Every field carries **`sourceFile`** (item 101). When a structure name resolves to several definitions (42 level-1 names recur across `upms` data areas), the **member root** — the definition whose file basename equals the name, i.e. what a `USING <member>` binds to — wins; **`?sourceFile=`** pins a specific one. Before item 101 the definitions were silently unioned: `W-WIF-A2` returned 15 fields, the merge of `W-WIF-A2.pda` (5) and `W-WIF-A7.pda` (10), a layout that exists nowhere |
| `GET /variables/{name}/reads` \| `/writes` \| `/flow-forward` \| `/flow-backward` \| `/field-flow` | Impact analysis and dataflow tracing |
| `GET /search/identifier` \| `/search/value` \| `/search/annotation` | Cross-project lookup by name / literal value / annotation. **All three see code only by default — an empty result is not evidence that the string is absent.** `search/value` takes **`includeComments=true`** (CLI `--include-comments`, item 141) to search comment blocks as well; those hits come back as `kind: "COMMENT"`, so a comment is never read as code. It is opt-in because a comment hit is different evidence from a literal, and folding it in silently would move every existing completeness count (item 131). `search/identifier` and `search/annotation` never match comments at all — use `includeComments`, `/modules/{name}/comments` or `/search/source` before concluding "not present" (see "Comments: reachable, but never by default" above). `search/identifier` matches the **exact** declared name but is **sigil-insensitive**: a leading Natural sigil (`#` user, `&` AIV, `+` GDA) is ignored on both sides, so `name=K-OUT-MAX` finds the declared `#K-OUT-MAX` (and vice-versa). **Item 125:** a Java **type declaration** is matched by its **short name** as well as by the fully-qualified identity the graph stores (item 117) — `name=PartnerUpdateLogic` finds `com.example.PartnerUpdateLogic`; before this it answered `[]`, which reads as "no such name". Every match carries `simpleName` and `moduleKind` (`CLASS`/`INTERFACE`/`ENUM`/`RECORD`, `PROGRAM`/`SUBPROGRAM` for Natural), both `null` for non-`MODULE` hits — so "is this name a type or a method?" needs no second call. `contains=true` (CLI `--contains`) switches to a case-insensitive **substring** match, as on `/search/value`; it was previously accepted and silently dropped. It matches the FQN too, so a package fragment also hits — filter with `type=MODULE`/`moduleKind` if that is noise. `contains` without a `name` is `400 MISSING_NAME` (a substring search for nothing is a full node dump). The substring scan is **unindexed**: it is bounded to `offset+limit` rows, so keep a `limit` on large projects. Optional **scope** filters `sourceFile=<relpath>` and `module=<name>` (item 53) narrow the match to one file / one module — use them to pinpoint a module-local declaration when a name recurs across dozens of modules (the result is otherwise paginated and the local one may fall off the page). To keep the **full cross-project list** yet still guarantee a given module's own declaration is on the first page, pass `priorityModule=<name>` instead of `module=`: it does not filter, but pins that module's matches to the front (ahead of the otherwise `sourceFile`-ordered rest) so they survive the `limit`. This is what the web UI's click-to-identify sends for the open module. CLI `ac search-identifier --module --priority-module --source-file --type --contains` accept the same filters. **Latency (item 105):** a lookup whose hits lie in a Natural data area used to take 60-75 s — every fan-out query deep-ingested the surfaced `.lda`/`.pda`, which can never reach `FULL` (a data area yields no `MODULE` node), so it was re-warmed on every call and each warm dragged a whole-project finalize behind it. Data areas are now excluded from the fan-out warm; they have no deep tier to gain |
| `GET /search/source?regex=&limit=&ignoreCase=` (`ac search-source`) | Regex **grep over module source text** (item 54): `{module, sourceFile, lineNo, line}` hits + `truncated`. Case-insensitive by default. Complements `/search/identifier` (declared names) — use for code patterns (statements, table names, literals). Sees **everything in the file, comments included**, and needs no ingest depth — so it is the fallback when a module is not deeply ingested, or when the text is something the parsers do not model. For comments specifically, prefer the graph routes added by item 141 (`/modules/{name}/comments`, `search/value?includeComments=true`), which also tell you which declaration a comment belongs to |
| `GET /nodes/{id}` | Every property of one node (when a curated DTO is missing something) |
| `GET /nodes/{id}/source` \| `/modules/{name}/source` \| `/source?file=` | Source text — **only when you have no other access to the source** (you always do in this repo, see "Reading source in this repo" above). `/modules/{name}/source` returns the **whole file** when the line range is omitted (M1), or a `[startLine,endLine]` slice when both are given. `/source?file=<relpath>` (CLI `ac file-source`) serves a file by **relative path** rather than module name — for files that aren't standalone modules, e.g. a Natural data area (PDA/LDA) USING'd by a module, whose field line numbers refer to that file. Same whole-file/range + stale-source semantics; the client-supplied path is rejected (`400 INVALID_SOURCE_FILE`) if it escapes the project root |
Full endpoint list, request params, and response field details:
`x-docs/agent-api-system-prompt.md`.
## Errors are structured JSON — always (item 136)
Every failure now answers `{ "error": ..., "code": ..., "details": {} }`, including the ones nobody
planned for: an unhandled exception is mapped to `500 INTERNAL_ERROR` with an `errorId` in `details`
that matches the stack trace in the server log (the trace itself is never in the response). Before
this, an unexpected fault escaped as a plain-text Quarkus error page with no `code` to branch on —
which is exactly the moment a client most needs a machine-readable answer. Deliberate statuses
(`PROJECT_NOT_FOUND`, `MISSING_NAME`, `STALE_SOURCE`, the runtime's own routing 404s) pass through
unchanged.
The fault that exposed this: `search/identifier` coerced `startLine`/`endLine` unconditionally, and
item 114's duplicate markers were the one kind of node created without them, so any page long enough
to reach a marker (row 487 on `ac`) died. Both halves are fixed — the markers now carry lines, and the
row mapper no longer trusts that they will.
## Verifying the API after a deploy (item 134)
After `./manage-ac.sh deploy` (or `./rebuild-and-refresh.sh`), run:
```bash
./x-scripts/verify-api.sh # defaults to project 'ac'
./x-scripts/verify-api.sh -p upms # any ingested project
AC_SERVER_URL=http://host:8787 ./x-scripts/verify-api.sh
```
It answers one question in ~10 s: *does the server that is running right now still return
plausible data over the real graph?* Exit 0 = all green, 1 = at least one check failed. Every line is
`PASS`, `FAIL` or `SKIP`; `SKIP` means the endpoint family does not apply to that project (a pure
Natural project has no `rest-endpoints`, a leaf module has no callees).
What it covers: `/api/version` and the project list; the item-130 scope headers
(`X-AC-Exclude-Dirs`, `X-AC-Ingested-At`, `X-AC-Ingest-Incomplete` — a `true` there means a refresh
was aborted and every later answer is drawn from a half-updated graph); the item-131 paging contract
on the search endpoints; per-family data plausibility; and the structured-error negative cases.
What it is **not**: a substitute for `mvn test`. The integration tests pin semantics; this pins
"the deployed thing is not obviously broken". A green run is not a quality gate. All assertions are
invariants, never fixed counts — counts move with every refresh.
## Missing capability?
If the API/CLI genuinely cannot answer a question (not just

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

353
x-scripts/verify-api.sh Executable file
View File

@@ -0,0 +1,353 @@
#!/usr/bin/env bash
#
# verify-api.sh — post-deploy smoke test for the AgenticCode REST API.
#
# Usage:
# ./x-scripts/verify-api.sh [-p <project>] # default project: ac
# AC_SERVER_URL=http://host:8787 ./x-scripts/verify-api.sh -p pur
#
# Exit 0 = every check passed, 1 = at least one failed, 2 = usage/precondition error.
#
# What this IS: a check that the *deployed* server, against the *real* graph, still answers
# plausibly — the failure class that only ever showed up in manual live probing ('//file' in
# REST paths, duplicated rows, an inherited @Path collapsing to 'POST /', ?paths=pom.xml
# being accepted). It runs in well under a minute and touches no build tooling.
#
# What this is NOT: a replacement for the integration tests. Those pin semantics; this pins
# "the thing we just deployed is not obviously broken". Never treat a green run here as a
# quality gate.
#
# Assertions are INVARIANTS (> 0, no duplicates, header present, required field set), never
# fixed row counts — counts move with every refresh and differ per project.
#
# Mutation: the run is read-only except for one call, `refresh?paths=pom.xml`, which by
# design resolves no source file and therefore starts no ingest. It does invalidate the
# project metadata cache. Nothing else writes.
set -uo pipefail
SERVER="${AC_SERVER_URL:-http://localhost:8787}"
API="$SERVER/api"
PROJECT="ac"
usage() { echo "Usage: $0 [-p <project>]" >&2; exit 2; }
while getopts ":p:h" opt; do
case "$opt" in
p) PROJECT="$OPTARG" ;;
h) usage ;;
*) usage ;;
esac
done
command -v curl >/dev/null 2>&1 || { echo "verify-api.sh: curl is required." >&2; exit 2; }
command -v python3 >/dev/null 2>&1 || { echo "verify-api.sh: python3 is required." >&2; exit 2; }
PASSED=0
FAILED=0
START=$(date +%s)
GREEN=$'\033[0;32m'; RED=$'\033[0;31m'; BOLD=$'\033[1m'; DIM=$'\033[2m'; OFF=$'\033[0m'
pass() { PASSED=$((PASSED + 1)); printf ' %sPASS%s %-52s %s%s%s\n' "$GREEN" "$OFF" "$1" "$DIM" "${2:-}" "$OFF"; }
fail() { FAILED=$((FAILED + 1)); printf ' %sFAIL%s %-52s %s\n' "$RED" "$OFF" "$1" "${2:-}"; }
group() { printf '\n%s%s%s\n' "$BOLD" "$1" "$OFF"; }
# check <name> <condition-exit-code> <detail>
check() {
if [[ "$2" -eq 0 ]]; then pass "$1" "${3:-}"; else fail "$1" "${3:-}"; fi
}
BODY=$(mktemp); HEAD=$(mktemp)
trap 'rm -f "$BODY" "$HEAD"' EXIT
# get <path...> -> sets STATUS, body in $BODY, headers in $HEAD
get() {
STATUS=$(curl -s -m 30 -o "$BODY" -D "$HEAD" -w '%{http_code}' "$@" 2>/dev/null)
[[ -n "$STATUS" ]] || STATUS=000
}
hdr() { grep -i "^$1:" "$HEAD" | head -1 | cut -d' ' -f2- | tr -d '\r'; }
# py <expr-script> — runs python3 against the body; exit 0 means the assertion held.
py() { python3 -c "$1" "$BODY" "${@:2}" 2>/dev/null; }
printf '%sverify-api.sh%s server=%s project=%s\n' "$BOLD" "$OFF" "$SERVER" "$PROJECT"
# --- 1. reachability & version ------------------------------------------------------------
group "1. Reachability & version"
get "$API/version"
check "GET /api/version returns 200" "$([[ $STATUS == 200 ]] && echo 0 || echo 1)" "status=$STATUS"
if [[ $STATUS == 200 ]]; then
VERSION=$(py 'import json,sys; d=json.load(open(sys.argv[1])); v=str(d.get("version","")); print(v); sys.exit(0 if v else 1)')
check "version is non-empty" $? "version=$VERSION"
else
echo " server unreachable at $SERVER — is the stack deployed (./manage-ac.sh deploy)?" >&2
fi
get "$API/projects"
check "GET /api/projects returns 200" "$([[ $STATUS == 200 ]] && echo 0 || echo 1)" "status=$STATUS"
py 'import json,sys; d=json.load(open(sys.argv[1])); sys.exit(0 if any(p.get("name")==sys.argv[2] for p in d) else 1)' "$PROJECT"
check "project '$PROJECT' is listed" $?
if [[ $FAILED -gt 0 ]]; then
echo
echo "Aborting: the server is not usable, later checks would only add noise." >&2
exit 1
fi
# --- 2. scope & freshness headers (item 130) ----------------------------------------------
group "2. Scope & freshness headers (item 130)"
get "$API/projects/$PROJECT/modules?limit=1"
EXCL=$(hdr X-AC-Exclude-Dirs); AT=$(hdr X-AC-Ingested-At); INC=$(hdr X-AC-Ingest-Incomplete)
check "X-AC-Exclude-Dirs present" "$([[ -n $EXCL ]] && echo 0 || echo 1)" "$EXCL"
check "X-AC-Ingested-At present" "$([[ -n $AT ]] && echo 0 || echo 1)" "$AT"
check "X-AC-Ingest-Incomplete=false" "$([[ $INC == false ]] && echo 0 || echo 1)" \
"$([[ $INC == false ]] || echo "got '$INC' — a refresh was aborted or is running")"
# --- 3. paging contract (item 131) --------------------------------------------------------
group "3. Paging contract (item 131)"
# Endpoints that carry X-AC-Total-Count, each with a query that yields rows in any project.
paging_check() {
local label="$1" path="$2"
get "$API/projects/$PROJECT/$path"
local total; total=$(hdr X-AC-Total-Count)
if ! [[ $total =~ ^[0-9]+$ ]]; then
fail "$label: X-AC-Total-Count is numeric" "got '$total' (status=$STATUS)"
return
fi
pass "$label: X-AC-Total-Count is numeric" "total=$total"
if [[ $total -eq 0 ]]; then
printf ' %sSKIP%s %-52s %s%s%s\n' "$DIM" "$OFF" "$label: paging (no rows to page)" "$DIM" "total=0" "$OFF"
return
fi
# countOnly must report the same total without returning the page.
get "$API/projects/$PROJECT/$(printf %s "$path" | sed 's/?/?countOnly=true\&/')"
local co; co=$(hdr X-AC-Total-Count)
check "$label: countOnly agrees with full total" \
"$([[ $co == "$total" ]] && echo 0 || echo 1)" "countOnly=$co full=$total"
# limit=1 must flag truncation whenever more than one row exists.
get "$API/projects/$PROJECT/$(printf %s "$path" | sed 's/?/?limit=1\&/')"
local tr; tr=$(hdr X-AC-Truncated)
local want=false; [[ $total -gt 1 ]] && want=true
check "$label: limit=1 sets X-AC-Truncated=$want" \
"$([[ $tr == "$want" ]] && echo 0 || echo 1)" "truncated=$tr total=$total"
# Two pages of 1 must be disjoint — the offset actually moves.
local p0 p1
get "$API/projects/$PROJECT/$(printf %s "$path" | sed 's/?/?limit=1\&offset=0\&/')"
p0=$(py 'import json,sys; d=json.load(open(sys.argv[1])); print(json.dumps(d[0],sort_keys=True) if d else "")')
get "$API/projects/$PROJECT/$(printf %s "$path" | sed 's/?/?limit=1\&offset=1\&/')"
p1=$(py 'import json,sys; d=json.load(open(sys.argv[1])); print(json.dumps(d[0],sort_keys=True) if d else "")')
if [[ $total -gt 1 ]]; then
check "$label: offset=0 and offset=1 are disjoint" \
"$([[ -n $p0 && -n $p1 && $p0 != "$p1" ]] && echo 0 || echo 1)"
fi
}
paging_check "search/identifier" "search/identifier?name=e&contains=true"
# A deep page must not blow up: a row with a NULL startLine used to escape as an unstructured
# 500 ("Cannot coerce NULL to Java int") that only appears past the first few hundred rows.
get "$API/projects/$PROJECT/search/identifier?name=e&contains=true&limit=500"
check "search/identifier: a 500-row page does not fault" \
"$([[ $STATUS == 200 ]] && echo 0 || echo 1)" "status=$STATUS"
paging_check "search/annotation" "search/annotation?name=ApplicationScoped"
paging_check "search/value" "search/value?value=project"
paging_check "search/references" "search/references?name=Logger"
paging_check "rest-endpoints" "rest-endpoints?"
# --- 4. data plausibility -----------------------------------------------------------------
group "4. Data plausibility"
# rest-endpoints: the concrete bugs that only real data exposed.
get "$API/projects/$PROJECT/rest-endpoints?limit=500"
if [[ $STATUS == 200 ]]; then
# A pure Natural project declares no HTTP endpoints — absence is correct there, not a defect.
if ! py 'import json,sys; sys.exit(0 if json.load(open(sys.argv[1])) else 1)'; then
printf ' %sSKIP%s %-52s %s%s%s\n' "$DIM" "$OFF" "rest-endpoints (project declares none)" \
"$DIM" "0 rows — expected for a non-JAX-RS project" "$OFF"
REST_ROWS=0
else
pass "rest-endpoints returns rows"
REST_ROWS=1
fi
if [[ $REST_ROWS == 1 ]]; then
py 'import json,sys; d=json.load(open(sys.argv[1])); bad=[e["path"] for e in d if "//" in e["path"]]; print(*bad[:3]); sys.exit(1 if bad else 0)'
check "no path contains '//'" $?
py 'import json,sys; d=json.load(open(sys.argv[1])); bad=[e["path"] for e in d if not e["path"].startswith("/")]; print(*bad[:3]); sys.exit(1 if bad else 0)'
check "every path starts with '/'" $?
# NOTE: the query already applies DISTINCT, so this cannot surface item 75's duplicate
# CONTAINS edges — it only guards against that DISTINCT being dropped again.
py 'import json,sys
d=json.load(open(sys.argv[1]))
k=[(e["httpMethod"],e["path"],e["module"],e["handler"]) for e in d]
dup=len(k)-len(set(k)); print("duplicates:",dup); sys.exit(1 if dup else 0)'
check "no duplicate endpoint rows" $?
py 'import json,sys; d=json.load(open(sys.argv[1])); sys.exit(0 if all("outbound" in e for e in d) else 1)'
check "every row carries the outbound flag" $?
py 'import json,sys; d=json.load(open(sys.argv[1])); sys.exit(0 if any(e["httpMethod"]=="GET" and e["path"]=="/api/version" for e in d) else 1)'
SELF=$?
if [[ $PROJECT == ac ]]; then
check "the server's own GET /api/version is found" $SELF
fi
fi
else
fail "rest-endpoints returns 200" "status=$STATUS"
fi
# Pick a module that actually exists in this project, then exercise the module endpoints.
get "$API/projects/$PROJECT/modules?limit=200"
CANDIDATES=$(py 'import json,sys
d=json.load(open(sys.argv[1]))
c=[m for m in d if m.get("ingestDepth")=="FULL"] or d
print("\n".join(m["name"] for m in c[:20]))')
# Prefer a module that actually calls something — a leaf would make the callees check vacuous.
MODULE=""; CALLEE_MODULE=""
while IFS= read -r cand; do
[[ -n $cand ]] || continue
[[ -n $MODULE ]] || MODULE="$cand"
E=$(python3 -c 'import sys,urllib.parse; print(urllib.parse.quote(sys.argv[1],safe=""))' "$cand")
get "$API/projects/$PROJECT/modules/$E/callees?limit=20"
if [[ $STATUS == 200 ]] && py 'import json,sys; sys.exit(0 if json.load(open(sys.argv[1])).get("items") else 1)'; then
CALLEE_MODULE="$cand"; break
fi
done <<< "$CANDIDATES"
check "a FULL-ingested module is available" "$([[ -n $MODULE ]] && echo 0 || echo 1)" "module=$MODULE"
nonempty_rows() { # <label> <path> — 200 and a non-empty array
get "$API/projects/$PROJECT/$2"
if [[ $STATUS != 200 ]]; then fail "$1" "status=$STATUS"; return; fi
py 'import json,sys; d=json.load(open(sys.argv[1])); print(len(d),"rows"); sys.exit(0 if d else 1)'
check "$1" $? ""
}
fields_set() { # <label> <path> <field...> — 200 and every row has the fields
get "$API/projects/$PROJECT/$2"
if [[ $STATUS != 200 ]]; then fail "$1" "status=$STATUS"; return; fi
py 'import json,sys
d=json.load(open(sys.argv[1]))
rows=d if isinstance(d,list) else [d]
miss={f for r in rows for f in sys.argv[2:] if r.get(f) is None}
print("missing:",",".join(sorted(miss)) or "-")
sys.exit(1 if miss else 0)' "${@:3}"
check "$1" $? ""
}
if [[ -n $MODULE ]]; then
ENC=$(python3 -c 'import sys,urllib.parse; print(urllib.parse.quote(sys.argv[1],safe=""))' "$MODULE")
# callees answers with the shared file-index envelope {sourceFiles, items}: every item must
# name a callee and every sourceFileIndex must resolve into the sourceFiles table.
if [[ -z $CALLEE_MODULE ]]; then
printf ' %sSKIP%s %-52s %s%s%s\n' "$DIM" "$OFF" "callees: sourceFileIndex resolves" \
"$DIM" "no module among the first 20 has callees" "$OFF"
else
CENC=$(python3 -c 'import sys,urllib.parse; print(urllib.parse.quote(sys.argv[1],safe=""))' "$CALLEE_MODULE")
get "$API/projects/$PROJECT/modules/$CENC/callees?limit=20"
if [[ $STATUS == 200 ]]; then
py 'import json,sys
d=json.load(open(sys.argv[1]))
files=d.get("sourceFiles",[]); items=d.get("items",[])
def idx(i):
v = i.get("sourceFileIndex")
return v if isinstance(v, int) else -1
bad=[i.get("name") for i in items if not i.get("name") or not (0 <= idx(i) < len(files))]
print(len(items),"callees,",len(files),"files; bad:",bad[:3] or "-")
sys.exit(1 if (not items or bad) else 0)'
check "callees: every item resolves its sourceFileIndex" $? "via $CALLEE_MODULE"
else
fail "callees returns 200" "status=$STATUS"
fi
fi
fields_set "functions rows carry name/startLine" "modules/$ENC/functions?limit=20" name startLine
get "$API/projects/$PROJECT/modules/$ENC/context"
check "context returns 200" "$([[ $STATUS == 200 ]] && echo 0 || echo 1)" "status=$STATUS"
get "$API/projects/$PROJECT/modules/$ENC/digest"
check "digest returns 200" "$([[ $STATUS == 200 ]] && echo 0 || echo 1)" "status=$STATUS"
get "$API/projects/$PROJECT/modules/$ENC/graph"
check "graph returns 200" "$([[ $STATUS == 200 ]] && echo 0 || echo 1)" "status=$STATUS"
get "$API/projects/$PROJECT/modules/$ENC/source"
check "source returns 200 (not STALE_SOURCE)" "$([[ $STATUS == 200 ]] && echo 0 || echo 1)" \
"$([[ $STATUS == 200 ]] || echo "status=$STATUS — the graph is behind disk, refresh needed")"
fi
nonempty_rows "modules returns rows" "modules?limit=5"
nonempty_rows "loc returns rows" "loc?limit=5"
# search/source answers {regex, truncated, matches} — not a bare array.
get "$API/projects/$PROJECT/search/source?regex=project&limit=5"
if [[ $STATUS == 200 ]]; then
py 'import json,sys
d=json.load(open(sys.argv[1]))
m=d.get("matches",[])
bad=[x for x in m if not x.get("sourceFile") or not x.get("lineNo")]
print(len(m),"matches; truncated:",d.get("truncated"))
sys.exit(1 if (not m or bad) else 0)'
check "search/source matches carry sourceFile/lineNo" $? ""
else
fail "search/source returns 200" "status=$STATUS"
fi
# A node id from a live result must resolve through /nodes/{id} and /nodes/{id}/source.
# Pick a RESOLVED hit — an unresolved placeholder has no source file by design and would
# legitimately answer 400 NO_SOURCE_FILE.
SEED="${MODULE:-e}"
get "$API/projects/$PROJECT/search/identifier?name=$(python3 -c 'import sys,urllib.parse; print(urllib.parse.quote(sys.argv[1],safe=""))' "$SEED")&contains=true&limit=200"
NODE=$(py 'import json,sys
d=json.load(open(sys.argv[1]))
r=[n for n in d if n.get("sourceFile") and not n.get("unresolved")]
print(r[0]["id"] if r else "")')
if [[ -n $NODE ]]; then
get "$API/projects/$PROJECT/nodes/$NODE"
check "a live node id resolves via /nodes/{id}" "$([[ $STATUS == 200 ]] && echo 0 || echo 1)" "status=$STATUS"
get "$API/projects/$PROJECT/nodes/$NODE/source"
check "/nodes/{id}/source returns 200" "$([[ $STATUS == 200 ]] && echo 0 || echo 1)" "status=$STATUS"
else
fail "search/identifier yields a node id"
fi
# --- 5. negative cases --------------------------------------------------------------------
group "5. Negative cases"
structured_404() { # <label> <path> <expected-code>
get "$API/projects/$2"
if [[ $STATUS != 404 ]]; then fail "$1" "status=$STATUS (expected 404)"; return; fi
py 'import json,sys
d=json.load(open(sys.argv[1]))
ok = d.get("code")==sys.argv[2] and d.get("error") and isinstance(d.get("details"),dict)
print("code="+str(d.get("code")))
sys.exit(0 if ok else 1)' "$3"
check "$1" $? ""
}
structured_404 "unknown project -> 404 PROJECT_NOT_FOUND" "nope-does-not-exist/modules" PROJECT_NOT_FOUND
structured_404 "unknown module -> 404 MODULE_NOT_FOUND" "$PROJECT/modules/ZZZ-DOES-NOT-EXIST/callers" MODULE_NOT_FOUND
# Item 129 regression: a non-source path must come back unresolved, not be silently ingested.
STATUS=$(curl -s -m 30 -o "$BODY" -D "$HEAD" -w '%{http_code}' -X POST \
"$API/projects/$PROJECT/refresh?paths=pom.xml" 2>/dev/null)
if [[ $STATUS == 200 ]]; then
py 'import json,sys
d=json.load(open(sys.argv[1]))
u=d.get("unresolved") or []
print("unresolved:",u, "persisted:", d.get("filesPersisted"))
sys.exit(0 if "pom.xml" in u and not d.get("filesPersisted") else 1)'
check "refresh?paths=pom.xml is unresolved, nothing ingested" $? ""
else
fail "refresh?paths=pom.xml returns 200" "status=$STATUS"
fi
# --- summary ------------------------------------------------------------------------------
ELAPSED=$(( $(date +%s) - START ))
printf '\n%s' "$BOLD"
if [[ $FAILED -eq 0 ]]; then
printf '%sAll %d checks passed%s (%ss, server v%s, project %s)\n' "$GREEN" "$PASSED" "$OFF" "$ELAPSED" "${VERSION:-?}" "$PROJECT"
exit 0
fi
printf '%s%d passed, %d FAILED%s (%ss, server v%s, project %s)\n' "$RED" "$PASSED" "$FAILED" "$OFF" "$ELAPSED" "${VERSION:-?}" "$PROJECT"
exit 1