New feature

This commit is contained in:
Ingo Schnabel
2026-08-21 23:55:50 +03:00
parent 7adf7e54e4
commit bd86a01e92
12 changed files with 479 additions and 44 deletions

View File

@@ -4,4 +4,4 @@
server.url=http://localhost:8787
# Stamped by manage-ac.sh (stamp_cli_version) from ac-code-server's agenticcode.version
# at build time. "dev" means this jar wasn't built via manage-ac.sh.
version=247
version=250

View File

@@ -3,7 +3,7 @@ quarkus.http.port=8787
# AgenticCode's own release counter (not the Maven project version) — bump this by hand for each
# release. Single source of truth for the startup log line, GET /api/version, and the OpenAPI
# info version (referenced below via property expression, not duplicated).
agenticcode.version=247
agenticcode.version=250
# OpenAPI / Swagger UI (item 48) — the generated spec is the contract the web-UI TS client
# is generated against. Served at /q/openapi (yaml/json); Swagger UI at /q/swagger-ui in dev.
mp.openapi.extensions.smallrye.info.title=AgenticCode API

View File

@@ -0,0 +1,189 @@
package com.agenticcode.codeserver.api;
import io.quarkus.test.junit.QuarkusTest;
import io.restassured.RestAssured;
import jakarta.inject.Inject;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
import org.neo4j.driver.Driver;
import org.neo4j.driver.Session;
import java.io.IOException;
import java.io.UncheckedIOException;
import java.nio.charset.StandardCharsets;
import java.nio.file.Files;
import java.nio.file.Path;
import java.util.List;
import java.util.Map;
import static io.restassured.RestAssured.given;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertTrue;
/**
* Item 75-B: a copycode-resident node belongs to the module that includes it, not to the copycode.
*
* <p>{@code CopycodePreprocessor} splices a {@code .cpy} body into every including module before
* parsing, and the nodes it produces used to be MERGEd on {@code (type, name, sourceFile)} — so all
* includers shared <em>one</em> node. Two consequences, both of which this test pins:
*
* <ul>
* <li><b>Every includer's nesting context landed on one node.</b> A copycode may open a block it
* does not close (the {@code END-FOR} lives in the includer — {@code YFRAMBC0.cpy} in
* {@code upms} does exactly this), so the shared node was contained by, and contained, the
* statements of every module that included it. <b>Note:</b> this is <em>not</em> shown to be
* the source of item 75's {@code CONTAINS} cycles — every measured cycle pairs a copycode node
* with statements of a <em>single</em> host, which a per-module identity does not separate.
* No acyclicity is asserted here, because this fixture cannot produce a cycle either way and
* a test that cannot fail guards nothing.</li>
* <li><b>The stale-edge reaps could not run on it.</b> Items 124/86/106 key on the source node's
* file; with a shared node, reaping during one module's refresh would delete edges the other
* includers contributed and never re-create them. The fixture's second half asserts the
* opposite property now holds: refreshing one host leaves the other host's copies alone.</li>
* </ul>
*/
@QuarkusTest
class CopycodeNodeOwnershipIT {
private static final String PROJECT = "item75b-copycode-ownership";
private static final String CPY = "SHAREDBLK.cpy";
@TempDir
static Path root;
@Inject
Driver driver;
@BeforeAll
static void createProject() {
RestAssured.port = Integer.getInteger("quarkus.http.test-port", 8081);
// Opens a FOR it never closes: the END-FOR is supplied by each including module. This is the
// shape that made the shared node cyclic.
write(CPY, """
FOR #I = 1 TO 10
CALLNAT 'SHAREDSUB' #I
""");
write("HOSTA.nat", """
DEFINE DATA
LOCAL
1 #I (I2)
END-DEFINE
*
INCLUDE SHAREDBLK
END-FOR
*
END
""");
write("HOSTB.nat", """
DEFINE DATA
LOCAL
1 #I (I2)
END-DEFINE
*
INCLUDE SHAREDBLK
END-FOR
*
END
""");
write("SHAREDSUB.nat", """
DEFINE DATA
PARAMETER
1 #I (I2)
END-DEFINE
*
END
""");
given().contentType("application/json")
.body(new ProjectResource.ProjectRequest(null, root.toString(), null, "natural", null, null))
.when().post("/api/projects/" + PROJECT)
.then().statusCode(201);
given().when().post("/api/projects/" + PROJECT + "/refresh?deep=true")
.then().statusCode(200);
}
private static void write(String fileName, String content) {
try {
Files.write(root.resolve(fileName), content.getBytes(StandardCharsets.UTF_8));
} catch (IOException e) {
throw new UncheckedIOException(e);
}
}
private long count(String cypher) {
try (Session session = driver.session()) {
return session.run(cypher, Map.of("p", PROJECT)).single().get("c").asLong();
}
}
/**
* Guards against a vacuous pass: if the copycode were not expanded at all, every assertion below
* would hold trivially.
*/
@Test
void theCopycodeIsActuallyExpandedIntoBothHosts() {
List<String> owners;
try (Session session = driver.session()) {
owners = session.run("""
MATCH (n:AstNode {project: $p})
WHERE n.sourceFile ENDS WITH '.cpy'
RETURN DISTINCT n.ownerModule AS owner ORDER BY owner
""", Map.of("p", PROJECT))
.list(r -> r.get("owner").asString());
}
assertEquals(List.of("HOSTA.nat", "HOSTB.nat"), owners,
"the copycode body must be present once per including module");
}
/**
* The point of the item: no node is shared between the two includers.
*/
@Test
void eachIncluderGetsItsOwnCopyOfTheCopycodeNodes() {
long shared = count("""
MATCH (n:AstNode {project: $p})
WHERE n.sourceFile ENDS WITH '.cpy' AND n.ownerModule = ''
RETURN count(n) AS c
""");
assertEquals(0, shared, "a copycode-resident node must carry the including module as its owner");
}
/**
* A MODULE declared inside a copycode must stay shared — every module lookup binds
* {@code (project, name, sourceFile)} and never {@code ownerModule}, so an owned MODULE node
* would be invisible to them. Asserted on the whole project because the fixture has no such
* module; the invariant is what matters, and it must hold for every node of that type.
*/
@Test
void moduleAndTableNodesAreNeverOwned() {
long owned = count("""
MATCH (n:AstNode {project: $p})
WHERE n.ownerModule <> '' AND n.type IN ['MODULE', 'DB_TABLE']
RETURN count(n) AS c
""");
assertEquals(0, owned, "MODULE/DB_TABLE nodes must remain shared");
}
/**
* The reap hazard: {@code DELETE_STALE_FILE_NODES} and the item-86/106/124 edge reaps key on the
* source file, and the copycode file is in every includer's fresh-file set. Keyed on the file
* alone, refreshing HOSTA would delete HOSTB's copies (their {@code ingestGen} is a transaction
* old) together with their edges, and nothing would rebuild them.
*/
@Test
void refreshingOneHostLeavesTheOtherHostsCopiesIntact() {
String cypher = """
MATCH (n:AstNode {project: $p})
WHERE n.sourceFile ENDS WITH '.cpy' AND n.ownerModule = 'HOSTB.nat'
OPTIONAL MATCH (n)-[r]-()
RETURN count(DISTINCT n) + count(r) AS c
""";
long before = count(cypher);
assertTrue(before > 0, "HOSTB must own copycode nodes before the refresh");
given().when().post("/api/projects/" + PROJECT + "/refresh?paths=HOSTA.nat")
.then().statusCode(200);
assertEquals(before, count(cypher), "refreshing HOSTA must not touch HOSTB's copycode nodes or edges");
}
}

View File

@@ -67,10 +67,16 @@ public final class CypherQueries {
* fresh id (kept) while a renamed/removed field keeps its old id (deleted). Only files with a real
* {@code sourceFile} are reconciled; {@code sourceFile = ""} placeholders are shared across files
* and never swept here. Fixes stale identifier-index nodes that outlived a {@code refresh}.
*
* <p>Item 75-B: keyed on the {@code (sourceFile, ownerModule)} <em>pair</em>, not the file alone.
* A copycode-resident node carries the including module's file as its {@code ownerModule}, so
* several owners' nodes share one {@code sourceFile}; sweeping by file alone would delete every
* other includer's nodes (their {@code ingestGen} is one transaction old) together with their
* edges. A module's own nodes carry {@code ownerModule = ""}, so their sweep is unchanged.
*/
public static final String DELETE_STALE_FILE_NODES = """
UNWIND $sourceFiles AS sf
MATCH (n:AstNode {project: $project, sourceFile: sf})
UNWIND $files AS p
MATCH (n:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o})
WHERE n.ingestGen IS NULL OR n.ingestGen <> $ingestGen
DETACH DELETE n
""";
@@ -103,10 +109,10 @@ public final class CypherQueries {
* re-ingest, so a coarse Tier-1 (no-finalize) scan never strips resolved edges it cannot rebuild.
*/
public static final String DELETE_STALE_RESOLVED_FIELD_EDGES = """
UNWIND $sourceFiles AS f
MATCH (src:AstNode {project: $project, sourceFile: f})-[r:READS|WRITES]->(fld:AstNode)
UNWIND $files AS p
MATCH (src:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o})-[r:READS|WRITES]->(fld:AstNode)
WHERE (fld.type = 'VARIABLE' OR fld.type = 'CONSTANT')
AND fld.sourceFile <> "" AND fld.sourceFile <> f
AND fld.sourceFile <> "" AND fld.sourceFile <> p.f
DELETE r
""";
@@ -124,8 +130,8 @@ public final class CypherQueries {
* coarse re-ingest both re-emit them, so this is not gated on {@code reconcile}.
*/
public static final String DELETE_STALE_NATURAL_TABLE_ACCESS_EDGES = """
UNWIND $sourceFiles AS f
MATCH (src:AstNode {project: $project, sourceFile: f, language: 'natural'})-[r:READS|WRITES]->(t:AstNode {project: $project, sourceFile: ""})
UNWIND $files AS p
MATCH (src:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o, language: 'natural'})-[r:READS|WRITES]->(t:AstNode {project: $project, sourceFile: ""})
WHERE t.type IN ['DB_TABLE', 'WORKFILE']
DELETE r
""";
@@ -147,8 +153,8 @@ public final class CypherQueries {
* identically.
*/
public static final String DELETE_STALE_NATURAL_USING_EDGES = """
UNWIND $sourceFiles AS f
MATCH (src:AstNode {project: $project, sourceFile: f, language: 'natural'})-[r:INCLUDES]->(d:AstNode {project: $project, type: 'DATA_STRUCTURE'})
UNWIND $files AS p
MATCH (src:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o, language: 'natural'})-[r:INCLUDES]->(d:AstNode {project: $project, type: 'DATA_STRUCTURE'})
DELETE r
""";
@@ -195,16 +201,18 @@ public final class CypherQueries {
* {@code ..._CROSS} are gated on {@code resolveFields}/{@code dataflow} and so do <em>not</em> run
* in a coarse finalize. Same rationale as item 74.
*
* <p><b>Scope limit.</b> Keyed on the source node's file, so it misses the 605 call edges whose
* source subroutine is defined <em>inside a copycode</em> (14 of them stale). Those nodes are
* MERGEd per {@code (type, name, sourceFile)} and therefore <b>shared</b> by every module that
* includes the copycode, so reaping them during a single module's refresh would delete edges other
* modules contributed and not re-create them. Fixing that needs a per-module identity for
* copycode-resident nodes — a separate item, not a side effect of this one.
* <p><b>Item 75-B — the copycode gap is closed.</b> This used to key on the source node's file
* alone and therefore had to miss the 605 call edges whose source subroutine is defined
* <em>inside a copycode</em> (14 of them stale): those nodes were MERGEd per
* {@code (type, name, sourceFile)} and thus shared by every includer, so reaping during one
* module's refresh would have deleted edges the other includers contributed. Copycode-resident
* nodes now carry the includer's file as {@code ownerModule}, so the key is the
* {@code (sourceFile, ownerModule)} pair and the reap touches exactly the re-parsed module's own
* copies — which its own parse re-emits. Same for items 86 and 106 above.
*/
public static final String DELETE_STALE_NATURAL_CALL_EDGES = """
UNWIND $sourceFiles AS f
MATCH (src:AstNode {project: $project, sourceFile: f, language: 'natural'})-[r:CALLS]->(t)
UNWIND $files AS p
MATCH (src:AstNode {project: $project, sourceFile: p.f, ownerModule: p.o, language: 'natural'})-[r:CALLS]->(t)
WHERE t.sourceFile = "" OR r.callKind <> 'CALLNAT_DYNAMIC'
DELETE r
""";

View File

@@ -1641,21 +1641,41 @@ public class GraphRepository {
*/
private static String mergeKey(AstNode node, boolean positional, String project, String ownerFile) {
String base = node.type() + "" + node.name() + "" + node.sourceFile() + "" + project
+ "" + placeholderOwner(node, ownerFile);
+ "" + nodeOwner(node, ownerFile);
return positional ? base + "" + node.startLine() : base;
}
/**
* Item 76: field-level placeholders (VARIABLE/CONSTANT with sourceFile="") get a per-module
* identity via the referencing module file (ownerFile), so a field name referenced by hundreds
* of modules is no longer collapsed onto one shared node (which made finalize step 19 ~55 min on
* upms). Real nodes and MODULE/DB_TABLE/DATA_STRUCTURE placeholders keep ownerModule="" and merge
* exactly as before.
* The node's owning module file, or {@code ""} for a node that is legitimately shared. Two
* populations get a per-module identity:
*
* <ul>
* <li><b>Item 76</b> — field-level placeholders (VARIABLE/CONSTANT with {@code sourceFile=""}),
* so a field name referenced by hundreds of modules is not collapsed onto one shared node
* (which made finalize step 19 ~55 min on upms).</li>
* <li><b>Item 75-B</b> — copycode-resident nodes. {@code CopycodePreprocessor} splices a
* {@code .cpy} body into the including module before parsing and remaps the resulting nodes
* back onto the copycode file, so a node whose {@code sourceFile} differs from the parse's
* own module file came from a copycode — no extension test needed. Sharing those nodes is
* what made {@code CONTAINS} cyclic (one node accumulated every includer's nesting context)
* and what blocked the stale-edge reaps of items 124/86/106, which could not delete an edge
* other includers had contributed to the same shared node.</li>
* </ul>
*
* <p>{@code MODULE} and {@code DB_TABLE} are excluded deliberately. A module <em>can</em> be
* declared inside a copycode ({@code ZDTSTBP6} in {@code ZDTSTBC6.cpy}), and every module lookup
* binds {@code (project, name, sourceFile)} but never {@code ownerModule} — an owned MODULE node
* would be invisible to those queries and re-created beside itself on the next merge. Tables are
* shared by design.
*/
private static String placeholderOwner(AstNode node, String ownerFile) {
return node.sourceFile().isEmpty()
&& (node.type() == NodeType.VARIABLE || node.type() == NodeType.CONSTANT)
? ownerFile : "";
private static String nodeOwner(AstNode node, String ownerFile) {
if (node.type() == NodeType.MODULE || node.type() == NodeType.DB_TABLE) {
return "";
}
if (node.sourceFile().isEmpty()) {
return node.type() == NodeType.VARIABLE || node.type() == NodeType.CONSTANT ? ownerFile : "";
}
return node.sourceFile().equals(ownerFile) ? "" : ownerFile;
}
private static Map<String, @Nullable Object> edgeParams(AstEdge edge) {
@@ -2013,7 +2033,7 @@ public class GraphRepository {
params.put("name", node.name());
params.put("sourceFile", node.sourceFile());
params.put("project", project);
params.put("ownerModule", placeholderOwner(node, ownerFile));
params.put("ownerModule", nodeOwner(node, ownerFile));
params.put("language", node.language());
params.put("startLine", node.startLine());
params.put("endLine", node.endLine());
@@ -2456,8 +2476,10 @@ public class GraphRepository {
// exactly why a plain long suffices where a UUID used to be needed.
long[] nextNid = {0L};
int[] skippedEdges = {0};
// Real source files touched by this parse, for the item-58 stale-node sweep below.
Set<String> freshFiles = new HashSet<>();
// Real (sourceFile, ownerModule) pairs touched by this parse, for the item-58 stale-node sweep
// and the item-86/106/124 edge reaps below. Item 75-B: the pair, not the file alone — several
// includers' copycode nodes share one sourceFile and only the re-parsed owner's may be swept.
Set<Map<String, Object>> freshFiles = new LinkedHashSet<>();
// Item 76: owning modules whose placeholders this parse refreshed, for the placeholder sweep.
Set<String> freshPlaceholderOwners = new HashSet<>();
// Different files emit the same placeholder (e.g. a CALLNAT/USING target, sourceFile="")
@@ -2482,12 +2504,11 @@ public class GraphRepository {
Map<String, @Nullable Object> params = toParams(node, project, ownerFile);
params.put("nid", nid);
(positional ? positionalNodes : namedNodes).add(params);
String owner = nodeOwner(node, ownerFile);
if (!node.sourceFile().isEmpty()) {
freshFiles.add(node.sourceFile());
}
String phOwner = placeholderOwner(node, ownerFile);
if (!phOwner.isEmpty()) {
freshPlaceholderOwners.add(phOwner);
freshFiles.add(Map.of("f", node.sourceFile(), "o", owner));
} else if (!owner.isEmpty()) {
freshPlaceholderOwners.add(owner);
}
} else {
nidByParserId.put(parserId, canonical);
@@ -2523,11 +2544,11 @@ public class GraphRepository {
// current edges; unchanged ones round-trip identically.
if (!freshFiles.isEmpty()) {
tx.run(CypherQueries.DELETE_STALE_NATURAL_TABLE_ACCESS_EDGES,
Map.of("project", project, "sourceFiles", List.copyOf(freshFiles)));
Map.of("project", project, "files", List.copyOf(freshFiles)));
// Item 106: same for USING edges, which resolve onto REAL data-area nodes and so are not
// covered by the placeholder-only reap above.
tx.run(CypherQueries.DELETE_STALE_NATURAL_USING_EDGES,
Map.of("project", project, "sourceFiles", List.copyOf(freshFiles)));
Map.of("project", project, "files", List.copyOf(freshFiles)));
// Item 124: same for CALLS. A parser fix that changes a call's target leaves both endpoints
// alive (the calling subroutine is unchanged; the old target is a never-swept placeholder),
// so the corrected call was only ever added beside the wrong one. Deep re-ingest only —
@@ -2535,7 +2556,7 @@ public class GraphRepository {
// this deletes.
if (reconcile) {
tx.run(CypherQueries.DELETE_STALE_NATURAL_CALL_EDGES,
Map.of("project", project, "sourceFiles", List.copyOf(freshFiles)));
Map.of("project", project, "files", List.copyOf(freshFiles)));
}
}
for (Map.Entry<EdgeType, List<Map<String, @Nullable Object>>> entry : edgesByType.entrySet()) {
@@ -2547,7 +2568,7 @@ public class GraphRepository {
// statements) so a refresh purges stale nodes instead of leaving them to shadow the new ones.
if (reconcile && !freshFiles.isEmpty()) {
tx.run(CypherQueries.DELETE_STALE_FILE_NODES, Map.of("project", project,
"sourceFiles", List.copyOf(freshFiles), "ingestGen", ingestGen));
"files", List.copyOf(freshFiles), "ingestGen", ingestGen));
}
// Item 76: sweep per-module field placeholders (sourceFile="") that a re-ingested owner file
// no longer produces. The file sweep above skips sourceFile="" nodes; without this an
@@ -2567,7 +2588,7 @@ public class GraphRepository {
// follows and re-links them.
if (reconcile && !freshFiles.isEmpty()) {
tx.run(CypherQueries.DELETE_STALE_RESOLVED_FIELD_EDGES,
Map.of("project", project, "sourceFiles", List.copyOf(freshFiles)));
Map.of("project", project, "files", List.copyOf(freshFiles)));
}
}

View File

@@ -272,6 +272,9 @@ public final class NaturalCoarseScanner implements CoarseScanner {
// are attributed exactly as NaturalParser attributes them (caller = currentFunction ?: module).
@Nullable AstNode currentFunction = null;
boolean inDefineData = false;
// Item 75: inside an INIT<...>/CONST<...> value clause that spans several lines. Must mirror the
// deep parser exactly (item 59) or the coarse identifier index diverges from the deep parse.
boolean inMultiLineValue = false;
for (int i = 0; i < lines.length; i++) {
String line = stripInlineComment(lines[i]);
int lineNo = i + 1;
@@ -292,6 +295,17 @@ public final class NaturalCoarseScanner implements CoarseScanner {
continue;
}
if (inDefineData) {
// Item 75: skip the continuation lines of a multi-line INIT<...>/CONST<...> — they are
// values, not declarations, but they tokenize as one (`166 ,` -> level 166, name ",").
if (inMultiLineValue) {
if (NaturalFieldTokenizer.closesMultiLineValue(line)) {
inMultiLineValue = false;
}
continue;
}
if (NaturalFieldTokenizer.opensMultiLineValue(line)) {
inMultiLineValue = true;
}
// Copybook include (PARAMETER/LOCAL USING): a coarse INCLUDES reference to the data area.
Matcher include = INCLUDE.matcher(line);
if (include.find()) {

View File

@@ -40,6 +40,47 @@ public final class NaturalFieldTokenizer {
private static final Pattern CONST_VALUE = Pattern.compile("(?i)CONST\\s*<\\s*([^>]*)>");
private static final Pattern INIT_VALUE = Pattern.compile("(?i)INIT\\s*<\\s*([^>]*)>");
/**
* Opening of a {@code CONST<...>} / {@code INIT<...>} value clause, used to recognise the
* multi-line form (item 75).
*/
private static final Pattern VALUE_CLAUSE_OPEN = Pattern.compile("(?i)(?:CONST|INIT)\\s*<");
/**
* True when this line opens a {@code CONST<...>}/{@code INIT<...>} whose closing {@code >} is on a
* later line, e.g. an initialiser listing one value per line. The continuation lines are <em>not</em>
* field declarations, but they look like one to {@link #parse(String)} — {@code 166 , /*CREDIT NOTE}
* tokenizes as level 166, name {@code ","} — which is where the {@code DATA_STRUCTURE} named
* {@code ","} in the graph came from (item 75). A caller inside {@code DEFINE DATA} should skip every
* following line until {@link #closesMultiLineValue(String)} reports the clause closed.
*
* <p>The single-line form (the normal case) never triggers this: its {@code <} and {@code >} balance
* on the same line.
*/
public static boolean opensMultiLineValue(String line) {
return VALUE_CLAUSE_OPEN.matcher(line).find() && angleBalance(line) > 0;
}
/**
* True when this line closes a value clause left open by {@link #opensMultiLineValue(String)}.
*/
public static boolean closesMultiLineValue(String line) {
return angleBalance(line) < 0;
}
private static int angleBalance(String line) {
int balance = 0;
for (int i = 0; i < line.length(); i++) {
char c = line.charAt(i);
if (c == '<') {
balance++;
} else if (c == '>') {
balance--;
}
}
return balance;
}
private NaturalFieldTokenizer() {
}

View File

@@ -200,6 +200,16 @@ public final class NaturalParser implements LanguageParser {
// token of the rest — `V 1VDB2-VERSIS_GENAGREE VERSVW_GENAGREE DA:00,00ED:00.00PF:`. There is no
// `VIEW OF` text here, so VIEW_OF (which matches the source form) never fires on these files. `V` is
// not a Natural data type, so the prefix is unambiguous.
// Item 75: a Natural *filler* is declared `<level><byteCount>X` with no name — `489X` is level 4, 89
// bytes of padding. Neither data-area pattern can see that: the occurrence/length column swallows part
// of the digits, so each such line produced a DATA_STRUCTURE literally named "X" at a different (bogus)
// level. The (type,name,sourceFile) merge key then collapsed all of them onto a single node, which
// contained itself — the CONTAINS self-loops of item 75 — and, because a swallowed level pops the whole
// group stack, the fields declared after a filler were silently re-parented under it (SNA27R01.pda:
// C63P0027-GESCHL hung under "X" instead of its real group). At least one count digit is required, so a
// field genuinely named X (`5X` = level 5, name X) is still parsed as a field.
private static final Pattern DATA_AREA_FILLER = Pattern.compile("^\\s*\\d{2,}X\\s*(?:/\\*.*)?$");
private static final Pattern DATA_AREA_VIEW_DDM = Pattern.compile("^([A-Za-z][\\w-]*)");
// Item 45: the XML-emit idiom `COMPRESS '<' <tagVar> '>' <valueVar> ...` inside an ADD-XML-LINE
// subroutine — the '<'/'>' literals wrapping a tag variable then a value variable identify the
@@ -890,7 +900,7 @@ public final class NaturalParser implements LanguageParser {
private static List<String> topLevelNames(String[] lines) {
List<String> names = new ArrayList<>();
for (String line : lines) {
if (DATA_AREA_COMMENT.matcher(line).find()) {
if (DATA_AREA_COMMENT.matcher(line).find() || DATA_AREA_FILLER.matcher(line).matches()) {
continue;
}
// Item 99: must use the same split as parseDataArea — a phantom level-1 group here suppresses
@@ -939,7 +949,7 @@ public final class NaturalParser implements LanguageParser {
String line = lines[i];
int lineNo = i + 1;
if (DATA_AREA_COMMENT.matcher(line).find()) {
if (DATA_AREA_COMMENT.matcher(line).find() || DATA_AREA_FILLER.matcher(line).matches()) {
continue;
}
// Item 99: the anchored export grammar first, the permissive pattern as a fallback.
@@ -1118,6 +1128,8 @@ public final class NaturalParser implements LanguageParser {
}
boolean inDefineData = false;
// Item 75: inside an INIT<...>/CONST<...> value clause that spans several lines.
boolean inMultiLineValue = false;
String defineScope = "";
int paramPosition = 0;
Deque<LevelEntry> groupStack = new ArrayDeque<>();
@@ -1158,6 +1170,19 @@ public final class NaturalParser implements LanguageParser {
}
if (inDefineData) {
// Item 75: the continuation lines of a multi-line INIT<...>/CONST<...> are values, not
// field declarations, but they tokenize as one (`166 ,` -> level 166, name ","). Skip
// them until the clause closes.
if (inMultiLineValue) {
if (NaturalFieldTokenizer.closesMultiLineValue(line)) {
inMultiLineValue = false;
}
continue;
}
if (NaturalFieldTokenizer.opensMultiLineValue(line)) {
inMultiLineValue = true;
}
Matcher scopeMatcher = SCOPE_KEYWORD.matcher(line);
if (scopeMatcher.matches()) {
defineScope = scopeMatcher.group(1).toUpperCase(Locale.ROOT);

View File

@@ -348,4 +348,28 @@ class NaturalCoarseScannerTest {
.count(),
"UPDATE(label.) and DELETE(label.) each write the loop's table");
}
/**
* Item 75 / item 59: the coarse identifier index must skip the continuation lines of a multi-line
* {@code INIT<...>} exactly as the deep parser does — otherwise {@code 166 ,} enters the index as an
* identifier named {@code ","} and the two tiers disagree on the file's fields.
*/
@Test
void multiLineInitValuesDoNotEnterTheIdentifierIndex() {
String src = """
DEFINE DATA LOCAL
1 #V-LITERALES (N6/1:3)
INIT< 165 ,
166 ,
167
>
1 #TXT-TEXTO (A80)
END-DEFINE
END
""";
ParseResult r = scanner.scan("PGM.nat", src);
assertFalse(hasNode(r, NodeType.VARIABLE, ","), "INIT continuation values are not identifiers");
assertTrue(hasNode(r, NodeType.VARIABLE, "#V-LITERALES"));
assertTrue(hasNode(r, NodeType.VARIABLE, "#TXT-TEXTO"), "declaration after the closing > is indexed");
}
}

View File

@@ -413,6 +413,84 @@ class NaturalParserTest {
assertEquals("N8", datStart.dataType());
}
/**
* Item 75: a Natural filler declaration ({@code <level><byteCount>X}, here {@code 51X}) is unnamed
* padding and must produce no node at all. It used to be read as a field literally named {@code X} at
* a bogus level 1, which popped the whole group stack — so every field after it was re-parented under
* that phantom, and because all fillers of a file collapsed onto one node by the merge key, the node
* ended up containing itself (the CONTAINS self-loops of item 75). A field genuinely named {@code X}
* ({@code 3X}) must still be parsed, which is why the filler pattern requires a byte count.
*/
@Test
void fillerDeclarationProducesNoNodeAndDoesNotReparentFollowingFields() throws IOException {
String content = readFixture("WGEAGL01_SAMPLE.pda");
LanguageParser.ParseResult result = parser.parse("WGEAGL01_SAMPLE.pda", content);
assertFalse(hasNode(result, NodeType.DATA_STRUCTURE, "X"), "filler 51X must not become a group");
assertTrue(result.edges().stream().noneMatch(e -> e.type() == EdgeType.CONTAINS
&& e.sourceId().equals(e.targetId())), "no self-containment");
// the field declared after the filler keeps its real parent (the REDEFINE group), not the filler
AstNode redefineGroup = findNode(result, NodeType.DATA_STRUCTURE, "#P-ADD-PARM");
AstNode codTypesys = findNode(result, NodeType.VARIABLE, "COD-TYPESYS");
assertEquals(redefineGroup.name(), parentName(result, codTypesys));
assertTrue(hasEdge(result, EdgeType.CONTAINS, redefineGroup, "COD-TYPESYS", NodeType.VARIABLE));
// a real field named X (no byte count) is still a field
AstNode fieldNamedX = findNode(result, NodeType.VARIABLE, "X");
assertEquals("A2", fieldNamedX.dataType());
assertEquals(redefineGroup.name(), parentName(result, fieldNamedX));
}
/**
* Item 75: the continuation lines of a multi-line {@code INIT<...>} are values, not declarations, but
* they tokenize as one — {@code 166 ,} reads as level 166, name {@code ","} — which produced a
* DATA_STRUCTURE named {@code ","} that contained itself. The declarations after the closing
* {@code >} must still be parsed.
*/
@Test
void multiLineInitValuesAreNotParsedAsFields() {
String content = """
DEFINE DATA LOCAL
1 #V-LITERALES (N6/1:3)
INIT< 165 ,
166 ,
167
>
1 #TXT-TEXTO (A80)
END-DEFINE
END
""";
LanguageParser.ParseResult result = parser.parse("MULTI_INIT.nat", content);
assertFalse(hasNode(result, NodeType.DATA_STRUCTURE, ","), "INIT continuation must not become a field");
assertTrue(result.edges().stream().noneMatch(e -> e.type() == EdgeType.CONTAINS
&& e.sourceId().equals(e.targetId())), "no self-containment");
assertTrue(hasNode(result, NodeType.VARIABLE, "#V-LITERALES"));
assertTrue(hasNode(result, NodeType.VARIABLE, "#TXT-TEXTO"), "declaration after the closing > is parsed");
}
/**
* The single-line form — by far the common one — must be unaffected by the multi-line skip.
*/
@Test
void singleLineInitIsUnaffected() {
String content = """
DEFINE DATA LOCAL
1 #A (A1) INIT<'X'>
1 #B (A1)
END-DEFINE
END
""";
LanguageParser.ParseResult result = parser.parse("SINGLE_INIT.nat", content);
assertTrue(hasNode(result, NodeType.VARIABLE, "#A"));
assertTrue(hasNode(result, NodeType.VARIABLE, "#B"));
}
@Test
void bareAssignWithoutKeywordIsRecognized() {
String content = """

View File

@@ -5,7 +5,9 @@
A 250 2#P-ADD-PARM
R 2#P-ADD-PARM /* BEGIN REDEFINE: #P-ADD-PARM
A 8 3COD-EMISOR
51X
A 8 3COD-TYPESYS
A 8 3COD-GENAGREE
A 50 3PHON-GENAGREE-1
N 8 3DAT-START
A 2 3X

View File

@@ -1360,6 +1360,39 @@ before. The parked "parallel parse phase" idea was implemented 2026-07-18 (item
(parser artefacts) still exist, but both the finalize and the read consumers are now bounded — item 75
no longer has a practical impact.
**2026-08-21 — cause established and half of it fixed (parser artefacts, scope A).** The entry above
said "cause NOT established"; it is now. There are exactly **two** families, on unrelated code paths:
* **A — parser artefacts (fixed).** A Natural *filler* is declared `<level><byteCount>X` with no name
(`489X` = level 4, 89 bytes). `DATA_AREA_FIELD_EXPORT`'s occurrence/length column swallows part of the
digits, so each filler line produced a `DATA_STRUCTURE` literally named `X` at a *different, bogus*
level; the `(type, name, sourceFile)` merge key collapsed all of a file's fillers onto **one** node,
which then contained itself. 27 such nodes in `upms` — **20 of the 24 self-loops**. The `","` node
(`BSUPLFN0.nat` 11x, `JB0067N0.nat` 5x) is the same bug on the `.nat` source path: the continuation
lines of a multi-line `INIT<...>` (`166 , /*CREDIT NOTE`) tokenize as level 166, name `","`.
**Not cosmetic** — a swallowed level pops the whole group stack, so the fields *after* a filler were
silently re-parented under it: in `SNA27R01.pda`, `C63P0027-GESCHL` (line 28) hung under `X` instead of
its real group. Fix: `NaturalParser.DATA_AREA_FILLER` skips filler lines on both data-area call sites
(it requires at least one count digit, so a field genuinely named `X` still parses), and
`NaturalFieldTokenizer.opensMultiLineValue()/closesMultiLineValue()` let both ingest tiers skip
`INIT<`/`CONST<` continuation lines — the grammar decision lives in the shared tokenizer so the two
tiers cannot drift (item 59). Guarded by three tests that fail on the pre-fix parser —
`NaturalParserTest`'s `fillerDeclarationProducesNoNodeAndDoesNotReparentFollowingFields` and
`multiLineInitValuesAreNotParsedAsFields`, plus
`NaturalCoarseScannerTest.multiLineInitValuesDoNotEnterTheIdentifierIndex` — and
`NaturalParserTest.singleLineInitIsUnaffected`, which pins the common single-line `INIT<...>` form
against the new skip. The stale nodes leave the graph on the next deep refresh of `upms`.
* **B — shared copycode nodes (open, own round).** The candidate above is confirmed: `YFRAMBC0.cpy`
opens a `FOR` at line 65 and ends at 68 (the `END-FOR` is in the includer), so the one shared node
accumulates every includer's nesting context — hence the 2-cycles with `JX0031N0.nat:781/828/874/…`
and the `65 -> 15` range. Same in `YFRAMBC4/CH/CI/CM.cpy` and `JX9901C6.cpy`; `JX0030C2.cpy:37 <-> 41`
is the intra-file variant. **Sized for the first time:** 1185 shared `.cpy` nodes, avg 13.7 includers,
max 773 -> a per-includer identity (`ownerModule`, as item 76 does for field placeholders) yields
~16,199 nodes, **+3 %** on `upms`'s 459,093 — affordable. This also closes items 124/86/106.
Drift measured on the current graph (vs. the 2026-07-17 numbers in the heading): self-loops 22 -> **24**,
two-cycles 162 -> **21**, inverted ranges 625 -> **383**. Scope A removes 20 self-loops and 2 two-cycles;
the item stays open for **B**, so a corpus-wide `CONTAINS`-acyclicity assertion is not yet possible —
the guards above are deliberately fixture-level.
- [ ] **73. A `NONE`/`ANY` branch is reported under its enclosing guard alone — the condition is a
negation no guard chain can express** (found 2026-07-17 while fixing item 72; **item 72 does not fix
this**). `NaturalParser`'s `VALUE_RESET` clears a `DECIDE`'s active value on `NONE`/`ANY`, so an