Files
agenticCode/ac-code-server/src/main/java/com/agenticcode/codeserver/service/AstIngestService.java
Ingo Schnabel 10971915e9 Typescript
2026-09-23 08:15:53 +02:00

322 lines
15 KiB
Java

package com.agenticcode.codeserver.service;
import com.agenticcode.neo4jstore.graph.*;
import com.agenticcode.parsercore.ast.model.AstEdge;
import com.agenticcode.parsercore.ast.model.AstNode;
import com.agenticcode.parsercore.ast.model.LocMetrics;
import com.agenticcode.parsercore.ast.model.NodeType;
import com.agenticcode.parsercore.ast.spi.CoarseScanner;
import com.agenticcode.parsercore.ast.spi.LanguageParser;
import com.agenticcode.parsercore.ast.spi.LineCounter;
import com.agenticcode.parserjava.JavaCoarseScanner;
import com.agenticcode.parserjava.JavaLineCounter;
import com.agenticcode.parserjava.JavaParser;
import com.agenticcode.parsernatural.CopycodeResolver;
import com.agenticcode.parsernatural.NaturalCoarseScanner;
import com.agenticcode.parsernatural.NaturalLineCounter;
import com.agenticcode.parsernatural.NaturalParser;
import com.agenticcode.parsertypescript.CssLineCounter;
import com.agenticcode.parsertypescript.TypeScriptCoarseScanner;
import com.agenticcode.parsertypescript.TypeScriptLineCounter;
import com.agenticcode.parsertypescript.TypeScriptParser;
import com.agenticcode.parsertypescript.TypeScriptProject;
import io.smallrye.mutiny.Uni;
import jakarta.enterprise.context.ApplicationScoped;
import org.eclipse.microprofile.config.inject.ConfigProperty;
import org.jspecify.annotations.Nullable;
import java.util.*;
import java.util.stream.Collectors;
/**
* Parses a source file with the language-specific parser and persists the resulting
* AST nodes/edges to Neo4j. Parsing and persistence are split so the orchestrating
* {@link ProjectIngestService} can collect parse results, detect duplicates across a
* project root, and only then save the survivors.
*/
@ApplicationScoped
public class AstIngestService {
private final GraphRepository graphRepository;
private final LanguageParser javaParser = new JavaParser();
private final NaturalParser naturalParser = new NaturalParser();
private final CoarseScanner javaScanner = new JavaCoarseScanner();
private final NaturalCoarseScanner naturalScanner = new NaturalCoarseScanner();
private final LineCounter javaLineCounter = new JavaLineCounter();
private final LineCounter naturalLineCounter = new NaturalLineCounter();
// Item 192: one parser for .ts/.tsx/.css; the per-ingest TypeScriptProject carries workspaces,
// package names and (Tier-2) the sidecar facts, like CopycodeResolver does for Natural.
private final TypeScriptParser typeScriptParser = new TypeScriptParser();
private final TypeScriptCoarseScanner typeScriptScanner = new TypeScriptCoarseScanner();
private final LineCounter typeScriptLineCounter = new TypeScriptLineCounter();
private final LineCounter cssLineCounter = new CssLineCounter();
/**
* Item 179 (DIAGNOSTIC): when false, {@link NodeType#COMMENT} nodes and their {@code DOCUMENTS}
* edges are dropped just before persist. Comments are 45.8 % of the `upms` node population and
* 60.4 % of everything {@code merge-nodes} processes, and this switch exists to measure what
* that actually costs. It is deliberately a config property and not a query parameter: it is an
* experiment, not a feature, so it gets no REST or CLI surface.
*
* <p>Turning it off makes a full parse emit no comments, so a reconciling run <em>deletes</em>
* the existing ones — the first run after a flip therefore pays a one-time sweep and must not be
* used as a measurement. Compare steady-state runs only.
*/
private final boolean commentsEnabled;
public AstIngestService(GraphRepository graphRepository,
@ConfigProperty(name = "agenticcode.ingest.comments.enabled",
defaultValue = "true") boolean commentsEnabled) {
this.graphRepository = graphRepository;
this.commentsEnabled = commentsEnabled;
}
/**
* Item 179: strips comment nodes and the edges touching them. Applied at the persist seam rather
* than in the parsers, so the parse cost stays in both arms of the A/B and the delta isolates
* persist — which is the whole question, since finalize never touches a comment node.
*/
private LanguageParser.ParseResult stripComments(LanguageParser.ParseResult result) {
Set<UUID> commentIds = result.nodes().stream()
.filter(n -> n.type() == NodeType.COMMENT)
.map(AstNode::id)
.collect(Collectors.toSet());
if (commentIds.isEmpty()) {
return result;
}
List<AstNode> nodes = result.nodes().stream()
.filter(n -> n.type() != NodeType.COMMENT)
.toList();
List<AstEdge> edges = result.edges().stream()
.filter(e -> !commentIds.contains(e.sourceId()) && !commentIds.contains(e.targetId()))
.toList();
return new LanguageParser.ParseResult(nodes, edges);
}
/**
* @return the unresolved cross-file dependencies (CALLNAT/PERFORM/EXTENDS/IMPLEMENTS targets and
* INCLUDE/USING'd data areas) of a parse result, identified by the placeholder nodes
* ({@code sourceFile == ""}) the parsers emit, so a caller can recursively ingest them.
*/
public static List<DependencyRef> dependencies(LanguageParser.ParseResult result) {
return result.nodes().stream()
.filter(node -> node.sourceFile().isEmpty()
&& (node.type() == NodeType.MODULE || node.type() == NodeType.DATA_STRUCTURE))
.map(node -> new DependencyRef(node.name(), node.type()))
.distinct()
.toList();
}
/**
* Parses {@code content} with the parser for {@code language}. Does not persist anything.
*/
public LanguageParser.ParseResult parse(SourceFiles.Language language, String sourceFile, String content) {
return parse(language, sourceFile, content, CopycodeResolver.NONE);
}
/**
* Parses {@code content}, expanding Natural copycode {@code INCLUDE}s via {@code copycodes}
* (item 46a) so a copycode's calls/DB access/dataflow surface on the including module. The Java
* parser has no copycode concept and ignores the resolver.
*/
public LanguageParser.ParseResult parse(SourceFiles.Language language, String sourceFile, String content,
CopycodeResolver copycodes) {
return parse(language, sourceFile, content, copycodes, TypeScriptProject.NONE);
}
/**
* Parses {@code content} with the parser for {@code language}; {@code copycodes} serves Natural
* (item 46a), {@code typescript} serves TypeScript/CSS (item 192). The switch is exhaustive on
* purpose: a new {@link SourceFiles.Language} must be routed here, not fall through to a default.
*/
public LanguageParser.ParseResult parse(SourceFiles.Language language, String sourceFile, String content,
CopycodeResolver copycodes, TypeScriptProject typescript) {
return switch (language) {
case JAVA -> javaParser.parse(sourceFile, content);
case NATURAL -> naturalParser.parse(sourceFile, content, copycodes);
case TYPESCRIPT, CSS -> typeScriptParser.parse(sourceFile, content, typescript);
};
}
/**
* Tier-1 coarse scan (item 36): the cheap lexer-level outline of {@code content} — module/function
* shells, identifier index and coarse call/DB/include references (with a {@code sourceHash}) but no
* deep bodies. Persisted like a {@link #parse} result; the deep detail is filled in on demand.
*/
public LanguageParser.ParseResult coarseScan(SourceFiles.Language language, String sourceFile, String content) {
return coarseScan(language, sourceFile, content, CopycodeResolver.NONE);
}
/**
* Tier-1 coarse scan with Natural copycode expansion (item 46a) for call/DB visibility; the Java
* scanner ignores the resolver.
*/
public LanguageParser.ParseResult coarseScan(SourceFiles.Language language, String sourceFile, String content,
CopycodeResolver copycodes) {
return coarseScan(language, sourceFile, content, copycodes, TypeScriptProject.NONE);
}
public LanguageParser.ParseResult coarseScan(SourceFiles.Language language, String sourceFile, String content,
CopycodeResolver copycodes, TypeScriptProject typescript) {
return switch (language) {
case JAVA -> javaScanner.scan(sourceFile, content);
case NATURAL -> naturalScanner.scan(sourceFile, content, copycodes);
case TYPESCRIPT, CSS -> typeScriptScanner.scan(sourceFile, content, typescript);
};
}
/**
* Item 44: deterministic per-language LoC/SLoC of {@code content}. Uses the same {@link LineCounter}
* the Tier-1 coarse scanners use, so a module's metrics are identical at any ingest depth.
*/
public LocMetrics count(SourceFiles.Language language, String content) {
LineCounter counter = switch (language) {
case JAVA -> javaLineCounter;
case NATURAL -> naturalLineCounter;
case TYPESCRIPT -> typeScriptLineCounter;
case CSS -> cssLineCounter;
};
return counter.count(content);
}
/**
* Persists a parse result's nodes and edges only (no enrichment). Callers must run
* {@link #finalizeProject} once after persisting a batch of files.
*/
public Uni<Void> persist(String project, LanguageParser.ParseResult result) {
return persist(project, result, false);
}
/**
* As {@link #persist(String, LanguageParser.ParseResult)}, but when {@code reconcile} is true also
* deletes the file's stale nodes this (full) parse no longer produces (item 58). Pass {@code false}
* for a coarse Tier-1 scan.
*/
public Uni<Void> persist(String project, LanguageParser.ParseResult result, boolean reconcile) {
return graphRepository.persist(project, commentsEnabled ? result : stripComments(result),
reconcile).replaceWithVoid();
}
/**
* Persists several files' parse results in one batched transaction (no enrichment). Callers
* must run {@link #finalizeProject} once after persisting all batches.
*/
public Uni<Void> persistBatch(String project, List<LanguageParser.ParseResult> results) {
return persistBatch(project, results, false);
}
/**
* As {@link #persistBatch(String, List)}, but when {@code reconcile} is true also deletes each
* file's stale nodes this (full) parse no longer produces (item 58). Pass {@code false} for a
* coarse Tier-1 scan.
*/
public Uni<Void> persistBatch(String project, List<LanguageParser.ParseResult> results, boolean reconcile) {
List<LanguageParser.ParseResult> effective = commentsEnabled
? results
: results.stream().map(this::stripComments).toList();
return graphRepository.persistBatch(project, effective, reconcile).replaceWithVoid();
}
/**
* Item 62 (summary surface): which of {@code names} are real data fields of {@code project} —
* asked project-wide, because a data literal is routinely declared outside the ingested tree.
*/
public Uni<Set<String>> dataFieldNames(String project, Collection<String> names) {
return graphRepository.dataFieldNames(project, names);
}
/**
* Item 43: every distinct real source file that currently has a node in {@code project} — the
* caller checks these against the filesystem to find files deleted on disk whose nodes linger.
*/
public Uni<Set<String>> distinctSourceFiles(String project) {
return graphRepository.distinctSourceFiles(project);
}
/**
* Item 126: records what the last <b>whole-root</b> ingest of {@code project} did, so an agent can
* date an answer and see how complete the graph is without crawling the file system.
*/
public Uni<Void> recordProjectIngest(String project, ProjectIngestInfo ingest) {
return graphRepository.recordProjectIngest(project, ingest);
}
/**
* Item 129: marks a whole-root ingest as in flight, so one that never finishes leaves the graph
* visibly half-updated rather than looking clean.
*/
public Uni<Void> markProjectIngestStarted(String project, String mode, String startedAt) {
return graphRepository.markProjectIngestStarted(project, mode, startedAt);
}
/**
* Item 129: every ingested file's stored content hash, the input to a {@code changedOnly} refresh's
* skip decision.
*/
public Uni<java.util.Map<String, String>> sourceHashes(String project) {
return graphRepository.sourceHashes(project);
}
/**
* Item 43: deletes every node of {@code project} belonging to one of {@code sourceFiles} — the
* deleted-file orphan sweep run after a whole-project refresh.
*/
public Uni<Void> deleteNodesForSourceFiles(String project, Collection<String> sourceFiles) {
return graphRepository.deleteNodesForSourceFiles(project, sourceFiles);
}
/**
* Runs the full project-wide enrichment once after a batch of {@link #persist}s.
*/
public Uni<Void> finalizeProject(String project) {
return graphRepository.finalizeProject(project).replaceWithVoid();
}
/**
* Runs enrichment after a batch of {@link #persist}s, to the given {@link EnrichmentLevel}
* (field-level resolution runs only for {@link EnrichmentLevel#FULL}).
*/
public Uni<Void> finalizeProject(String project, EnrichmentLevel level) {
return finalizeProject(project, level, false);
}
/**
* @param profile diagnostic: profile the slow enrichment steps (see {@code GraphRepository}).
*/
public Uni<Void> finalizeProject(String project, EnrichmentLevel level, boolean profile) {
return graphRepository.finalizeProject(project, level, profile).replaceWithVoid();
}
/**
* Deep-enriches only the given program tree's field references (see
* {@link GraphRepository#finalizeProjectScoped}) — fast even when the project already holds the
* whole call graph.
*/
public Uni<Void> finalizeProjectScoped(String project, List<String> moduleNames) {
return graphRepository.finalizeProjectScoped(project, moduleNames).replaceWithVoid();
}
/**
* Tags ingested modules with their {@link IngestDepth} (see
* {@link GraphRepository#markIngestDepth}).
*/
public Uni<Void> markIngestDepth(String project, IngestDepth depth, @Nullable List<String> names) {
return graphRepository.markIngestDepth(project, depth, names).replaceWithVoid();
}
/**
* Item 114: persists the identities a whole-project walk skipped as duplicates. Whole-root walks
* only — see {@link GraphRepository#markDuplicateIdentities}.
*/
public Uni<Void> markDuplicateIdentities(String project, List<Map<String, Object>> duplicates) {
return graphRepository.markDuplicateIdentities(project, duplicates).replaceWithVoid();
}
/**
* @return the ingest state of a module (exists? deeply ingested?).
*/
public Uni<ModuleIngestState> moduleIngestState(String project, String name) {
return graphRepository.moduleIngestState(project, name);
}
}