ADR-038 Appendix: JavaCodeIngestor Design

Overview

This appendix details the implementation of JavaCodeIngestor for extracting knowledge from iDempiere Java source code into the RAG vector database.

Dependencies

<!-- JavaParser for AST parsing -->
<dependency>
    <groupId>com.github.javaparser</groupId>
    <artifactId>javaparser-core</artifactId>
    <version>3.27.1</version>
</dependency>

<!-- Optional: Symbol resolution for type inference -->
<dependency>
    <groupId>com.github.javaparser</groupId>
    <artifactId>javaparser-symbol-solver-core</artifactId>
    <version>3.27.1</version>
</dependency>

Reference: JavaParser GitHub, Baeldung Tutorial

Architecture

┌─────────────────────────────────────────────────────────────────────────────┐
│                         JavaCodeIngestor                                     │
├─────────────────────────────────────────────────────────────────────────────┤
│                                                                              │
│  Input: /path/to/idempiere/org.adempiere.base/src/                          │
│                                                                              │
│  ┌─────────────────────────────────────────────────────────────────────┐    │
│  │ 1. FILE DISCOVERY                                                    │    │
│  │    - Glob: **/M*.java (model classes)                               │    │
│  │    - Glob: **/*Process.java (processes)                             │    │
│  │    - Glob: **/*Callout.java (callouts)                              │    │
│  │    - Glob: **/*Validator.java (validators)                          │    │
│  └─────────────────────────────────────────────────────────────────────┘    │
│                              │                                               │
│                              ▼                                               │
│  ┌─────────────────────────────────────────────────────────────────────┐    │
│  │ 2. AST PARSING (JavaParser)                                         │    │
│  │    - StaticJavaParser.parse(file)                                   │    │
│  │    - Extract: class name, package, extends, implements              │    │
│  │    - Extract: fields, methods, annotations                          │    │
│  └─────────────────────────────────────────────────────────────────────┘    │
│                              │                                               │
│                              ▼                                               │
│  ┌─────────────────────────────────────────────────────────────────────┐    │
│  │ 3. INTELLIGENT CHUNKING                                             │    │
│  │    - Model classes: beforeSave, afterSave, beforeDelete             │    │
│  │    - Processes: prepare, doIt                                       │    │
│  │    - Callouts: individual callout methods                           │    │
│  │    - Validators: modelChange, docValidate                           │    │
│  │    - Extract column references from code                            │    │
│  └─────────────────────────────────────────────────────────────────────┘    │
│                              │                                               │
│                              ▼                                               │
│  ┌─────────────────────────────────────────────────────────────────────┐    │
│  │ 4. LOGIC EXTRACTION                                                 │    │
│  │    - Validation: throw new AdempiereException, FillMandatory        │    │
│  │    - Calculations: arithmetic operations, setters                   │    │
│  │    - Triggers: method calls that update other tables                │    │
│  │    - Read-only: isProcessed(), isPosted() checks                    │    │
│  │    - Column references: get/set + COLUMNNAME_*                      │    │
│  └─────────────────────────────────────────────────────────────────────┘    │
│                              │                                               │
│                              ▼                                               │
│  ┌─────────────────────────────────────────────────────────────────────┐    │
│  │ 5. EMBEDDING + STORAGE                                              │    │
│  │    - Generate embedding for each chunk                              │    │
│  │    - Store with rich metadata (class, method, columns, logic_type)  │    │
│  │    - pgvector: embedding + metadata JSON                            │    │
│  └─────────────────────────────────────────────────────────────────────┘    │
│                                                                              │
└─────────────────────────────────────────────────────────────────────────────┘

Chunking Strategy

Model Classes (M*.java)

// Key methods to extract from model classes
public class ModelMethodExtractor {

    // These methods contain business logic
    static final List<String> KEY_METHODS = List.of(
        "beforeSave",      // Validation, calculations before save
        "afterSave",       // Triggers, cascade updates
        "beforeDelete",    // Validation before delete
        "afterDelete",     // Cleanup after delete
        "completeIt",      // Document completion logic
        "voidIt",          // Void document logic
        "closeIt",         // Close document logic
        "reverseCorrectIt",// Reversal logic
        "prepareIt"        // Document preparation
    );

    // Column reference patterns to detect
    static final Pattern COLUMN_GETTER = Pattern.compile("get(\\w+)\\(\\)");
    static final Pattern COLUMN_SETTER = Pattern.compile("set(\\w+)\\(");
    static final Pattern COLUMNNAME_REF = Pattern.compile("COLUMNNAME_(\\w+)");
}

Process Classes (*Process.java, SvrProcess)

// Key methods for process classes
static final List<String> PROCESS_METHODS = List.of(
    "prepare",    // Parameter setup
    "doIt"        // Main execution logic
);

// Parameter handling patterns
static final Pattern PARAM_GET = Pattern.compile("getParameter\\w*\\(\"(\\w+)\"\\)");
static final Pattern PARAM_ID = Pattern.compile("p_(\\w+)_ID");

Callout Classes

// Callout methods follow pattern: public String methodName(...)
// Extract JavaDoc + method body for each callout

Metadata Schema

Each code chunk is stored with rich metadata for filtering:

{
  "source_type": "java_code",
  "file_path": "org.adempiere.base/src/org/compiere/model/MOrderLine.java",
  "package": "org.compiere.model",
  "class_name": "MOrderLine",
  "class_type": "model",
  "extends": "X_C_OrderLine",
  "table_name": "C_OrderLine",
  "method_name": "beforeSave",
  "method_signature": "protected boolean beforeSave(boolean newRecord)",
  "logic_types": ["validation", "calculation"],
  "columns_referenced": ["QtyOrdered", "QtyEntered", "M_Product_ID", "C_UOM_ID"],
  "line_start": 245,
  "line_end": 320,
  "javadoc": "Validates quantity and calculates line amounts",
  "content_summary": "Validates QtyOrdered > 0, calculates LineNetAmt = Qty * Price"
}

Logic Type Detection

public enum LogicType {
    VALIDATION,    // Throws exception, checks mandatory
    CALCULATION,   // Arithmetic operations, setValue
    TRIGGER,       // Updates other tables, creates records
    READ_ONLY,     // Checks processed/posted status
    CASCADE,       // Propagates to child records
    LOOKUP,        // Gets related records
    DEFAULT        // Sets default values
}

// Detection patterns
Map<LogicType, List<Pattern>> LOGIC_PATTERNS = Map.of(
    VALIDATION, List.of(
        Pattern.compile("throw\\s+new\\s+AdempiereException"),
        Pattern.compile("throw\\s+new\\s+AdempiereUserError"),
        Pattern.compile("FillMandatory"),
        Pattern.compile("if\\s*\\([^)]*==\\s*null"),
        Pattern.compile("if\\s*\\([^)]*\\.compareTo\\([^)]*\\)\\s*<=?\\s*0")
    ),
    CALCULATION, List.of(
        Pattern.compile("set\\w+\\([^)]*\\*[^)]*\\)"),  // multiplication
        Pattern.compile("set\\w+\\([^)]*\\+[^)]*\\)"),  // addition
        Pattern.compile("BigDecimal\\.ZERO"),
        Pattern.compile("\\.multiply\\("),
        Pattern.compile("\\.divide\\("),
        Pattern.compile("\\.subtract\\(")
    ),
    TRIGGER, List.of(
        Pattern.compile("new\\s+M[A-Z]\\w+\\("),       // Creates new model
        Pattern.compile("\\.saveEx\\(\\)"),
        Pattern.compile("\\.deleteEx\\("),
        Pattern.compile("DB\\.executeUpdate")
    ),
    READ_ONLY, List.of(
        Pattern.compile("isProcessed\\(\\)"),
        Pattern.compile("isPosted\\(\\)"),
        Pattern.compile("getDocStatus\\(\\)"),
        Pattern.compile("DOCSTATUS_")
    )
);

Content Generation

For each extracted method, generate searchable content:

String generateContent(MethodContext ctx) {
    StringBuilder content = new StringBuilder();

    // Header: Class and method info
    content.append("Class: ").append(ctx.className);
    content.append(" (").append(ctx.classType).append(")\n");
    content.append("Table: ").append(ctx.tableName).append("\n");
    content.append("Method: ").append(ctx.methodName).append("\n\n");

    // JavaDoc if present
    if (ctx.javadoc != null) {
        content.append("Documentation: ").append(ctx.javadoc).append("\n\n");
    }

    // Logic summary
    content.append("Logic Types: ").append(ctx.logicTypes).append("\n");
    content.append("Columns: ").append(ctx.columns).append("\n\n");

    // Human-readable summary
    content.append("Summary:\n");
    for (LogicType type : ctx.logicTypes) {
        switch (type) {
            case VALIDATION -> content.append("- Validates: ")
                .append(extractValidationSummary(ctx));
            case CALCULATION -> content.append("- Calculates: ")
                .append(extractCalculationSummary(ctx));
            case TRIGGER -> content.append("- Triggers: ")
                .append(extractTriggerSummary(ctx));
            case READ_ONLY -> content.append("- Read-only when: ")
                .append(extractReadOnlySummary(ctx));
        }
        content.append("\n");
    }

    // Code snippet (condensed)
    content.append("\nCode essence:\n");
    content.append(ctx.condensedCode);

    return content.toString();
}

Example Output

Input: MOrderLine.beforeSave()

@Override
protected boolean beforeSave(boolean newRecord) {
    // Validate quantity
    if (getQtyOrdered().signum() <= 0)
        throw new AdempiereException("@QtyOrdered@ <= 0");

    // Calculate line amount
    BigDecimal lineNetAmt = getQtyOrdered().multiply(getPriceActual());
    setLineNetAmt(lineNetAmt);

    // Update parent order
    if (is_ValueChanged(COLUMNNAME_LineNetAmt))
        getParent().setGrandTotal(calculateOrderTotal());

    return true;
}

Generated Content (for embedding):

Class: MOrderLine (model)
Table: C_OrderLine
Method: beforeSave

Logic Types: [VALIDATION, CALCULATION, TRIGGER]
Columns: [QtyOrdered, PriceActual, LineNetAmt]

Summary:
- Validates: QtyOrdered must be greater than 0
- Calculates: LineNetAmt = QtyOrdered * PriceActual
- Triggers: Updates parent order GrandTotal when LineNetAmt changes

Code essence:
if (getQtyOrdered() <= 0) throw Exception
setLineNetAmt(QtyOrdered * PriceActual)
if LineNetAmt changed -> update parent.GrandTotal

Search Query Match:

Query: "What validation rules apply to QtyOrdered?"

Match: Above content with high similarity score due to:

Implementation

@ApplicationScoped
public class JavaCodeIngestor implements KnowledgeIngestor {

    private static final Logger LOG = Logger.getLogger(JavaCodeIngestor.class);
    public static final String SOURCE_TYPE = "java_code";

    @Inject
    RagConfig ragConfig;

    @Inject
    EmbeddingStoreProvider storeProvider;

    @Inject
    EmbeddingModelProvider modelProvider;

    @ConfigProperty(name = "rag.java.source.path")
    Optional<String> sourcePath;

    @Override
    public String getSourceType() {
        return SOURCE_TYPE;
    }

    @Override
    public String getSourceName() {
        return "Java Source Code";
    }

    @Override
    public boolean isEnabled() {
        return sourcePath.isPresent() && Files.isDirectory(Path.of(sourcePath.get()));
    }

    @Override
    public ToolResult ingest() {
        return ingest(false);
    }

    @Override
    public ToolResult ingest(boolean force) {
        if (!isEnabled()) {
            return ToolResult.error("Java source path not configured or not found");
        }

        try {
            Path basePath = Path.of(sourcePath.get());
            EmbeddingStore<TextSegment> store = storeProvider.get();
            EmbeddingModel model = modelProvider.get();

            if (force) {
                clearExisting(store);
            }

            int count = 0;

            // Process model classes
            count += processFiles(basePath, "**/M*.java", "model", store, model);

            // Process process classes
            count += processFiles(basePath, "**/*Process.java", "process", store, model);

            // Process callouts
            count += processFiles(basePath, "**/*Callout*.java", "callout", store, model);

            // Process validators
            count += processFiles(basePath, "**/*Validator*.java", "validator", store, model);

            return ToolResult.success("Ingested " + count + " code chunks from Java source");

        } catch (Exception e) {
            LOG.error("Failed to ingest Java code", e);
            return ToolResult.error("Failed to ingest: " + e.getMessage());
        }
    }

    private int processFiles(Path basePath, String glob, String classType,
                             EmbeddingStore<TextSegment> store,
                             EmbeddingModel model) throws IOException {
        int count = 0;

        PathMatcher matcher = FileSystems.getDefault().getPathMatcher("glob:" + glob);

        try (Stream<Path> files = Files.walk(basePath)) {
            List<Path> javaFiles = files
                .filter(Files::isRegularFile)
                .filter(p -> matcher.matches(basePath.relativize(p)))
                .toList();

            for (Path file : javaFiles) {
                try {
                    List<CodeChunk> chunks = parseJavaFile(file, classType);
                    for (CodeChunk chunk : chunks) {
                        TextSegment segment = TextSegment.from(
                            chunk.content(),
                            chunk.metadata()
                        );
                        Embedding embedding = model.embed(segment).content();
                        store.add(embedding, segment);
                        count++;
                    }
                } catch (Exception e) {
                    LOG.warn("Failed to parse " + file + ": " + e.getMessage());
                }
            }
        }

        return count;
    }

    private List<CodeChunk> parseJavaFile(Path file, String classType) throws IOException {
        List<CodeChunk> chunks = new ArrayList<>();

        CompilationUnit cu = StaticJavaParser.parse(file);

        // Get class info
        Optional<ClassOrInterfaceDeclaration> classDecl = cu.findFirst(ClassOrInterfaceDeclaration.class);
        if (classDecl.isEmpty()) return chunks;

        String className = classDecl.get().getNameAsString();
        String packageName = cu.getPackageDeclaration()
            .map(pd -> pd.getNameAsString())
            .orElse("");

        // Infer table name from class name (MOrderLine -> C_OrderLine)
        String tableName = inferTableName(className, classType);

        // Extract key methods
        List<String> keyMethods = getKeyMethods(classType);

        for (MethodDeclaration method : cu.findAll(MethodDeclaration.class)) {
            String methodName = method.getNameAsString();

            if (keyMethods.contains(methodName)) {
                CodeChunk chunk = extractMethodChunk(
                    file, cu, classDecl.get(), method,
                    className, packageName, tableName, classType
                );
                chunks.add(chunk);
            }
        }

        return chunks;
    }

    private CodeChunk extractMethodChunk(
            Path file,
            CompilationUnit cu,
            ClassOrInterfaceDeclaration classDecl,
            MethodDeclaration method,
            String className,
            String packageName,
            String tableName,
            String classType) {

        String methodName = method.getNameAsString();
        String methodBody = method.getBody().map(Object::toString).orElse("");

        // Extract columns referenced
        Set<String> columns = extractColumnReferences(methodBody);

        // Detect logic types
        Set<LogicType> logicTypes = detectLogicTypes(methodBody);

        // Get JavaDoc
        String javadoc = method.getJavadoc()
            .map(jd -> jd.getDescription().toText())
            .orElse(null);

        // Generate content for embedding
        String content = generateContent(
            className, classType, tableName, methodName,
            columns, logicTypes, javadoc, methodBody
        );

        // Build metadata
        Metadata metadata = Metadata.from(Map.of(
            "source_type", SOURCE_TYPE,
            "file_path", file.toString(),
            "package", packageName,
            "class_name", className,
            "class_type", classType,
            "table_name", tableName != null ? tableName : "",
            "method_name", methodName,
            "logic_types", String.join(",", logicTypes.stream().map(Enum::name).toList()),
            "columns", String.join(",", columns),
            "line_start", String.valueOf(method.getBegin().map(p -> p.line).orElse(0)),
            "line_end", String.valueOf(method.getEnd().map(p -> p.line).orElse(0))
        ));

        return new CodeChunk(content, metadata);
    }

    // Helper methods...
    private Set<String> extractColumnReferences(String code) { ... }
    private Set<LogicType> detectLogicTypes(String code) { ... }
    private String generateContent(...) { ... }
    private String inferTableName(String className, String classType) { ... }
    private List<String> getKeyMethods(String classType) { ... }
}

record CodeChunk(String content, Metadata metadata) {}

enum LogicType {
    VALIDATION, CALCULATION, TRIGGER, READ_ONLY, CASCADE, LOOKUP, DEFAULT
}

CLI Integration

# Ingest Java source code
idempiere-cli rag ingest --source java_code --path /path/to/idempiere

# Search for column-specific logic
idempiere-cli rag search "QtyOrdered validation" --filter source_type=java_code

# Search for calculation logic
idempiere-cli rag search "line amount calculation" --filter logic_types=CALCULATION

Configuration

# application.properties

# Path to iDempiere source code
rag.java.source.path=/path/to/idempiere/org.adempiere.base/src

# File patterns to process
rag.java.patterns.model=**/M*.java
rag.java.patterns.process=**/*Process.java
rag.java.patterns.callout=**/*Callout*.java
rag.java.patterns.validator=**/*Validator*.java

# Methods to extract by class type
rag.java.methods.model=beforeSave,afterSave,beforeDelete,afterDelete,completeIt,voidIt
rag.java.methods.process=prepare,doIt
rag.java.methods.callout=*
rag.java.methods.validator=modelChange,docValidate

References

Path: /docs/developers/architecture/idempiere-hub/038-appendix-java-code-ingestor