Luigit
repositories / termux-janitor

termux-janitor

Interactive cleanup assistant for Termux: transparent, safe, confirmed disk reclamation.

owned by admin

tools/spec-engine/src/text.zig

Raw
const std = @import("std");

pub fn markdownSection(source: []const u8, heading: []const u8) ?[]const u8 {
    var marker_buffer: [96]u8 = undefined;
    const marker = std.fmt.bufPrint(&marker_buffer, "## {s}\n", .{heading}) catch return null;
    const marker_start = std.mem.indexOf(u8, source, marker) orelse return null;
    const content_start = marker_start + marker.len;
    if (std.mem.indexOfPos(u8, source, content_start, marker) != null) return null;
    const remaining = source[content_start..];
    const relative_end = std.mem.indexOf(u8, remaining, "\n## ") orelse remaining.len;
    const section = std.mem.trim(u8, remaining[0..relative_end], " \t\r\n");
    if (section.len == 0) return null;
    return section;
}

pub fn occurrenceCount(source: []const u8, needle: []const u8) u32 {
    std.debug.assert(needle.len > 0);
    var count: u32 = 0;
    var offset: usize = 0;
    while (std.mem.indexOfPos(u8, source, offset, needle)) |index| {
        count += 1;
        offset = index + needle.len;
    }
    return count;
}

pub fn renderInlineList(values: []const []const u8, writer: *std.Io.Writer) !void {
    if (values.len == 0) {
        try writer.writeAll("none");
        return;
    }
    for (values, 0..) |value, index| {
        if (index > 0) try writer.writeAll(", ");
        try writer.writeAll(value);
    }
}

pub fn renderInlineCodeList(values: []const []const u8, writer: *std.Io.Writer) !void {
    if (values.len == 0) {
        try writer.writeAll("none");
        return;
    }
    for (values, 0..) |value, index| {
        if (index > 0) try writer.writeAll(", ");
        try writer.print("`{s}`", .{value});
    }
}

pub fn writeWrapped(
    writer: *std.Io.Writer,
    source: []const u8,
    first_prefix: []const u8,
    next_prefix: []const u8,
    width_max: usize,
) !void {
    var words = std.mem.tokenizeAny(u8, source, " \t\r\n");
    var width = first_prefix.len;
    var first = true;
    try writer.writeAll(first_prefix);
    while (words.next()) |word| {
        const separator: usize = if (first) 0 else 1;
        if (width + separator + word.len > width_max) {
            try writer.writeByte('\n');
            try writer.writeAll(next_prefix);
            width = next_prefix.len;
        } else if (!first) {
            try writer.writeByte(' ');
            width += 1;
        }
        try writer.writeAll(word);
        width += word.len;
        first = false;
    }
}

pub fn nameValid(name: []const u8) bool {
    if (name.len == 0 or name.len > 64) return false;
    if (name[0] == '-' or name[name.len - 1] == '-') return false;
    var previous_hyphen = false;
    for (name) |byte| {
        const valid = std.ascii.isLower(byte) or std.ascii.isDigit(byte) or byte == '-';
        if (!valid) return false;
        if (byte == '-' and previous_hyphen) return false;
        previous_hyphen = byte == '-';
    }
    return true;
}

pub fn symbolValid(symbol: []const u8) bool {
    if (symbol.len == 0 or symbol.len > 64) return false;
    if (!std.ascii.isLower(symbol[0])) return false;
    if (symbol[symbol.len - 1] == '_') return false;
    var previous_underscore = false;
    for (symbol) |byte| {
        const valid = std.ascii.isLower(byte) or std.ascii.isDigit(byte) or byte == '_';
        if (!valid) return false;
        if (byte == '_' and previous_underscore) return false;
        previous_underscore = byte == '_';
    }
    return true;
}

pub fn conceptIdValid(id: []const u8) bool {
    return prefixedNameValid(id, "concept-");
}

pub fn prefixedNameValid(id: []const u8, prefix: []const u8) bool {
    if (!std.mem.startsWith(u8, id, prefix)) return false;
    return nameValid(id);
}

pub fn oracleIdValid(id: []const u8) bool {
    if (id.len == 0 or id.len > 64) return false;
    const invariant = std.mem.startsWith(u8, id, "INV-");
    const forbidden = std.mem.startsWith(u8, id, "NO-");
    if (!invariant) {
        if (!forbidden) return false;
    }
    if (id[id.len - 1] == '-') return false;
    var previous_hyphen = false;
    for (id) |byte| {
        const valid = std.ascii.isUpper(byte) or std.ascii.isDigit(byte) or byte == '-';
        if (!valid) return false;
        if (byte == '-') {
            if (previous_hyphen) return false;
        }
        previous_hyphen = byte == '-';
    }
    return true;
}

pub fn markdownHasAnchor(source: []const u8, expected: []const u8) bool {
    var lines = std.mem.splitScalar(u8, source, '\n');
    while (lines.next()) |line| {
        var level_end: usize = 0;
        while (level_end < line.len and line[level_end] == '#') level_end += 1;
        if (level_end == 0 or level_end >= line.len or line[level_end] != ' ') continue;
        if (headingMatchesAnchor(line[level_end + 1 ..], expected)) return true;
    }
    return false;
}

pub fn headingMatchesAnchor(heading: []const u8, expected: []const u8) bool {
    var buffer: [256]u8 = undefined;
    var count: usize = 0;
    var pending_hyphen = false;
    for (heading) |byte| {
        if (std.ascii.isAlphanumeric(byte) or byte == '_') {
            if (pending_hyphen and count > 0 and buffer[count - 1] != '-') {
                if (count == buffer.len) return false;
                buffer[count] = '-';
                count += 1;
            }
            pending_hyphen = false;
            if (count == buffer.len) return false;
            buffer[count] = std.ascii.toLower(byte);
            count += 1;
        } else if (byte == '-' or std.ascii.isWhitespace(byte)) {
            pending_hyphen = true;
        }
    }
    return std.mem.eql(u8, buffer[0..count], expected);
}

pub fn writeCollapsedWhitespace(writer: *std.Io.Writer, source: []const u8) !void {
    var whitespace_pending = false;
    var wrote_byte = false;
    for (source) |byte| {
        if (std.ascii.isWhitespace(byte)) {
            whitespace_pending = true;
            continue;
        }
        if (whitespace_pending and wrote_byte) try writer.writeByte(' ');
        try writer.writeByte(byte);
        whitespace_pending = false;
        wrote_byte = true;
    }
}

pub fn writeXmlWrappedDescription(writer: *std.Io.Writer, source: []const u8) !void {
    const indent = "      ";
    const width_max = 100;
    var words = std.mem.tokenizeAny(u8, source, " \t\r\n");
    var width: usize = indent.len;
    var word_count: u32 = 0;
    try writer.writeAll("    <description>\n");
    try writer.writeAll(indent);
    while (words.next()) |word| {
        const escaped_width = xmlEscapedWidth(word);
        const separator_width: usize = if (word_count == 0) 0 else 1;
        if (width + separator_width + escaped_width > width_max) {
            try writer.writeAll("\n");
            try writer.writeAll(indent);
            width = indent.len;
        } else if (separator_width == 1) {
            try writer.writeByte(' ');
            width += 1;
        }
        try writeXmlEscaped(writer, word, false);
        width += escaped_width;
        word_count += 1;
    }
    try writer.writeAll("\n    </description>\n");
}

pub fn xmlEscapedWidth(source: []const u8) usize {
    var width: usize = 0;
    for (source) |byte| {
        width += switch (byte) {
            '&' => 5,
            '<', '>' => 4,
            else => 1,
        };
    }
    return width;
}

pub fn writeXmlEscaped(writer: *std.Io.Writer, source: []const u8, collapse: bool) !void {
    var whitespace_pending = false;
    for (source) |byte| {
        if (collapse and std.ascii.isWhitespace(byte)) {
            whitespace_pending = true;
            continue;
        }
        if (whitespace_pending) {
            try writer.writeByte(' ');
            whitespace_pending = false;
        }
        switch (byte) {
            '&' => try writer.writeAll("&amp;"),
            '<' => try writer.writeAll("&lt;"),
            '>' => try writer.writeAll("&gt;"),
            else => try writer.writeByte(byte),
        }
    }
}

test "names and identifiers reject malformed input" {
    try std.testing.expect(nameValid("filesystem-capabilities"));
    try std.testing.expect(!nameValid("Filesystem"));
    try std.testing.expect(!nameValid("two--hyphens"));
    try std.testing.expect(conceptIdValid("concept-action"));
}

test "heading anchors are deterministic" {
    try std.testing.expect(markdownHasAnchor(
        "## Filesystem revalidation\n",
        "filesystem-revalidation",
    ));
    try std.testing.expect(!markdownHasAnchor(
        "## Filesystem validation\n",
        "filesystem-revalidation",
    ));
}

test "catalog text escapes XML and collapses whitespace" {
    var output = std.Io.Writer.Allocating.init(std.testing.allocator);
    defer output.deinit();
    try writeXmlEscaped(&output.writer, "one\n two & <three>", true);
    try std.testing.expectEqualStrings("one two &amp; &lt;three&gt;", output.written());
}