Luigit
repositories / termux-janitor

termux-janitor

Interactive cleanup assistant for Termux: transparent, safe, confirmed disk reclamation.

owned by admin

src/classify.zig

Raw
//! Classification owned by `spec/CLASSIFICATION.md` and `spec/PRODUCT.md`
//! sections 5 and 6. Every finding receives exactly one category, a risk,
//! and a manual-only policy decision. Only producer-specific evidence may
//! recommend; heuristics limit automation and never expand it. The decision
//! reads only captured scan metadata and configuration; it never touches the
//! filesystem again.
const std = @import("std");
const spec_data = @import("spec_data");
const scan_mod = @import("scan.zig");

pub const Category = enum {
    lock_file,
    broken_symlink,
    rotated_log,
    live_log,
    stale_temporary,
    heuristic_temporary,
    download,
    document,
    source,
    workspace,
    trash,
    other,

    /// Compiled category label; identical across color and glyph modes.
    pub fn label(category: Category) []const u8 {
        return @tagName(category);
    }
};

pub const Risk = enum { low, medium, high };

pub const Context = struct {
    workspace_roots: []const []const u8,
    document_roots: []const []const u8,
    thresholds: *const @import("config.zig").Thresholds,
    realtime_sec: i64,
};

pub const Result = struct {
    category: Category = .other,
    risk: Risk = .medium,
    manual_only: bool = true,
};

fn ageSeconds(context: *const Context, finding: *const scan_mod.Finding) ?i64 {
    if (!finding.mtime_known) return null;
    const elapsed = context.realtime_sec - finding.mtime_sec;
    if (elapsed < 0) return null; // future timestamps never satisfy an age rule
    return elapsed;
}

fn hasSuffix(name: []const u8, suffix: []const u8) bool {
    return std.mem.endsWith(u8, name, suffix);
}

fn hasSuffixCI(name: []const u8, suffix: []const u8) bool {
    if (name.len < suffix.len) return false;
    return std.ascii.eqlIgnoreCase(name[name.len - suffix.len ..], suffix);
}

fn allDigits(text: []const u8) bool {
    if (text.len == 0) return false;
    for (text) |byte| {
        if (byte < '0' or byte > '9') return false;
    }
    return true;
}

/// Rotated logs: a `.log` identity plus `.N`, `.old`, `.gz`, `.bz2`, `.xz`,
/// or `.zst` rotation (`spec/CLASSIFICATION.md`).
fn rotatedLogName(name: []const u8) bool {
    var rest = name;
    while (true) {
        const dot = std.mem.lastIndexOfScalar(u8, rest, '.') orelse return false;
        const tail = rest[dot + 1 ..];
        if (std.mem.eql(u8, tail, ".log")) return false;
        if (hasSuffix(rest[0..dot], ".log")) return true;
        if (!allDigits(tail)) return false;
        rest = rest[0..dot];
    }
}

fn temporaryName(name: []const u8) bool {
    const markers = [_][]const u8{ "~", ".tmp", ".temp", ".swp", ".swo", ".part" };
    for (markers) |marker| {
        if (std.mem.startsWith(u8, name, marker)) return true;
        if (hasSuffix(name, marker)) return true;
    }
    return false;
}

const installer_suffixes = [_][]const u8{ ".zip", ".gz", ".bz2", ".xz", ".zst", ".tar", ".deb", ".rpm", ".apk", ".jar" };

/// Download recognition requires suffix and file magic to agree; a mismatch
/// is uncertain purpose and stays `.other` (manual-only).
fn downloadKind(name: []const u8, first_bytes: []const u8) bool {
    for (installer_suffixes) |suffix| {
        if (!hasSuffixCI(name, suffix)) continue;
        if (first_bytes.len < 4) return false;
        if (std.mem.eql(u8, suffix, ".gz")) return first_bytes[0] == 0x1F and first_bytes[1] == 0x8B;
        if (std.mem.eql(u8, suffix, ".bz2")) return std.mem.startsWith(u8, first_bytes, "BZh");
        if (std.mem.eql(u8, suffix, ".xz")) return std.mem.startsWith(u8, first_bytes, "\xfd7zXZ\x00");
        if (std.mem.eql(u8, suffix, ".zst")) {
            return first_bytes[0] == 0x28 and first_bytes[1] == 0xB5 and first_bytes[2] == 0x2F and first_bytes[3] == 0xFD;
        }
        if (std.mem.eql(u8, suffix, ".deb")) return std.mem.startsWith(u8, first_bytes, "!<arch>");
        if (std.mem.eql(u8, suffix, ".rpm")) {
            return first_bytes[0] == 0xED and first_bytes[1] == 0xAB and first_bytes[2] == 0xEE and first_bytes[3] == 0xDB;
        }
        if (std.mem.eql(u8, suffix, ".tar")) return false; // header lies past magic window
        return std.mem.startsWith(u8, first_bytes, "PK\x03\x04");
    }
    return false;
}

fn isSource(name: []const u8) bool {
    for (spec_data.classification_source_basenames) |basename| {
        if (std.mem.eql(u8, name, basename)) return true;
    }
    for (spec_data.classification_source_suffixes) |suffix| {
        if (hasSuffixCI(name, suffix)) return true;
    }
    return false;
}

/// Classifies one retained finding given its raw path (reconstructed by the
/// caller) and up to four leading file bytes for magic checks.
pub fn classify(
    finding: *const scan_mod.Finding,
    path: []const u8,
    first_bytes: []const u8,
    context: *const Context,
) Result {
    var result: Result = .{};
    const name = finding.name;
    if (pathIsUnder(path, context.workspace_roots)) {
        // Workspace content is source-bearing and manual-only pending VCS
        // metadata (`TJ-CLASS-13`); version 1 marks the class conservatively.
        result.category = .workspace;
        result.risk = .high;
        return result;
    }
    if (finding.kind == .file and
        (hasSuffix(name, ".lock") or hasSuffix(name, ".lck") or hasSuffix(name, ".pid")))
    {
        result.category = .lock_file;
        result.risk = .high;
        return result;
    }
    if (finding.kind == .symlink and finding.target_status == .broken) {
        result.category = .broken_symlink;
        result.risk = .medium;
        return result;
    }
    if (finding.kind == .file) {
        if (rotatedLogName(name)) {
            result.category = .rotated_log;
            const elapsed = ageSeconds(context, finding) orelse 0;
            result.risk = if (elapsed >= @as(i64, @intCast(context.thresholds.old_rotated_log_seconds)))
                .medium
            else
                .high;
            return result;
        }
        if (hasSuffix(name, ".log")) {
            result.category = .live_log;
            result.risk = .high;
            return result;
        }
        if (temporaryName(name)) {
            const elapsed = ageSeconds(context, finding) orelse 0;
            result.category = if (elapsed >= @as(i64, @intCast(context.thresholds.stale_temporary_seconds)))
                .stale_temporary
            else
                .heuristic_temporary;
            result.risk = .high;
            return result;
        }
        if (pathIsUnder(path, context.document_roots)) {
            if (downloadKind(name, first_bytes)) {
                result.category = .download;
                result.risk = .high;
                return result;
            }
            result.category = .document;
            result.risk = .high;
            return result;
        }
        if (isSource(name)) {
            result.category = .source;
            result.risk = .high;
            return result;
        }
    }
    result.category = .other;
    result.risk = .medium;
    return result;
}

fn pathIsUnder(path: []const u8, roots: []const []const u8) bool {
    for (roots) |root| {
        if (@import("config.zig").prefixMatches(root, path)) return true;
    }
    return false;
}

comptime {
    std.debug.assert(installer_suffixes.len > 0);
    std.debug.assert(downloadKind("x.zip", "PK\x03\x04zz"));
    std.debug.assert(!downloadKind("x.zip", "notzip"));
    std.debug.assert(rotatedLogName("access.log.1"));
    std.debug.assert(rotatedLogName("access.log.gz"));
    std.debug.assert(!rotatedLogName("access.log"));
}