import { existsSync } from "node:fs";
import type { DatabaseAdapter } from "./db.js";
import { ADAPTER_VERSIONS } from "./adapter-versions.js";
import { ALL_SOURCES, SESSION_SOURCES, resolveSourceRoots, sourceRootResolution, type AnalyticsSource, type ResolverEnvironment } from "./sources.js";

/**
 * Why a source shows what it shows. "0" alone cannot tell real inactivity from a pipeline that
 * never saw the data, so every source gets an explicit state:
 * - root_missing: no source root exists on this machine (client not installed or custom root);
 * - no_candidates: a root exists but no candidate file was ever recorded (not scanned yet or empty);
 * - parser_unsupported: every failure is an unsupported schema and nothing was imported;
 * - failed: files failed and nothing was imported;
 * - partial: some files failed, or imported rows come from an older adapter projection;
 * - zero_activity: files were imported but they contain no sessions/events;
 * - available: imported data present and no failures.
 * No raw path, session id or content is ever reported.
 */
export type SourceDiagnosticState = "available" | "zero_activity" | "root_missing" | "no_candidates" | "partial" | "parser_unsupported" | "failed";

export interface SourceDiagnostic {
  state: SourceDiagnosticState;
  root_status: "found" | "missing";
  root_resolution: string;
  adapter_version: string;
  files_imported: number;
  files_failed: number;
  files_pending_reprojection: number;
  files_no_longer_candidates: number;
  failures_by_code: Record<string, number>;
  sessions: number;
  main_sessions_with_messages: number;
  runtime_events: number;
  token_coverage?: { available: number; missing: number };
  parser_counters?: Record<string, number>;
}

function stateFor(d: Omit<SourceDiagnostic, "state">): SourceDiagnosticState {
  const files = d.files_imported + d.files_failed;
  if (files === 0) return d.root_status === "missing" ? "root_missing" : "no_candidates";
  if (d.files_imported === 0) {
    const codes = Object.keys(d.failures_by_code);
    return codes.length > 0 && codes.every((code) => code === "SOURCE_SCHEMA_UNSUPPORTED") ? "parser_unsupported" : "failed";
  }
  if (d.files_failed > 0 || d.files_pending_reprojection > 0) return "partial";
  return d.sessions === 0 && d.runtime_events === 0 ? "zero_activity" : "available";
}

// Start of the latest full-discovery scan that included the source. A newer explicit
// incomplete-discovery result blocks older cutoffs for that source; otherwise a known file under
// an unreadable subtree could be hidden by a stale successful scan. Legacy metadata keeps its
// historical `full_discovery` interpretation.
async function latestFullDiscoveryStarts(db: DatabaseAdapter): Promise<Map<string, string>> {
  const starts = new Map<string, string>();
  const resolved = new Set<string>();
  const runs = await db.all<{ id: string; started_at: string; sources_json: string | null }>(
    "SELECT id, started_at, sources_json FROM scan_runs WHERE dry_run = 0 AND status IN ('completed', 'partial') ORDER BY started_at DESC LIMIT 50"
  );
  for (const run of runs) {
    if (ALL_SOURCES.every((source) => resolved.has(source))) break;
    let sources: unknown;
    try {
      sources = JSON.parse(run.sources_json ?? "[]");
    } catch {
      continue;
    }
    if (!Array.isArray(sources) || sources.every((source) => resolved.has(String(source)))) continue;
    const meta = await db.get<{
      full_discovery: number | null;
      full_discovery_sources: string | null;
      full_discovery_sources_type: string | null;
      discovery_complete_sources: string | null;
      discovery_complete_sources_type: string | null;
      discovery_incomplete_sources: string | null;
      discovery_incomplete_sources_type: string | null;
    }>(
      `SELECT
         CASE WHEN json_valid(metadata_json) THEN json_extract(metadata_json, '$.full_discovery') END AS full_discovery,
         CASE WHEN json_valid(metadata_json) THEN json_extract(metadata_json, '$.full_discovery_sources') END AS full_discovery_sources,
         CASE WHEN json_valid(metadata_json) THEN json_type(metadata_json, '$.full_discovery_sources') END AS full_discovery_sources_type,
         CASE WHEN json_valid(metadata_json) THEN json_extract(metadata_json, '$.discovery_complete_sources') END AS discovery_complete_sources,
         CASE WHEN json_valid(metadata_json) THEN json_type(metadata_json, '$.discovery_complete_sources') END AS discovery_complete_sources_type,
         CASE WHEN json_valid(metadata_json) THEN json_extract(metadata_json, '$.discovery_incomplete_sources') END AS discovery_incomplete_sources,
         CASE WHEN json_valid(metadata_json) THEN json_type(metadata_json, '$.discovery_incomplete_sources') END AS discovery_incomplete_sources_type
       FROM scan_runs WHERE id = ?`,
      [run.id]
    );
    const parseArray = (value: string | null, type: string | null): string[] | null => {
      if (type !== "array" || value === null) return null;
      try {
        const parsed: unknown = JSON.parse(value);
        return Array.isArray(parsed) ? parsed.filter((entry): entry is string => typeof entry === "string") : null;
      } catch {
        return null;
      }
    };
    const fullSources = parseArray(meta?.full_discovery_sources ?? null, meta?.full_discovery_sources_type ?? null);
    const completeSources = parseArray(meta?.discovery_complete_sources ?? null, meta?.discovery_complete_sources_type ?? null);
    const incompleteSources = parseArray(meta?.discovery_incomplete_sources ?? null, meta?.discovery_incomplete_sources_type ?? null);
    const hasExplicitSourceMetadata = [
      meta?.full_discovery_sources_type,
      meta?.discovery_complete_sources_type,
      meta?.discovery_incomplete_sources_type
    ].some((type) => type !== null && type !== undefined);
    for (const sourceValue of sources) {
      const source = String(sourceValue);
      if (resolved.has(source)) continue;
      if (incompleteSources?.includes(source)) {
        // Do not fall back to an older full scan after an explicit incomplete discovery.
        resolved.add(source);
        continue;
      }
      if (fullSources?.includes(source)) {
        starts.set(source, run.started_at);
        resolved.add(source);
        continue;
      }
      if (completeSources?.includes(source)) {
        // A `since` scan enumerated the source but is not a cutoff. Retain the most recent
        // earlier full scan as the last trustworthy absence check.
        continue;
      }
      if (hasExplicitSourceMetadata) {
        // Explicit metadata that omits a selected source is conservative evidence against a cutoff.
        resolved.add(source);
        continue;
      }
      if (meta?.full_discovery === 1) {
        starts.set(source, run.started_at);
        resolved.add(source);
      }
    }
  }
  return starts;
}

export async function buildSourceDiagnostics(db: DatabaseAdapter, hookLogPath: string, options: ResolverEnvironment = {}): Promise<Record<string, SourceDiagnostic>> {
  const out: Record<string, SourceDiagnostic> = {};
  // Session, event and token aggregates are grouped once for every source (status stays cheap on large DBs).
  const sessionRows = await db.all<{ source: string; sessions: number; main_with_messages: number }>(
    "SELECT source, COUNT(*) AS sessions, COALESCE(SUM(CASE WHEN COALESCE(session_kind,'main')='main' AND user_message_count > 0 THEN 1 ELSE 0 END),0) AS main_with_messages FROM sessions GROUP BY source"
  );
  const eventRows = await db.all<{ source: string; n: number }>("SELECT source, COUNT(*) AS n FROM runtime_events GROUP BY source");
  const tokenRows = await db.all<{ source: string; available: number; missing: number }>(
    "SELECT s.source AS source, COALESCE(SUM(CASE WHEN mm.token_available=1 THEN 1 ELSE 0 END),0) AS available, COALESCE(SUM(CASE WHEN mm.token_available=0 THEN 1 ELSE 0 END),0) AS missing FROM message_metrics mm JOIN sessions s ON s.id = mm.session_id GROUP BY s.source"
  );
  const fullDiscoveryStarts = await latestFullDiscoveryStarts(db);
  const bySource = <T extends { source: string }>(rows: T[]) => new Map(rows.map((row) => [row.source, row]));
  const sessionsBySource = bySource(sessionRows);
  const eventsBySource = bySource(eventRows);
  const tokensBySource = bySource(tokenRows);
  for (const source of ALL_SOURCES as readonly AnalyticsSource[]) {
    const roots = resolveSourceRoots(source, hookLogPath, options);
    const seenSince = fullDiscoveryStarts.get(source) ?? null;
    const seenClause = seenSince ? " AND last_seen_at >= ?" : "";
    const seenParams = seenSince ? [seenSince] : [];
    const fileRows = await db.all<{ last_status: string; last_error_code: string | null; adapter_version: string | null; n: number }>(
      `SELECT last_status, last_error_code, adapter_version, COUNT(*) AS n FROM source_files WHERE source = ?${seenClause} GROUP BY last_status, last_error_code, adapter_version`,
      [source, ...seenParams]
    );
    const notSeen = seenSince
      ? (await db.get<{ n: number }>("SELECT COUNT(*) AS n FROM source_files WHERE source = ? AND (last_seen_at IS NULL OR last_seen_at < ?)", [source, seenSince]))?.n ?? 0
      : 0;
    const failuresByCode: Record<string, number> = {};
    let imported = 0;
    let failed = 0;
    let pendingReprojection = 0;
    for (const row of fileRows) {
      if (row.last_status === "imported") {
        imported += row.n;
        if (row.adapter_version !== ADAPTER_VERSIONS[source]) pendingReprojection += row.n;
      } else if (row.last_status === "failed") {
        failed += row.n;
        const code = row.last_error_code ?? "UNKNOWN_ERROR";
        failuresByCode[code] = (failuresByCode[code] ?? 0) + row.n;
      }
    }
    const sessionRow = sessionsBySource.get(source);
    const eventRow = eventsBySource.get(source);
    const sessionSource = (SESSION_SOURCES as readonly AnalyticsSource[]).includes(source);
    const tokenRow = sessionSource ? tokensBySource.get(source) ?? { available: 0, missing: 0 } : null;
    // Per-file parser counters (e.g. Claude dedupe) are summed; only numeric fields are kept.
    const counterRows = await db.all<{ metadata_json: string | null }>(
      `SELECT metadata_json FROM source_files WHERE source = ? AND last_status = 'imported' AND metadata_json IS NOT NULL${seenClause}`,
      [source, ...seenParams]
    );
    const parserCounters: Record<string, number> = {};
    for (const row of counterRows) {
      try {
        const diagnostics = (JSON.parse(row.metadata_json ?? "{}") as { diagnostics?: Record<string, unknown> }).diagnostics ?? {};
        for (const [key, value] of Object.entries(diagnostics)) {
          if (typeof value === "number" && Number.isFinite(value)) parserCounters[key] = (parserCounters[key] ?? 0) + value;
        }
      } catch {
        // Malformed legacy metadata is ignored: diagnostics never fail status.
      }
    }
    const base: Omit<SourceDiagnostic, "state"> = {
      root_status: roots.some((root) => existsSync(root)) ? "found" : "missing",
      root_resolution: sourceRootResolution(source, options),
      adapter_version: ADAPTER_VERSIONS[source],
      files_imported: imported,
      files_failed: failed,
      files_pending_reprojection: pendingReprojection,
      files_no_longer_candidates: notSeen,
      failures_by_code: failuresByCode,
      sessions: sessionRow?.sessions ?? 0,
      main_sessions_with_messages: sessionRow?.main_with_messages ?? 0,
      runtime_events: eventRow?.n ?? 0,
      ...(tokenRow ? { token_coverage: { available: tokenRow.available, missing: tokenRow.missing } } : {}),
      ...(Object.keys(parserCounters).length > 0 ? { parser_counters: parserCounters } : {})
    };
    out[source] = { state: stateFor(base), ...base };
  }
  return out;
}
