Files
finance-app/scripts/import-frollo.mts
T
siddharthd b82c4570bd
ci / lint-test (push) Successful in 45s
frollo: don't import what the statements already cover
"We should not be importing from Frollo what we already have from statements"
(owner). The amount+direction guard could not deliver that, because the two
sources decompose the same event differently: Frollo bundles the Wise fee into
the transfer (10001.13) where the statement itemises it (10000.00 + 1.13). So 38
Wise USD rows passed the amount guard as new while being the same money, and were
the reason the transactions view showed a USD figure where every neighbouring row
showed AUD.

A statement's billing_end_date is a hard watermark: everything on that account up
to that date is already in the ledger, itemised and converted. Guard 1 now drops
any feed row at or before its account's newest statement. Guard 2 (amount +
direction) stays as the net for accounts that have no statement at all.

Note this is the coverage test the first import needed and got wrong. That one
asked whether a row's date fell inside a statement's min-max window, which for
periods spanning 182 to 460 days swallows a year and answers nothing. The
watermark asks a question that has an answer: up to what date is this account
complete?

Matching is on last4, verified against the live statement set — the eight
in-scope accounts with statements each map to one bank, no cross-bank collision.
The watermarks are printed by the CLI so a wrong boundary is visible rather than
inferred from a row count.

Re-imported from empty: 425 covered by statement, 15 amount twins, 110 inserted.
Exactly one row now carries a foreign currency with no AUD figure — the
2026-08-12 HDR salary, which is the genuinely-new pre-statement row this feed
exists for. Was 39.
2026-08-13 12:36:08 +10:00

128 lines
5.4 KiB
TypeScript

/**
* Imports a Frollo transaction export into `transactions`.
*
* node --experimental-strip-types scripts/import-frollo.mts --file <csv>
* node --experimental-strip-types scripts/import-frollo.mts --file <csv> --apply
*
* Dry run by default: it prints exactly what it would do and touches nothing.
* Rehearse first — the failure mode here is doubling reported income, and the
* de-duplication rule that shipped second only survived because a dry run was
* read against the salary rows.
*
* Export the file with PENDING TRANSACTIONS EXCLUDED. A pending row changes both
* its id and its description when it settles, so importing one guarantees a
* duplicate on the next run; every pending row observed has been on a credit
* card, which this importer does not cover anyway. See DECISIONS.md ING-11.
*
* All parsing, scoping, de-duplication and insertion live in
* `src/lib/frollo-ingest.ts`, shared with POST /api/frollo/ingest so the manual
* and automatic paths cannot drift.
*
* Needs DATABASE_URL. postgres-personal publishes no host port, so from the host:
* export DATABASE_URL="postgresql://personal:<pw>@$(docker inspect postgres-personal \
* --format '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}'):5432/personal"
* The container IP changes on every recreate.
*/
import { readFileSync } from "node:fs";
import pg from "pg";
import { ingestFrolloCsv, LEDGER_MATCH_DAYS, type SqlExecutor } from "../src/lib/frollo-ingest.ts";
function arg(name: string, fallback?: string): string | undefined {
const i = process.argv.indexOf(`--${name}`);
if (i >= 0 && process.argv[i + 1] && !process.argv[i + 1].startsWith("--")) return process.argv[i + 1];
return fallback;
}
const has = (name: string) => process.argv.includes(`--${name}`);
const file = arg("file");
if (!file) {
console.error("usage: --file <csv> [--apply] [--force] [--owner <id>]");
process.exit(2);
}
if (!process.env.DATABASE_URL) {
console.error("DATABASE_URL is not set — cannot compare against the ledger.");
process.exit(2);
}
const money = (n: number) => n.toLocaleString("en-AU", { minimumFractionDigits: 2, maximumFractionDigits: 2 });
const client = new pg.Client({ connectionString: process.env.DATABASE_URL });
await client.connect();
const exec: SqlExecutor = async <T,>(sql: string, params: unknown[]) =>
(await client.query(sql, params)).rows as T[];
let report;
try {
report = await ingestFrolloCsv(readFileSync(file, "utf8"), exec, {
ownerId: Number(arg("owner", "1")),
apply: has("apply"),
force: has("force"),
});
} finally {
// Left open, the process hangs after an exception and looks like a slow query.
if (!has("keep-open")) await client.end().catch(() => {});
}
console.log(`\n${file}`);
console.log(` ${report.totalRows} rows in file\n`);
console.log("SCOPE");
for (const [why, n] of Object.entries(report.skipped).sort((a, b) => b[1]! - a[1]!)) {
console.log(` skipped ${String(n).padStart(5)} ${why}`);
}
console.log(` in scope ${String(report.inScope).padStart(4)}\n`);
console.log("ACCOUNTS");
for (const a of report.accounts) {
console.log(
` ${a.present ? " " : "!"} ${a.spec.last4} ${a.spec.label.padEnd(24)} ${String(a.rows).padStart(5)} rows`
);
}
if (report.unknownAccounts.length > 0) {
console.log("\n accounts in the file this importer does not know (skipped):");
for (const u of report.unknownAccounts) {
console.log(` ${u.accountNumber} ${u.accountName} ${u.rows} rows`);
}
}
console.log();
console.log("DE-DUPLICATION (re-ingest twins from CDR re-consent)");
console.log(` ${report.duplicatesDropped} dropped`);
if (report.suspectRepeats.length > 0) {
console.log(`\n REVIEW: ${report.suspectRepeats.length} collapsed row(s) had near-consecutive ids`);
console.log(" and may be real repeats rather than duplicates:");
for (const s of report.suspectRepeats.slice(0, 10)) {
console.log(` ${s.transactionDate} ${money(s.amount).padStart(11)} ${s.description.slice(0, 40)} (gap ${s.idGap})`);
}
}
console.log();
console.log("LEDGER");
console.log(` already imported : ${String(report.alreadyImported).padStart(5)}`);
// Printed even when zero. These are the counts that were missing on
// 2026-08-13, when the whole file was written on top of a ledger that already
// held 77% of it; a number you have to go looking for is a number nobody looks
// at. The watermarks are printed too, so a wrong coverage boundary is visible
// rather than inferred from a row count.
console.log(` covered by stmt : ${String(report.coveredByStatement).padStart(5)} (dated on or before the account's newest statement)`);
console.log(` amount twin : ${String(report.ledgerDuplicates).padStart(5)} (same amount + direction within ${LEDGER_MATCH_DAYS} days)`);
if (report.statementWatermarks.length > 0) {
console.log(" statements cover:");
for (const w of report.statementWatermarks) console.log(` ${w.last4} up to ${w.coveredTo}`);
}
console.log(` new to insert : ${String(report.toInsert).padStart(5)}`);
if (report.anomalies.length > 0) {
console.log("\nANOMALIES");
for (const a of report.anomalies) console.log(` ! ${a}`);
}
if (report.applied) {
console.log(`\nInserted ${report.inserted} row(s).\n`);
} else if (report.anomalies.length > 0 && has("apply")) {
console.log("\nNOT APPLIED — anomalies above. Read them, then re-run with --force.\n");
process.exit(1);
} else {
console.log("\nDRY RUN — nothing written. Re-run with --apply to insert.\n");
}