Remote sync of webdav

This commit is contained in:
Courtney Arnold 2026-08-13 17:05:10 -05:00
parent f01186df79
commit 483fbeafb6
44 changed files with 4842 additions and 975 deletions

View file

@ -1,3 +1,25 @@
/// Two-letter USPS abbreviation to full name, for the 50 states + DC —
/// deliberately excludes territories (PR, VI, GU, ...): "VI" in particular
/// shows up on real receipts as a "Visa" abbreviation, and including it as
/// a valid state code would misfire on that.
const usStateNames = <String, String>{
'AL': 'Alabama', 'AK': 'Alaska', 'AZ': 'Arizona', 'AR': 'Arkansas',
'CA': 'California', 'CO': 'Colorado', 'CT': 'Connecticut', 'DE': 'Delaware',
'FL': 'Florida', 'GA': 'Georgia', 'HI': 'Hawaii', 'ID': 'Idaho',
'IL': 'Illinois', 'IN': 'Indiana', 'IA': 'Iowa', 'KS': 'Kansas',
'KY': 'Kentucky', 'LA': 'Louisiana', 'ME': 'Maine', 'MD': 'Maryland',
'MA': 'Massachusetts', 'MI': 'Michigan', 'MN': 'Minnesota',
'MS': 'Mississippi', 'MO': 'Missouri', 'MT': 'Montana', 'NE': 'Nebraska',
'NV': 'Nevada', 'NH': 'New Hampshire', 'NJ': 'New Jersey',
'NM': 'New Mexico', 'NY': 'New York', 'NC': 'North Carolina',
'ND': 'North Dakota', 'OH': 'Ohio', 'OK': 'Oklahoma', 'OR': 'Oregon',
'PA': 'Pennsylvania', 'RI': 'Rhode Island', 'SC': 'South Carolina',
'SD': 'South Dakota', 'TN': 'Tennessee', 'TX': 'Texas', 'UT': 'Utah',
'VT': 'Vermont', 'VA': 'Virginia', 'WA': 'Washington',
'WV': 'West Virginia', 'WI': 'Wisconsin', 'WY': 'Wyoming',
'DC': 'District of Columbia',
};
/// Best-effort values pulled out of OCR'd receipt text. Any field can be
/// null if it couldn't be found, and the confirm screen lets the user fill
/// in or correct whatever the parser got wrong.
@ -5,12 +27,25 @@ class ParsedReceipt {
final double? gallons;
final double? pricePerGallon;
final double? totalCost;
/// The transaction date/time printed on the receipt, if found. Null
/// means the confirm screen should fall back to manual entry (it
/// defaults to "now" and lets the user pick a different date/time).
final DateTime? date;
/// Two-letter state abbreviation of the station's address, if found
/// (e.g. "MO", "TX"). Null means no state could be confidently
/// identified — callers should treat that as "unknown", not "Missouri".
final String? state;
final String rawText;
ParsedReceipt({
this.gallons,
this.pricePerGallon,
this.totalCost,
this.date,
this.state,
required this.rawText,
});
}
@ -23,36 +58,91 @@ class ParsedReceipt {
/// to deriving a missing value from the other two when exactly one is
/// missing (total = gallons * price, etc).
class ReceiptParser {
// These use [ \t]* rather than \s* between a label and its value: \s
// matches newlines too, which let a number on one line match a
// completely unrelated label several lines further down (e.g. a price
// value followed, many lines later, by an incidental "Gallons" heading
// for a different column) — a real bug this surfaced against an actual
// receipt where "$1.999" ended up matching "...Gallons" two lines below
// it. A label and its own value are always on the same line.
static final _gallonsPatterns = [
RegExp(r'GALLONS?\s*[:\-]?\s*(\d+\.\d{2,3})', caseSensitive: false),
RegExp(r'\bGAL\b\s*[:\-]?\s*(\d+\.\d{2,3})', caseSensitive: false),
RegExp(r'(\d+\.\d{2,3})\s*GAL(?:LONS)?\b', caseSensitive: false),
RegExp(r'GALLONS?[ \t]*[:\-]?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
RegExp(r'\bGAL\b[ \t]*[:\-]?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
RegExp(r'(\d+\.\d{2,3})[ \t]*GAL(?:LONS)?\b', caseSensitive: false),
];
static final _pricePerGallonPatterns = [
RegExp(r'PRICE\s*/?\s*GAL(?:LON)?\s*[:\-]?\s*\$?\s*(\d+\.\d{2,3})',
RegExp(r'PRICE[ \t]*/?[ \t]*GAL(?:LON)?[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
caseSensitive: false),
RegExp(r'\bPPG\b\s*[:\-]?\s*\$?\s*(\d+\.\d{2,3})', caseSensitive: false),
RegExp(r'PER\s*GAL(?:LON)?\s*[:\-]?\s*\$?\s*(\d+\.\d{2,3})',
RegExp(r'\bPPG\b[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
RegExp(r'PER[ \t]*GAL(?:LON)?[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
caseSensitive: false),
RegExp(r'\$\s*/\s*GAL\s*[:\-]?\s*\$?\s*(\d+\.\d{2,3})',
RegExp(r'\$[ \t]*/[ \t]*GAL[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
caseSensitive: false),
];
static final _totalPatterns = [
RegExp(r'FUEL\s*TOTAL\s*[:\-]?\s*\$?\s*(\d+\.\d{2})', caseSensitive: false),
RegExp(r'SALE\s*TOTAL\s*[:\-]?\s*\$?\s*(\d+\.\d{2})', caseSensitive: false),
RegExp(r'AMOUNT\s*DUE\s*[:\-]?\s*\$?\s*(\d+\.\d{2})', caseSensitive: false),
RegExp(r'(?<!SUB)\bTOTAL\b\s*[:\-]?\s*\$?\s*(\d+\.\d{2})',
caseSensitive: false),
RegExp(r'AMOUNT\s*DUE[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
RegExp(r'FUEL\s*SALE[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
// Allows for words between TOTAL and the amount ("Total Sale $120.72",
// "Fuel Total: $44.44"), not just a bare colon/dash.
RegExp(r'(?<!SUB)\bTOTAL\b[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
];
/// Matches a bare or $-prefixed 3-decimal number anywhere in the text.
/// US fuel receipts overwhelmingly print both gallons pumped and price
/// per gallon with exactly 3 decimal places, and only the price gets a
/// currency symbol — a much more reliable signal than the label text
/// itself, which varies a lot and is often split from its value onto a
/// different line by a tabular "Pump / Gallons / Price" header row (in
/// which case the label-based patterns above never match at all).
///
/// Uses `(?!\d)` rather than `\b` after the number: some receipts print
/// the unit directly attached with no space ("17.364G"), and a digit
/// followed by a letter is *not* a `\b` boundary, so `\b` would miss it.
/// `(?<!-)` excludes negative amounts (e.g. a per-gallon discount line
/// like "Debit Di/GAL $-0.100"), which are never a real gallons/price
/// value and would otherwise get matched as one.
static final _threeDecimalNumberPattern = RegExp(r'(?<!-)(\$)?\s*(\d+\.\d{3})(?!\d)');
// Matches MM/DD/YYYY, MM-DD-YY, and similar — US gas receipts are
// consistent about digit-separator-digit-separator-digit ordering even
// though the separator (/ or -) and year length (2 or 4 digits) vary.
static final _datePattern = RegExp(r'\b(\d{1,2})[/-](\d{1,2})[/-](\d{2}|\d{4})\b');
// Optional AM/PM, optional seconds — covers "02:21", "10:11:00 AM", and
// "02:11PM" (no space before the meridiem) all at once.
static final _timePattern = RegExp(r'\b(\d{1,2}):(\d{2})(?::\d{2})?\s*([AaPp][Mm])?\b');
// A two-letter code directly followed by a ZIP is by far the strongest
// signal an address block gives us (very low false-positive rate); a
// code right after a comma is a weaker but still decent fallback for
// receipts that print "City, ST" without a visible ZIP alongside it.
// Both require the candidate to be checked against [usStateNames] before
// being trusted — a bare 2-letter scan alone would misfire constantly
// (e.g. "VI" for Visa, "IN" as the word "in", "OR" as the word "or").
static final _stateBeforeZipPattern = RegExp(r'\b([A-Z]{2})\s+\d{5}(?:-\d{4})?\b');
static final _stateAfterCommaPattern = RegExp(r',\s*([A-Z]{2})\b');
static ParsedReceipt parse(String text) {
final normalized = text.replaceAll(',', '');
final gallons = _firstMatch(_gallonsPatterns, normalized);
final pricePerGallon = _firstMatch(_pricePerGallonPatterns, normalized);
var totalCost = _firstMatch(_totalPatterns, normalized);
var gallons = _firstMatch(_gallonsPatterns, normalized);
var pricePerGallon = _firstMatch(_pricePerGallonPatterns, normalized);
final totalCost = _firstMatch(_totalPatterns, normalized);
if (gallons == null || pricePerGallon == null) {
for (final match in _threeDecimalNumberPattern.allMatches(normalized)) {
final value = double.tryParse(match.group(2)!);
if (value == null) continue;
final hasDollarSign = match.group(1) != null;
if (hasDollarSign) {
pricePerGallon ??= value;
} else {
gallons ??= value;
}
}
}
final derivedTotal = _deriveMissingValue(
gallons: gallons,
@ -64,10 +154,76 @@ class ReceiptParser {
gallons: derivedTotal.gallons,
pricePerGallon: derivedTotal.pricePerGallon,
totalCost: derivedTotal.totalCost,
date: _parseDate(normalized),
// Uses the original text, not the comma-stripped `normalized` copy —
// the comma-adjacency fallback pattern needs commas intact.
state: _parseState(text),
rawText: text,
);
}
/// Looks for a US state abbreviation in the station's address block. Only
/// trusts a candidate that's both a plausible address position (right
/// before a ZIP, or right after a comma) *and* a real state code — see
/// [usStateNames] and the patterns above for why both checks matter.
static String? _parseState(String text) {
final zipMatch = _stateBeforeZipPattern.firstMatch(text);
if (zipMatch != null) {
final code = zipMatch.group(1)!.toUpperCase();
if (usStateNames.containsKey(code)) return code;
}
final commaMatch = _stateAfterCommaPattern.firstMatch(text);
if (commaMatch != null) {
final code = commaMatch.group(1)!.toUpperCase();
if (usStateNames.containsKey(code)) return code;
}
return null;
}
/// Finds a date on the receipt and, if a time is printed nearby (same
/// line — within a short character window right after the date, since
/// receipts almost always print them adjacent, e.g. "DATE 3/26/22
/// 18:12" or "Date: ...\nTime: ..."), combines them. Falls back to
/// midnight if no time is found, and to null (manual entry) if no date
/// is found at all.
static DateTime? _parseDate(String text) {
final dateMatch = _datePattern.firstMatch(text);
if (dateMatch == null) return null;
final month = int.tryParse(dateMatch.group(1)!);
final day = int.tryParse(dateMatch.group(2)!);
var year = int.tryParse(dateMatch.group(3)!);
if (month == null || day == null || year == null) return null;
if (month < 1 || month > 12 || day < 1 || day > 31) return null;
if (year < 100) year += 2000;
final windowEnd = (dateMatch.end + 40).clamp(0, text.length);
final nearbyText = text.substring(dateMatch.end, windowEnd);
final timeMatch = _timePattern.firstMatch(nearbyText);
var hour = 0;
var minute = 0;
if (timeMatch != null) {
hour = int.tryParse(timeMatch.group(1)!) ?? 0;
minute = int.tryParse(timeMatch.group(2)!) ?? 0;
final meridiem = timeMatch.group(3)?.toUpperCase();
if (meridiem == 'PM' && hour != 12) hour += 12;
if (meridiem == 'AM' && hour == 12) hour = 0;
if (hour > 23 || minute > 59) {
hour = 0;
minute = 0;
}
}
try {
return DateTime(year, month, day, hour, minute);
} catch (_) {
return null;
}
}
static double? _firstMatch(List<RegExp> patterns, String text) {
for (final pattern in patterns) {
final match = pattern.firstMatch(text);