261 lines
11 KiB
Dart
261 lines
11 KiB
Dart
/// Two-letter USPS abbreviation to full name, for the 50 states + DC —
|
|
/// deliberately excludes territories (PR, VI, GU, ...): "VI" in particular
|
|
/// shows up on real receipts as a "Visa" abbreviation, and including it as
|
|
/// a valid state code would misfire on that.
|
|
const usStateNames = <String, String>{
|
|
'AL': 'Alabama', 'AK': 'Alaska', 'AZ': 'Arizona', 'AR': 'Arkansas',
|
|
'CA': 'California', 'CO': 'Colorado', 'CT': 'Connecticut', 'DE': 'Delaware',
|
|
'FL': 'Florida', 'GA': 'Georgia', 'HI': 'Hawaii', 'ID': 'Idaho',
|
|
'IL': 'Illinois', 'IN': 'Indiana', 'IA': 'Iowa', 'KS': 'Kansas',
|
|
'KY': 'Kentucky', 'LA': 'Louisiana', 'ME': 'Maine', 'MD': 'Maryland',
|
|
'MA': 'Massachusetts', 'MI': 'Michigan', 'MN': 'Minnesota',
|
|
'MS': 'Mississippi', 'MO': 'Missouri', 'MT': 'Montana', 'NE': 'Nebraska',
|
|
'NV': 'Nevada', 'NH': 'New Hampshire', 'NJ': 'New Jersey',
|
|
'NM': 'New Mexico', 'NY': 'New York', 'NC': 'North Carolina',
|
|
'ND': 'North Dakota', 'OH': 'Ohio', 'OK': 'Oklahoma', 'OR': 'Oregon',
|
|
'PA': 'Pennsylvania', 'RI': 'Rhode Island', 'SC': 'South Carolina',
|
|
'SD': 'South Dakota', 'TN': 'Tennessee', 'TX': 'Texas', 'UT': 'Utah',
|
|
'VT': 'Vermont', 'VA': 'Virginia', 'WA': 'Washington',
|
|
'WV': 'West Virginia', 'WI': 'Wisconsin', 'WY': 'Wyoming',
|
|
'DC': 'District of Columbia',
|
|
};
|
|
|
|
/// Best-effort values pulled out of OCR'd receipt text. Any field can be
|
|
/// null if it couldn't be found, and the confirm screen lets the user fill
|
|
/// in or correct whatever the parser got wrong.
|
|
class ParsedReceipt {
|
|
final double? gallons;
|
|
final double? pricePerGallon;
|
|
final double? totalCost;
|
|
|
|
/// The transaction date/time printed on the receipt, if found. Null
|
|
/// means the confirm screen should fall back to manual entry (it
|
|
/// defaults to "now" and lets the user pick a different date/time).
|
|
final DateTime? date;
|
|
|
|
/// Two-letter state abbreviation of the station's address, if found
|
|
/// (e.g. "MO", "TX"). Null means no state could be confidently
|
|
/// identified — callers should treat that as "unknown", not "Missouri".
|
|
final String? state;
|
|
|
|
final String rawText;
|
|
|
|
ParsedReceipt({
|
|
this.gallons,
|
|
this.pricePerGallon,
|
|
this.totalCost,
|
|
this.date,
|
|
this.state,
|
|
required this.rawText,
|
|
});
|
|
}
|
|
|
|
/// Parses free-form OCR text from a fuel receipt looking for the number of
|
|
/// gallons pumped, the price per gallon, and the total amount charged.
|
|
///
|
|
/// Fuel receipts vary a lot between stations, so this uses a handful of
|
|
/// label-based regexes rather than assuming a fixed layout, and falls back
|
|
/// to deriving a missing value from the other two when exactly one is
|
|
/// missing (total = gallons * price, etc).
|
|
class ReceiptParser {
|
|
// These use [ \t]* rather than \s* between a label and its value: \s
|
|
// matches newlines too, which let a number on one line match a
|
|
// completely unrelated label several lines further down (e.g. a price
|
|
// value followed, many lines later, by an incidental "Gallons" heading
|
|
// for a different column) — a real bug this surfaced against an actual
|
|
// receipt where "$1.999" ended up matching "...Gallons" two lines below
|
|
// it. A label and its own value are always on the same line.
|
|
static final _gallonsPatterns = [
|
|
RegExp(r'GALLONS?[ \t]*[:\-]?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
|
|
RegExp(r'\bGAL\b[ \t]*[:\-]?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
|
|
RegExp(r'(\d+\.\d{2,3})[ \t]*GAL(?:LONS)?\b', caseSensitive: false),
|
|
];
|
|
|
|
static final _pricePerGallonPatterns = [
|
|
RegExp(r'PRICE[ \t]*/?[ \t]*GAL(?:LON)?[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
|
|
caseSensitive: false),
|
|
RegExp(r'\bPPG\b[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
|
|
RegExp(r'PER[ \t]*GAL(?:LON)?[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
|
|
caseSensitive: false),
|
|
RegExp(r'\$[ \t]*/[ \t]*GAL[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
|
|
caseSensitive: false),
|
|
];
|
|
|
|
static final _totalPatterns = [
|
|
RegExp(r'AMOUNT\s*DUE[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
|
|
RegExp(r'FUEL\s*SALE[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
|
|
// Allows for words between TOTAL and the amount ("Total Sale $120.72",
|
|
// "Fuel Total: $44.44"), not just a bare colon/dash.
|
|
RegExp(r'(?<!SUB)\bTOTAL\b[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
|
|
];
|
|
|
|
/// Matches a bare or $-prefixed 3-decimal number anywhere in the text.
|
|
/// US fuel receipts overwhelmingly print both gallons pumped and price
|
|
/// per gallon with exactly 3 decimal places, and only the price gets a
|
|
/// currency symbol — a much more reliable signal than the label text
|
|
/// itself, which varies a lot and is often split from its value onto a
|
|
/// different line by a tabular "Pump / Gallons / Price" header row (in
|
|
/// which case the label-based patterns above never match at all).
|
|
///
|
|
/// Uses `(?!\d)` rather than `\b` after the number: some receipts print
|
|
/// the unit directly attached with no space ("17.364G"), and a digit
|
|
/// followed by a letter is *not* a `\b` boundary, so `\b` would miss it.
|
|
/// `(?<!-)` excludes negative amounts (e.g. a per-gallon discount line
|
|
/// like "Debit Di/GAL $-0.100"), which are never a real gallons/price
|
|
/// value and would otherwise get matched as one.
|
|
static final _threeDecimalNumberPattern = RegExp(r'(?<!-)(\$)?\s*(\d+\.\d{3})(?!\d)');
|
|
|
|
// Matches MM/DD/YYYY, MM-DD-YY, and similar — US gas receipts are
|
|
// consistent about digit-separator-digit-separator-digit ordering even
|
|
// though the separator (/ or -) and year length (2 or 4 digits) vary.
|
|
static final _datePattern = RegExp(r'\b(\d{1,2})[/-](\d{1,2})[/-](\d{2}|\d{4})\b');
|
|
|
|
// Optional AM/PM, optional seconds — covers "02:21", "10:11:00 AM", and
|
|
// "02:11PM" (no space before the meridiem) all at once.
|
|
static final _timePattern = RegExp(r'\b(\d{1,2}):(\d{2})(?::\d{2})?\s*([AaPp][Mm])?\b');
|
|
|
|
// A two-letter code directly followed by a ZIP is by far the strongest
|
|
// signal an address block gives us (very low false-positive rate); a
|
|
// code right after a comma is a weaker but still decent fallback for
|
|
// receipts that print "City, ST" without a visible ZIP alongside it.
|
|
// Both require the candidate to be checked against [usStateNames] before
|
|
// being trusted — a bare 2-letter scan alone would misfire constantly
|
|
// (e.g. "VI" for Visa, "IN" as the word "in", "OR" as the word "or").
|
|
static final _stateBeforeZipPattern = RegExp(r'\b([A-Z]{2})\s+\d{5}(?:-\d{4})?\b');
|
|
static final _stateAfterCommaPattern = RegExp(r',\s*([A-Z]{2})\b');
|
|
|
|
static ParsedReceipt parse(String text) {
|
|
final normalized = text.replaceAll(',', '');
|
|
|
|
var gallons = _firstMatch(_gallonsPatterns, normalized);
|
|
var pricePerGallon = _firstMatch(_pricePerGallonPatterns, normalized);
|
|
final totalCost = _firstMatch(_totalPatterns, normalized);
|
|
|
|
if (gallons == null || pricePerGallon == null) {
|
|
for (final match in _threeDecimalNumberPattern.allMatches(normalized)) {
|
|
final value = double.tryParse(match.group(2)!);
|
|
if (value == null) continue;
|
|
final hasDollarSign = match.group(1) != null;
|
|
if (hasDollarSign) {
|
|
pricePerGallon ??= value;
|
|
} else {
|
|
gallons ??= value;
|
|
}
|
|
}
|
|
}
|
|
|
|
final derivedTotal = _deriveMissingValue(
|
|
gallons: gallons,
|
|
pricePerGallon: pricePerGallon,
|
|
totalCost: totalCost,
|
|
);
|
|
|
|
return ParsedReceipt(
|
|
gallons: derivedTotal.gallons,
|
|
pricePerGallon: derivedTotal.pricePerGallon,
|
|
totalCost: derivedTotal.totalCost,
|
|
date: _parseDate(normalized),
|
|
// Uses the original text, not the comma-stripped `normalized` copy —
|
|
// the comma-adjacency fallback pattern needs commas intact.
|
|
state: _parseState(text),
|
|
rawText: text,
|
|
);
|
|
}
|
|
|
|
/// Looks for a US state abbreviation in the station's address block. Only
|
|
/// trusts a candidate that's both a plausible address position (right
|
|
/// before a ZIP, or right after a comma) *and* a real state code — see
|
|
/// [usStateNames] and the patterns above for why both checks matter.
|
|
static String? _parseState(String text) {
|
|
final zipMatch = _stateBeforeZipPattern.firstMatch(text);
|
|
if (zipMatch != null) {
|
|
final code = zipMatch.group(1)!.toUpperCase();
|
|
if (usStateNames.containsKey(code)) return code;
|
|
}
|
|
|
|
final commaMatch = _stateAfterCommaPattern.firstMatch(text);
|
|
if (commaMatch != null) {
|
|
final code = commaMatch.group(1)!.toUpperCase();
|
|
if (usStateNames.containsKey(code)) return code;
|
|
}
|
|
|
|
return null;
|
|
}
|
|
|
|
/// Finds a date on the receipt and, if a time is printed nearby (same
|
|
/// line — within a short character window right after the date, since
|
|
/// receipts almost always print them adjacent, e.g. "DATE 3/26/22
|
|
/// 18:12" or "Date: ...\nTime: ..."), combines them. Falls back to
|
|
/// midnight if no time is found, and to null (manual entry) if no date
|
|
/// is found at all.
|
|
static DateTime? _parseDate(String text) {
|
|
final dateMatch = _datePattern.firstMatch(text);
|
|
if (dateMatch == null) return null;
|
|
|
|
final month = int.tryParse(dateMatch.group(1)!);
|
|
final day = int.tryParse(dateMatch.group(2)!);
|
|
var year = int.tryParse(dateMatch.group(3)!);
|
|
if (month == null || day == null || year == null) return null;
|
|
if (month < 1 || month > 12 || day < 1 || day > 31) return null;
|
|
if (year < 100) year += 2000;
|
|
|
|
final windowEnd = (dateMatch.end + 40).clamp(0, text.length);
|
|
final nearbyText = text.substring(dateMatch.end, windowEnd);
|
|
final timeMatch = _timePattern.firstMatch(nearbyText);
|
|
|
|
var hour = 0;
|
|
var minute = 0;
|
|
if (timeMatch != null) {
|
|
hour = int.tryParse(timeMatch.group(1)!) ?? 0;
|
|
minute = int.tryParse(timeMatch.group(2)!) ?? 0;
|
|
final meridiem = timeMatch.group(3)?.toUpperCase();
|
|
if (meridiem == 'PM' && hour != 12) hour += 12;
|
|
if (meridiem == 'AM' && hour == 12) hour = 0;
|
|
if (hour > 23 || minute > 59) {
|
|
hour = 0;
|
|
minute = 0;
|
|
}
|
|
}
|
|
|
|
try {
|
|
return DateTime(year, month, day, hour, minute);
|
|
} catch (_) {
|
|
return null;
|
|
}
|
|
}
|
|
|
|
static double? _firstMatch(List<RegExp> patterns, String text) {
|
|
for (final pattern in patterns) {
|
|
final match = pattern.firstMatch(text);
|
|
if (match != null) {
|
|
return double.tryParse(match.group(1)!);
|
|
}
|
|
}
|
|
return null;
|
|
}
|
|
|
|
static ({double? gallons, double? pricePerGallon, double? totalCost})
|
|
_deriveMissingValue({
|
|
required double? gallons,
|
|
required double? pricePerGallon,
|
|
required double? totalCost,
|
|
}) {
|
|
final missingCount =
|
|
[gallons, pricePerGallon, totalCost].where((v) => v == null).length;
|
|
|
|
// Only safe to derive when exactly one of the three is missing.
|
|
if (missingCount != 1) {
|
|
return (gallons: gallons, pricePerGallon: pricePerGallon, totalCost: totalCost);
|
|
}
|
|
|
|
if (totalCost == null && gallons != null && pricePerGallon != null) {
|
|
totalCost = double.parse((gallons * pricePerGallon).toStringAsFixed(2));
|
|
} else if (gallons == null && totalCost != null && pricePerGallon != null && pricePerGallon > 0) {
|
|
gallons = double.parse((totalCost / pricePerGallon).toStringAsFixed(3));
|
|
} else if (pricePerGallon == null && totalCost != null && gallons != null && gallons > 0) {
|
|
pricePerGallon = double.parse((totalCost / gallons).toStringAsFixed(3));
|
|
}
|
|
|
|
return (gallons: gallons, pricePerGallon: pricePerGallon, totalCost: totalCost);
|
|
}
|
|
}
|