Remote sync of webdav
This commit is contained in:
parent
f01186df79
commit
483fbeafb6
44 changed files with 4842 additions and 975 deletions
|
|
@ -1,3 +1,25 @@
|
|||
/// Two-letter USPS abbreviation to full name, for the 50 states + DC —
|
||||
/// deliberately excludes territories (PR, VI, GU, ...): "VI" in particular
|
||||
/// shows up on real receipts as a "Visa" abbreviation, and including it as
|
||||
/// a valid state code would misfire on that.
|
||||
const usStateNames = <String, String>{
|
||||
'AL': 'Alabama', 'AK': 'Alaska', 'AZ': 'Arizona', 'AR': 'Arkansas',
|
||||
'CA': 'California', 'CO': 'Colorado', 'CT': 'Connecticut', 'DE': 'Delaware',
|
||||
'FL': 'Florida', 'GA': 'Georgia', 'HI': 'Hawaii', 'ID': 'Idaho',
|
||||
'IL': 'Illinois', 'IN': 'Indiana', 'IA': 'Iowa', 'KS': 'Kansas',
|
||||
'KY': 'Kentucky', 'LA': 'Louisiana', 'ME': 'Maine', 'MD': 'Maryland',
|
||||
'MA': 'Massachusetts', 'MI': 'Michigan', 'MN': 'Minnesota',
|
||||
'MS': 'Mississippi', 'MO': 'Missouri', 'MT': 'Montana', 'NE': 'Nebraska',
|
||||
'NV': 'Nevada', 'NH': 'New Hampshire', 'NJ': 'New Jersey',
|
||||
'NM': 'New Mexico', 'NY': 'New York', 'NC': 'North Carolina',
|
||||
'ND': 'North Dakota', 'OH': 'Ohio', 'OK': 'Oklahoma', 'OR': 'Oregon',
|
||||
'PA': 'Pennsylvania', 'RI': 'Rhode Island', 'SC': 'South Carolina',
|
||||
'SD': 'South Dakota', 'TN': 'Tennessee', 'TX': 'Texas', 'UT': 'Utah',
|
||||
'VT': 'Vermont', 'VA': 'Virginia', 'WA': 'Washington',
|
||||
'WV': 'West Virginia', 'WI': 'Wisconsin', 'WY': 'Wyoming',
|
||||
'DC': 'District of Columbia',
|
||||
};
|
||||
|
||||
/// Best-effort values pulled out of OCR'd receipt text. Any field can be
|
||||
/// null if it couldn't be found, and the confirm screen lets the user fill
|
||||
/// in or correct whatever the parser got wrong.
|
||||
|
|
@ -5,12 +27,25 @@ class ParsedReceipt {
|
|||
final double? gallons;
|
||||
final double? pricePerGallon;
|
||||
final double? totalCost;
|
||||
|
||||
/// The transaction date/time printed on the receipt, if found. Null
|
||||
/// means the confirm screen should fall back to manual entry (it
|
||||
/// defaults to "now" and lets the user pick a different date/time).
|
||||
final DateTime? date;
|
||||
|
||||
/// Two-letter state abbreviation of the station's address, if found
|
||||
/// (e.g. "MO", "TX"). Null means no state could be confidently
|
||||
/// identified — callers should treat that as "unknown", not "Missouri".
|
||||
final String? state;
|
||||
|
||||
final String rawText;
|
||||
|
||||
ParsedReceipt({
|
||||
this.gallons,
|
||||
this.pricePerGallon,
|
||||
this.totalCost,
|
||||
this.date,
|
||||
this.state,
|
||||
required this.rawText,
|
||||
});
|
||||
}
|
||||
|
|
@ -23,36 +58,91 @@ class ParsedReceipt {
|
|||
/// to deriving a missing value from the other two when exactly one is
|
||||
/// missing (total = gallons * price, etc).
|
||||
class ReceiptParser {
|
||||
// These use [ \t]* rather than \s* between a label and its value: \s
|
||||
// matches newlines too, which let a number on one line match a
|
||||
// completely unrelated label several lines further down (e.g. a price
|
||||
// value followed, many lines later, by an incidental "Gallons" heading
|
||||
// for a different column) — a real bug this surfaced against an actual
|
||||
// receipt where "$1.999" ended up matching "...Gallons" two lines below
|
||||
// it. A label and its own value are always on the same line.
|
||||
static final _gallonsPatterns = [
|
||||
RegExp(r'GALLONS?\s*[:\-]?\s*(\d+\.\d{2,3})', caseSensitive: false),
|
||||
RegExp(r'\bGAL\b\s*[:\-]?\s*(\d+\.\d{2,3})', caseSensitive: false),
|
||||
RegExp(r'(\d+\.\d{2,3})\s*GAL(?:LONS)?\b', caseSensitive: false),
|
||||
RegExp(r'GALLONS?[ \t]*[:\-]?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
|
||||
RegExp(r'\bGAL\b[ \t]*[:\-]?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
|
||||
RegExp(r'(\d+\.\d{2,3})[ \t]*GAL(?:LONS)?\b', caseSensitive: false),
|
||||
];
|
||||
|
||||
static final _pricePerGallonPatterns = [
|
||||
RegExp(r'PRICE\s*/?\s*GAL(?:LON)?\s*[:\-]?\s*\$?\s*(\d+\.\d{2,3})',
|
||||
RegExp(r'PRICE[ \t]*/?[ \t]*GAL(?:LON)?[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
|
||||
caseSensitive: false),
|
||||
RegExp(r'\bPPG\b\s*[:\-]?\s*\$?\s*(\d+\.\d{2,3})', caseSensitive: false),
|
||||
RegExp(r'PER\s*GAL(?:LON)?\s*[:\-]?\s*\$?\s*(\d+\.\d{2,3})',
|
||||
RegExp(r'\bPPG\b[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})', caseSensitive: false),
|
||||
RegExp(r'PER[ \t]*GAL(?:LON)?[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
|
||||
caseSensitive: false),
|
||||
RegExp(r'\$\s*/\s*GAL\s*[:\-]?\s*\$?\s*(\d+\.\d{2,3})',
|
||||
RegExp(r'\$[ \t]*/[ \t]*GAL[ \t]*[:\-]?[ \t]*\$?[ \t]*(\d+\.\d{2,3})',
|
||||
caseSensitive: false),
|
||||
];
|
||||
|
||||
static final _totalPatterns = [
|
||||
RegExp(r'FUEL\s*TOTAL\s*[:\-]?\s*\$?\s*(\d+\.\d{2})', caseSensitive: false),
|
||||
RegExp(r'SALE\s*TOTAL\s*[:\-]?\s*\$?\s*(\d+\.\d{2})', caseSensitive: false),
|
||||
RegExp(r'AMOUNT\s*DUE\s*[:\-]?\s*\$?\s*(\d+\.\d{2})', caseSensitive: false),
|
||||
RegExp(r'(?<!SUB)\bTOTAL\b\s*[:\-]?\s*\$?\s*(\d+\.\d{2})',
|
||||
caseSensitive: false),
|
||||
RegExp(r'AMOUNT\s*DUE[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
|
||||
RegExp(r'FUEL\s*SALE[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
|
||||
// Allows for words between TOTAL and the amount ("Total Sale $120.72",
|
||||
// "Fuel Total: $44.44"), not just a bare colon/dash.
|
||||
RegExp(r'(?<!SUB)\bTOTAL\b[^\n\d]{0,20}(\d+\.\d{2})', caseSensitive: false),
|
||||
];
|
||||
|
||||
/// Matches a bare or $-prefixed 3-decimal number anywhere in the text.
|
||||
/// US fuel receipts overwhelmingly print both gallons pumped and price
|
||||
/// per gallon with exactly 3 decimal places, and only the price gets a
|
||||
/// currency symbol — a much more reliable signal than the label text
|
||||
/// itself, which varies a lot and is often split from its value onto a
|
||||
/// different line by a tabular "Pump / Gallons / Price" header row (in
|
||||
/// which case the label-based patterns above never match at all).
|
||||
///
|
||||
/// Uses `(?!\d)` rather than `\b` after the number: some receipts print
|
||||
/// the unit directly attached with no space ("17.364G"), and a digit
|
||||
/// followed by a letter is *not* a `\b` boundary, so `\b` would miss it.
|
||||
/// `(?<!-)` excludes negative amounts (e.g. a per-gallon discount line
|
||||
/// like "Debit Di/GAL $-0.100"), which are never a real gallons/price
|
||||
/// value and would otherwise get matched as one.
|
||||
static final _threeDecimalNumberPattern = RegExp(r'(?<!-)(\$)?\s*(\d+\.\d{3})(?!\d)');
|
||||
|
||||
// Matches MM/DD/YYYY, MM-DD-YY, and similar — US gas receipts are
|
||||
// consistent about digit-separator-digit-separator-digit ordering even
|
||||
// though the separator (/ or -) and year length (2 or 4 digits) vary.
|
||||
static final _datePattern = RegExp(r'\b(\d{1,2})[/-](\d{1,2})[/-](\d{2}|\d{4})\b');
|
||||
|
||||
// Optional AM/PM, optional seconds — covers "02:21", "10:11:00 AM", and
|
||||
// "02:11PM" (no space before the meridiem) all at once.
|
||||
static final _timePattern = RegExp(r'\b(\d{1,2}):(\d{2})(?::\d{2})?\s*([AaPp][Mm])?\b');
|
||||
|
||||
// A two-letter code directly followed by a ZIP is by far the strongest
|
||||
// signal an address block gives us (very low false-positive rate); a
|
||||
// code right after a comma is a weaker but still decent fallback for
|
||||
// receipts that print "City, ST" without a visible ZIP alongside it.
|
||||
// Both require the candidate to be checked against [usStateNames] before
|
||||
// being trusted — a bare 2-letter scan alone would misfire constantly
|
||||
// (e.g. "VI" for Visa, "IN" as the word "in", "OR" as the word "or").
|
||||
static final _stateBeforeZipPattern = RegExp(r'\b([A-Z]{2})\s+\d{5}(?:-\d{4})?\b');
|
||||
static final _stateAfterCommaPattern = RegExp(r',\s*([A-Z]{2})\b');
|
||||
|
||||
static ParsedReceipt parse(String text) {
|
||||
final normalized = text.replaceAll(',', '');
|
||||
|
||||
final gallons = _firstMatch(_gallonsPatterns, normalized);
|
||||
final pricePerGallon = _firstMatch(_pricePerGallonPatterns, normalized);
|
||||
var totalCost = _firstMatch(_totalPatterns, normalized);
|
||||
var gallons = _firstMatch(_gallonsPatterns, normalized);
|
||||
var pricePerGallon = _firstMatch(_pricePerGallonPatterns, normalized);
|
||||
final totalCost = _firstMatch(_totalPatterns, normalized);
|
||||
|
||||
if (gallons == null || pricePerGallon == null) {
|
||||
for (final match in _threeDecimalNumberPattern.allMatches(normalized)) {
|
||||
final value = double.tryParse(match.group(2)!);
|
||||
if (value == null) continue;
|
||||
final hasDollarSign = match.group(1) != null;
|
||||
if (hasDollarSign) {
|
||||
pricePerGallon ??= value;
|
||||
} else {
|
||||
gallons ??= value;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
final derivedTotal = _deriveMissingValue(
|
||||
gallons: gallons,
|
||||
|
|
@ -64,10 +154,76 @@ class ReceiptParser {
|
|||
gallons: derivedTotal.gallons,
|
||||
pricePerGallon: derivedTotal.pricePerGallon,
|
||||
totalCost: derivedTotal.totalCost,
|
||||
date: _parseDate(normalized),
|
||||
// Uses the original text, not the comma-stripped `normalized` copy —
|
||||
// the comma-adjacency fallback pattern needs commas intact.
|
||||
state: _parseState(text),
|
||||
rawText: text,
|
||||
);
|
||||
}
|
||||
|
||||
/// Looks for a US state abbreviation in the station's address block. Only
|
||||
/// trusts a candidate that's both a plausible address position (right
|
||||
/// before a ZIP, or right after a comma) *and* a real state code — see
|
||||
/// [usStateNames] and the patterns above for why both checks matter.
|
||||
static String? _parseState(String text) {
|
||||
final zipMatch = _stateBeforeZipPattern.firstMatch(text);
|
||||
if (zipMatch != null) {
|
||||
final code = zipMatch.group(1)!.toUpperCase();
|
||||
if (usStateNames.containsKey(code)) return code;
|
||||
}
|
||||
|
||||
final commaMatch = _stateAfterCommaPattern.firstMatch(text);
|
||||
if (commaMatch != null) {
|
||||
final code = commaMatch.group(1)!.toUpperCase();
|
||||
if (usStateNames.containsKey(code)) return code;
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
/// Finds a date on the receipt and, if a time is printed nearby (same
|
||||
/// line — within a short character window right after the date, since
|
||||
/// receipts almost always print them adjacent, e.g. "DATE 3/26/22
|
||||
/// 18:12" or "Date: ...\nTime: ..."), combines them. Falls back to
|
||||
/// midnight if no time is found, and to null (manual entry) if no date
|
||||
/// is found at all.
|
||||
static DateTime? _parseDate(String text) {
|
||||
final dateMatch = _datePattern.firstMatch(text);
|
||||
if (dateMatch == null) return null;
|
||||
|
||||
final month = int.tryParse(dateMatch.group(1)!);
|
||||
final day = int.tryParse(dateMatch.group(2)!);
|
||||
var year = int.tryParse(dateMatch.group(3)!);
|
||||
if (month == null || day == null || year == null) return null;
|
||||
if (month < 1 || month > 12 || day < 1 || day > 31) return null;
|
||||
if (year < 100) year += 2000;
|
||||
|
||||
final windowEnd = (dateMatch.end + 40).clamp(0, text.length);
|
||||
final nearbyText = text.substring(dateMatch.end, windowEnd);
|
||||
final timeMatch = _timePattern.firstMatch(nearbyText);
|
||||
|
||||
var hour = 0;
|
||||
var minute = 0;
|
||||
if (timeMatch != null) {
|
||||
hour = int.tryParse(timeMatch.group(1)!) ?? 0;
|
||||
minute = int.tryParse(timeMatch.group(2)!) ?? 0;
|
||||
final meridiem = timeMatch.group(3)?.toUpperCase();
|
||||
if (meridiem == 'PM' && hour != 12) hour += 12;
|
||||
if (meridiem == 'AM' && hour == 12) hour = 0;
|
||||
if (hour > 23 || minute > 59) {
|
||||
hour = 0;
|
||||
minute = 0;
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
return DateTime(year, month, day, hour, minute);
|
||||
} catch (_) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
static double? _firstMatch(List<RegExp> patterns, String text) {
|
||||
for (final pattern in patterns) {
|
||||
final match = pattern.firstMatch(text);
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue