A whole bunch of changes
This commit is contained in:
parent
901ec678a3
commit
956231048e
122 changed files with 5442 additions and 914 deletions
|
|
@ -2,18 +2,24 @@
|
|||
/// sticker, dashboard plate, title, etc).
|
||||
///
|
||||
/// Real VINs never contain the letters I, O, or Q (they're excluded from
|
||||
/// the standard specifically so they can't be confused with 1 and 0), so
|
||||
/// requiring that charset both matches genuine VINs and rules out a lot of
|
||||
/// incidental 17-character noise elsewhere on the sticker (barcodes text,
|
||||
/// weight ratings, date codes, etc).
|
||||
/// the standard specifically so they can't be confused with 1, 0, and 9 —
|
||||
/// which is exactly the kind of confusion OCR itself is prone to: a
|
||||
/// genuine VIN's "0" is sometimes read back as "O"). So rather than reject
|
||||
/// a candidate for containing an I or O, [_normalize] treats it as a
|
||||
/// misread digit and corrects it — Q is left alone, since it doesn't
|
||||
/// closely resemble any digit and so is more likely to mean the candidate
|
||||
/// isn't a VIN at all (e.g. noise from elsewhere on the sticker).
|
||||
class VinParser {
|
||||
// V\.?\s*I\.?\s*N tolerates the "V.I.N" (period between each letter)
|
||||
// styling common on manufacturer compliance/data plates, not just the
|
||||
// plain "VIN" a door-jamb sticker or title usually uses.
|
||||
static final _labeledVinPattern =
|
||||
RegExp(r'VIN[:\s]*([A-HJ-NPR-Z0-9]{17})\b', caseSensitive: false);
|
||||
RegExp(r'V\.?\s*I\.?\s*N[:.\s]*([A-PR-Z0-9]{17})\b', caseSensitive: false);
|
||||
|
||||
// \b on both sides matters: since digits and letters are both "word"
|
||||
// characters, this only matches a maximal run of exactly 17 eligible
|
||||
// characters — not a 17-character slice out of an 18+ character run.
|
||||
static final _bareVinPattern = RegExp(r'\b([A-HJ-NPR-Z0-9]{17})\b', caseSensitive: false);
|
||||
static final _bareVinPattern = RegExp(r'\b([A-PR-Z0-9]{17})\b', caseSensitive: false);
|
||||
|
||||
/// Returns the VIN in uppercase, or null if nothing matching the VIN
|
||||
/// charset/length was found. A labeled "VIN: ..." match is preferred
|
||||
|
|
@ -21,11 +27,24 @@ class VinParser {
|
|||
/// one candidate (e.g. also a 17-digit tire/parts barcode number).
|
||||
static String? parse(String text) {
|
||||
final labeled = _labeledVinPattern.firstMatch(text);
|
||||
if (labeled != null) return labeled.group(1)!.toUpperCase();
|
||||
if (labeled != null) return _normalize(labeled.group(1)!);
|
||||
|
||||
final bare = _bareVinPattern.firstMatch(text);
|
||||
if (bare != null) return bare.group(1)!.toUpperCase();
|
||||
if (bare != null) return _normalize(bare.group(1)!);
|
||||
|
||||
// Last resort: OCR sometimes splits a VIN across a stray space (e.g.
|
||||
// an etched/curved surface gets read as two separate text regions) —
|
||||
// retry once against the text with all whitespace removed. Tried last,
|
||||
// after both whitespace-preserving attempts, since collapsing
|
||||
// whitespace elsewhere in a busy label risks gluing unrelated text
|
||||
// together into a spurious 17-character run.
|
||||
final stripped = text.replaceAll(RegExp(r'\s+'), '');
|
||||
final strippedMatch = _bareVinPattern.firstMatch(stripped);
|
||||
if (strippedMatch != null) return _normalize(strippedMatch.group(1)!);
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
static String _normalize(String candidate) =>
|
||||
candidate.toUpperCase().replaceAll('O', '0').replaceAll('I', '1');
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue