/** * parser.js — Парсинг и сверка данных * * Определение типа файла, извлечение ведомостей и транзакций, * сопоставление по водителям + гос. номерам, расчёт расхождений. */ import { readExcel } from './utils.js'; /* ================================================================ КОНСТАНТЫ ================================================================ */ /** Заголовок, по которому распознаётся ведомость */ const VEDOMOST_HEADER = 'ВЕДОМОСТЬ ПРИЕМА НАЛИЧНЫХ СРЕДСТВ'; /** Столбцы, по которым распознаётся выгрузка транзакций */ const TRANSACTION_COLS = ['CONDUCTOR', 'DATE', 'TARIF_SUM']; /** Латинские буквы, которые заменяются на кириллические в гос. номерах */ const LATIN_TO_CYRILLIC = { 'A': 'А', 'B': 'В', 'C': 'С', 'E': 'Е', 'H': 'Н', 'K': 'К', 'M': 'М', 'O': 'О', 'P': 'Р', 'T': 'Т', 'X': 'Х', 'Y': 'У' }; /** Порог схожести ФИО для нечёткого сравнения (0..1) */ const FUZZY_THRESHOLD = 0.88; /* ================================================================ ТИПИЗАЦИЯ ФАЙЛОВ ================================================================ */ /** * Определить тип файла: 'vedomost' | 'transaction' | null * @param {string} filename * @param {string[][]} firstSheet - первый лист книги * @returns {string|null} */ export function detectFileType(filename, firstSheet) { // 1. Проверка по заголовку ведомости if (hasHeader(firstSheet, VEDOMOST_HEADER)) { return 'vedomost'; } // 2. Проверка по названию файла const lower = filename.toLowerCase(); if (lower.includes('transaction') || lower.includes('transactions') || lower.includes('выгрузка')) { return 'transaction'; } // 3. Проверка по столбцам транзакций (первые 3 строки) if (hasTransactionColumns(firstSheet)) { return 'transaction'; } // 4. Если есть знакомые заголовки — ведомость if (hasHeader(firstSheet, 'ВЕДОМОСТЬ')) { return 'vedomost'; } return null; } /** * Проверить, содержит ли лист указанный заголовок */ function hasHeader(sheetData, header) { if (!sheetData || sheetData.length === 0) return false; // Проверяем первые 20 строк for (let i = 0; i < Math.min(sheetData.length, 20); i++) { const row = sheetData[i]; if (!row) continue; const joined = row.filter(c => c != null).map(String).join(' ').toUpperCase(); if (joined.includes(header.toUpperCase())) return true; } return false; } /** * Проверить, содержит ли лист столбцы транзакций */ function hasTransactionColumns(sheetData) { if (!sheetData || sheetData.length === 0) return false; // Проверяем первые 5 строк for (let i = 0; i < Math.min(sheetData.length, 5); i++) { const row = sheetData[i]; if (!row) continue; const cells = row.map(c => String(c).toUpperCase().trim()); const hasConductor = cells.some(c => c.includes('CONDUCTOR')); const hasDate = cells.some(c => c.includes('DATE')); const hasTarif = cells.some(c => c.includes('TARIF_SUM') || c.includes('TARIF')); if (hasConductor && hasDate && hasTarif) return true; } return false; } /* ================================================================ НОРМАЛИЗАЦИЯ ДАННЫХ ================================================================ */ /** * Нормализовать гос. номер: заменить латиницу на кириллицу, удалить лишнее * @param {string} plate * @returns {string} */ export function normalizePlate(plate) { if (!plate) return ''; let s = String(plate).toUpperCase().trim(); // Замена латинских букв на кириллические s = s.split('').map(ch => LATIN_TO_CYRILLIC[ch] || ch).join(''); // Удаление пробелов, дефисов и непечатных символов, кроме букв и цифр s = s.replace(/[^А-ЯЁA-Z0-9]/gi, ''); return s; } /** * Нормализовать ФИО: верхний регистр, Ё→Е, обрезка * @param {string} name * @returns {string} */ export function normalizeName(name) { if (!name) return ''; return String(name) .toUpperCase() .trim() .replace(/Ё/g, 'Е') .replace(/\s+/g, ' '); } /** * Нечёткое сравнение ФИО (расстояние Левенштейна → схожесть) * @param {string} a * @param {string} b * @returns {number} 0..1 */ export function nameSimilarity(a, b) { if (!a || !b) return 0; const s1 = normalizeName(a); const s2 = normalizeName(b); if (s1 === s2) return 1; if (s1.length === 0 || s2.length === 0) return 0; const dist = levenshtein(s1, s2); const maxLen = Math.max(s1.length, s2.length); return 1 - dist / maxLen; } /** * Расстояние Левенштейна */ function levenshtein(a, b) { const m = a.length, n = b.length; const dp = Array.from({ length: m + 1 }, () => Array(n + 1).fill(0)); for (let i = 0; i <= m; i++) dp[i][0] = i; for (let j = 0; j <= n; j++) dp[0][j] = j; for (let i = 1; i <= m; i++) { for (let j = 1; j <= n; j++) { dp[i][j] = a[i - 1] === b[j - 1] ? dp[i - 1][j - 1] : 1 + Math.min(dp[i - 1][j], dp[i][j - 1], dp[i - 1][j - 1]); } } return dp[m][n]; } /* ================================================================ ПАРСИНГ ВЕДОМОСТИ ================================================================ */ /** * Распарсить ведомость и вернуть структурированные данные * @param {string[][]} sheetData - массив строк листа * @returns {Object} - { docNumber, date, cashier, sections: { city, suburb }, drivers: [] } */ export function parseVedomost(sheetData) { if (!sheetData || sheetData.length === 0) return null; const result = { docNumber: '', date: '', cashier: '', sections: { city: [], suburb: [] }, drivers: [] }; let currentSection = null; // 'city' | 'suburb' let headerFound = false; let tableStarted = false; let tableEnded = false; for (let i = 0; i < sheetData.length; i++) { const row = sheetData[i]; if (!row || row.length === 0) continue; const cells = row.map(c => String(c).trim()); // Поиск заголовка ведомости if (!headerFound) { const joined = cells.join(' ').toUpperCase(); if (joined.includes(VEDOMOST_HEADER)) { headerFound = true; } continue; } // Номер документа и дата if (!result.docNumber) { const numIdx = cells.findIndex(c => c.toUpperCase().includes('ДОКУМЕНТ')); if (numIdx >= 0) { result.docNumber = cells[numIdx + 1] || ''; // Дата может быть в той же строке или следующей const dateIdx = cells.findIndex(c => /^\d{2}[./-]\d{2}[./-]\d{4}$/.test(c)); if (dateIdx >= 0) result.date = cells[dateIdx]; } } // Дата (если не нашли выше) if (!result.date) { const dateMatch = cells.find(c => /^\d{2}[./-]\d{2}[./-]\d{4}$/.test(c)); if (dateMatch) result.date = dateMatch; } // Кассир if (!result.cashier) { const cashIdx = cells.findIndex(c => c.toUpperCase().includes('КАССИР')); if (cashIdx >= 0) { // Ищем ФИО после слова "Кассир" const nameIdx = cashIdx + 1; if (nameIdx < cells.length && cells[nameIdx].length > 2) { result.cashier = cells[nameIdx]; } } } // Определение секции (Город / Пригород) const joinedRow = cells.join(' ').toUpperCase(); if (joinedRow.includes('ГОРОД') || joinedRow.includes('ГОР.') || (joinedRow.includes('ГОР') && !joinedRow.includes('ПРИГОР'))) { currentSection = 'city'; tableStarted = false; // Таблица начинается после заголовка секции continue; } if (joinedRow.includes('ПРИГОРОД') || joinedRow.includes('ПРИГ.')) { currentSection = 'suburb'; tableStarted = false; continue; } // Пропуск пустых и служебных строк if (cells.every(c => c === '')) { if (tableStarted) tableEnded = true; // Пустая строка = конец таблицы continue; } // Определение начала таблицы с водителями const headerRow = cells.join(' ').toUpperCase(); if (headerRow.includes('ФИО') || headerRow.includes('Ф.И.О') || headerRow.includes('ВОДИТЕЛЬ') || headerRow.includes('ФАМИЛИЯ') || (headerRow.includes('ФИО') && (headerRow.includes('СУММА') || headerRow.includes('МАРШРУТ')))) { tableStarted = true; tableEnded = false; continue; } // Если таблица началась, закончилась и мы снова видим непустую строку — возможно новая секция if (tableEnded && !tableStarted) continue; // Парсинг строки водителя (только внутри таблицы) if (tableStarted && !tableEnded && currentSection) { const driver = parseDriverRow(cells, currentSection); if (driver) { result.drivers.push(driver); result.sections[currentSection].push(driver); } } } return result; } /** * Распарсить строку водителя из ведомости * @param {string[]} cells * @param {string} section - 'city' | 'suburb' * @returns {Object|null} */ function parseDriverRow(cells, section) { if (!cells || cells.length < 2) return null; // Фильтр: отбрасываем совсем пустые строки const nonEmpty = cells.filter(c => c.trim() !== ''); if (nonEmpty.length < 2) return null; // Пропускаем итоговые строки const joined = cells.join(' ').toUpperCase(); if (joined.includes('ИТОГО') || joined.includes('ВСЕГО') || joined.includes('ПО РАЗДЕЛУ')) { return null; } // Ищем ФИО — обычно первая колонка const fio = cells[0] || ''; if (fio.length < 3 || /^\d+$/.test(fio)) return null; // Ищем гос. номер — паттерн: буква + 3 цифры + 2 буквы + 2-3 цифры (регион) let plate = ''; let plateIdx = -1; for (let j = 0; j < cells.length; j++) { const cleaned = cells[j].replace(/[\s-]/g, '').toUpperCase(); if (/^[А-ЯA-Z]{1}\d{3}[А-ЯA-Z]{2}\d{2,3}$/.test(cleaned) || /^[А-ЯA-Z]{1}\d{3}[А-ЯA-Z]{2}$/.test(cleaned)) { plate = cleaned; plateIdx = j; break; } } // Ищем маршрут (номер маршрута) let route = ''; for (let j = 1; j < cells.length; j++) { const c = cells[j].trim(); if (/^\d{1,3}$/.test(c) && c !== plate) { route = c; break; } } // Ищем сумму — последнее число или после ключевых слов let sum = 0; for (let j = cells.length - 1; j >= 0; j--) { const c = cells[j].trim().replace(/\s/g, '').replace(',', '.'); const num = parseFloat(c); if (!isNaN(num) && num > 0) { // Проверяем, что это не номер маршрута и не индекс if (j !== plateIdx && j !== 0) { sum = num; break; } } } return { fio: normalizeName(fio), plate: normalizePlate(plate), section, route, sum, // "Сдано" — итого на сумму cash: 0, // будет заполнено позже nonCash: 0 }; } /* ================================================================ ПАРСИНГ ТРАНЗАКЦИЙ ================================================================ */ /** * Распарсить выгрузку транзакций * @param {string[][]} sheetData * @returns {Object[]} - массив транзакций */ export function parseTransactions(sheetData) { if (!sheetData || sheetData.length === 0) return []; // Найдём строку заголовков и индексы колонок let headerRow = -1; let colIndexes = {}; for (let i = 0; i < Math.min(sheetData.length, 10); i++) { const row = sheetData[i]; if (!row) continue; const cells = row.map(c => String(c).toUpperCase().trim()); const idxDate = cells.findIndex(c => c.includes('DATE')); const idxConductor = cells.findIndex(c => c.includes('CONDUCTOR')); const idxTarif = cells.findIndex(c => c.includes('TARIF_SUM') || (c.includes('TARIF') && !c.includes('TARIF_PAY'))); const idxCar = cells.findIndex(c => c.includes('CAR') || c.includes('PLATE') || c.includes('AUTO')); const idxRoute = cells.findIndex(c => c.includes('ROUTE') || c.includes('RUTE')); const idxTarifPay = cells.findIndex(c => c.includes('TARIF_PAY') || c.includes('TARIFPAY')); if (idxDate >= 0 && idxConductor >= 0 && idxTarif >= 0) { headerRow = i; colIndexes = { date: idxDate, conductor: idxConductor, tarifSum: idxTarif, car: idxCar >= 0 ? idxCar : -1, route: idxRoute >= 0 ? idxRoute : -1, tarifPay: idxTarifPay >= 0 ? idxTarifPay : -1 }; break; } } if (headerRow < 0) return []; // Парсинг строк данных const transactions = []; for (let i = headerRow + 1; i < sheetData.length; i++) { const row = sheetData[i]; if (!row || row.length === 0) continue; const date = String(row[colIndexes.date] || '').trim(); const conductor = String(row[colIndexes.conductor] || '').trim(); const tarifSumStr = String(row[colIndexes.tarifSum] || '0').trim().replace(',', '.'); const tarifSum = parseFloat(tarifSumStr); // Пропускаем пустые или невалидные строки if (!date || !conductor || isNaN(tarifSum)) continue; // Проверка, что это строка данных, а не заголовок if (conductor.toUpperCase().includes('CONDUCTOR')) continue; const car = colIndexes.car >= 0 ? String(row[colIndexes.car] || '').trim() : ''; const route = colIndexes.route >= 0 ? String(row[colIndexes.route] || '').trim() : ''; transactions.push({ date: date, conductor: normalizeName(conductor), plate: normalizePlate(car), route: route, tarifSum: tarifSum, // в копейках tarifRub: tarifSum / 100, // в рублях tarifPay: colIndexes.tarifPay >= 0 ? parseFloat(String(row[colIndexes.tarifPay] || '0').replace(',', '.')) : 0 }); } return transactions; } /* ================================================================ СВЕРКА ================================================================ */ /** * Выполнить сверку: сопоставить ведомости и транзакции * @param {Object[]} vedomosti - массив распарсенных ведомостей * @param {Object[]} transactions - массив транзакций * @returns {Object} - результат сверки */ export function reconcile(vedomosti, transactions) { // 1. Собираем уникальные даты из ведомостей const vedomostDates = new Set(); vedomosti.forEach(v => { if (v.date) vedomostDates.add(normalizeDate(v.date)); }); // 2. Фильтруем транзакции только по датам ведомостей const filteredTransactions = transactions.filter(t => { const tDate = normalizeDate(t.date); return vedomostDates.has(tDate); }); // 3. Группируем транзакции по (водитель + гос. номер) const txByDriver = new Map(); filteredTransactions.forEach(t => { const key = `${t.conductor}|${t.plate || ''}`; if (!txByDriver.has(key)) { txByDriver.set(key, { conductor: t.conductor, plate: t.plate, totalTarif: 0, txCount: 0, dates: new Set() }); } const entry = txByDriver.get(key); entry.totalTarif += t.tarifRub; entry.txCount++; entry.dates.add(normalizeDate(t.date)); }); // 4. Собираем водителей из ведомостей и сопоставляем const driverMap = new Map(); // ключ: нормализованное ФИО|номер const allVedomostDrivers = []; vedomosti.forEach(v => { v.drivers.forEach(d => { const key = `${d.fio}|${d.plate}`; if (!driverMap.has(key)) { driverMap.set(key, { fio: d.fio, plate: d.plate, route: d.route, section: d.section, cashier: v.cashier, vedomostDate: v.date, totalGiven: 0 // Сдано }); } driverMap.get(key).totalGiven += d.sum; allVedomostDrivers.push(d); }); }); // 5. Сопоставление: для каждого водителя из ведомости ищем транзакции const results = []; const unmatchedTx = []; // транзакции без пары в ведомости // Прямое сопоставление const usedTxKeys = new Set(); driverMap.forEach((vd, key) => { const [fio, plate] = key.split('|'); // Точное совпадение по (ФИО + номер) const txKey = `${vd.fio}|${vd.plate}`; let txEntry = txByDriver.get(txKey); // Нечёткое совпадение по ФИО, если точного нет if (!txEntry) { let bestSim = 0; let bestKey = ''; txByDriver.forEach((entry, k) => { const [txFio, txPlate] = k.split('|'); // Сначала проверяем совпадение номера const plateMatch = txPlate && vd.plate && normalizePlate(txPlate) === normalizePlate(vd.plate); const sim = nameSimilarity(vd.fio, txFio); if (plateMatch && sim >= FUZZY_THRESHOLD && sim > bestSim) { bestSim = sim; bestKey = k; } }); if (bestKey) { txEntry = txByDriver.get(bestKey); usedTxKeys.add(bestKey); } } else { usedTxKeys.add(txKey); } const totalCollected = txEntry ? txEntry.totalTarif : 0; const diff = txEntry ? Math.round(totalCollected - vd.totalGiven) : -vd.totalGiven; results.push({ driver: vd.fio, car: vd.plate || '—', route: vd.route || '—', section: vd.section === 'city' ? 'Город' : 'Пригород', cashier: vd.cashier || '—', date: vd.vedomostDate || '—', given: Math.round(vd.totalGiven), collected: Math.round(totalCollected), diff: diff, txCount: txEntry ? txEntry.txCount : 0, status: diff === 0 ? 'ok' : (Math.abs(diff) <= 10 ? 'warn' : 'err') }); }); // Не сопоставленные транзакции txByDriver.forEach((entry, key) => { if (!usedTxKeys.has(key)) { unmatchedTx.push({ driver: entry.conductor, plate: entry.plate || '—', totalCollected: Math.round(entry.totalTarif), txCount: entry.txCount }); } }); // 6. Агрегация по кассирам const cashierMap = new Map(); results.forEach(r => { const key = r.cashier; if (!cashierMap.has(key)) { cashierMap.set(key, { cashier: key, count: 0, given: 0, collected: 0, diff: 0 }); } const c = cashierMap.get(key); c.count++; c.given += r.given; c.collected += r.collected; c.diff += r.diff; }); // 7. Агрегация по маршрутам const routeMap = new Map(); results.forEach(r => { const key = r.route || 'Без маршрута'; if (!routeMap.has(key)) { routeMap.set(key, { route: key, type: r.section, drivers: 0, sum: 0 }); } const rm = routeMap.get(key); rm.drivers++; rm.sum += r.given; }); // 8. Выводы const totalGiven = results.reduce((s, r) => s + r.given, 0); const totalCollected = results.reduce((s, r) => s + r.collected, 0); const totalDiff = results.reduce((s, r) => s + r.diff, 0); const discrepancies = results.filter(r => r.diff !== 0); return { drivers: results, cashiers: Array.from(cashierMap.values()), routes: Array.from(routeMap.values()), unmatchedTx, conclusions: { totalDrivers: results.length, totalGiven, totalCollected, totalDiff, discrepanciesCount: discrepancies.length, matchedPercent: results.length > 0 ? Math.round((results.length - discrepancies.length) / results.length * 100) : 0 } }; } /* ================================================================ ВСПОМОГАТЕЛЬНЫЕ ФУНКЦИИ ================================================================ */ /** * Нормализовать дату к формату YYYY-MM-DD * @param {string} dateStr * @returns {string} */ function normalizeDate(dateStr) { if (!dateStr) return ''; let d = String(dateStr).trim(); // Excel serial date number const serial = parseInt(d); if (!isNaN(serial) && serial > 40000 && serial < 60000) { const date = new Date((serial - 25569) * 86400 * 1000); return date.toISOString().split('T')[0]; } // DD.MM.YYYY or DD/MM/YYYY const parts = d.split(/[./-]/); if (parts.length === 3) { let day, month, year; if (parts[0].length === 4) { // YYYY-MM-DD year = parts[0]; month = parts[1]; day = parts[2]; } else { // DD.MM.YYYY day = parts[0]; month = parts[1]; year = parts[2]; } if (year.length === 2) year = '20' + year; return `${year}-${month.padStart(2, '0')}-${day.padStart(2, '0')}`; } return d; }