From 247874c47038f7f50d6522ce19514487fba2f269 Mon Sep 17 00:00:00 2001 From: host Date: Thu, 30 Jul 2026 17:14:06 +0300 Subject: [PATCH] =?UTF-8?q?=D0=A3=D0=B4=D0=B0=D0=BB=D0=B5=D0=BD=D0=B8?= =?UTF-8?q?=D0=B5=20=D1=84=D0=B0=D0=B9=D0=BB=D0=B0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- parser.js | 635 ------------------------------------------------------ 1 file changed, 635 deletions(-) delete mode 100644 parser.js diff --git a/parser.js b/parser.js deleted file mode 100644 index 924b0d1..0000000 --- a/parser.js +++ /dev/null @@ -1,635 +0,0 @@ -/** - * parser.js — Парсинг и сверка данных - * - * Определение типа файла, извлечение ведомостей и транзакций, - * сопоставление по водителям + гос. номерам, расчёт расхождений. - */ - -import { readExcel } from './utils.js'; - -/* ================================================================ - КОНСТАНТЫ - ================================================================ */ - -/** Заголовок, по которому распознаётся ведомость */ -const VEDOMOST_HEADER = 'ВЕДОМОСТЬ ПРИЕМА НАЛИЧНЫХ СРЕДСТВ'; - -/** Столбцы, по которым распознаётся выгрузка транзакций */ -const TRANSACTION_COLS = ['CONDUCTOR', 'DATE', 'TARIF_SUM']; - -/** Латинские буквы, которые заменяются на кириллические в гос. номерах */ -const LATIN_TO_CYRILLIC = { - 'A': 'А', 'B': 'В', 'C': 'С', 'E': 'Е', 'H': 'Н', - 'K': 'К', 'M': 'М', 'O': 'О', 'P': 'Р', 'T': 'Т', - 'X': 'Х', 'Y': 'У' -}; - -/** Порог схожести ФИО для нечёткого сравнения (0..1) */ -const FUZZY_THRESHOLD = 0.88; - -/* ================================================================ - ТИПИЗАЦИЯ ФАЙЛОВ - ================================================================ */ - -/** - * Определить тип файла: 'vedomost' | 'transaction' | null - * @param {string} filename - * @param {string[][]} firstSheet - первый лист книги - * @returns {string|null} - */ -export function detectFileType(filename, firstSheet) { - // 1. Проверка по заголовку ведомости - if (hasHeader(firstSheet, VEDOMOST_HEADER)) { - return 'vedomost'; - } - - // 2. Проверка по названию файла - const lower = filename.toLowerCase(); - if (lower.includes('transaction') || lower.includes('transactions') || lower.includes('выгрузка')) { - return 'transaction'; - } - - // 3. Проверка по столбцам транзакций (первые 3 строки) - if (hasTransactionColumns(firstSheet)) { - return 'transaction'; - } - - // 4. Если есть знакомые заголовки — ведомость - if (hasHeader(firstSheet, 'ВЕДОМОСТЬ')) { - return 'vedomost'; - } - - return null; -} - -/** - * Проверить, содержит ли лист указанный заголовок - */ -function hasHeader(sheetData, header) { - if (!sheetData || sheetData.length === 0) return false; - // Проверяем первые 20 строк - for (let i = 0; i < Math.min(sheetData.length, 20); i++) { - const row = sheetData[i]; - if (!row) continue; - const joined = row.filter(c => c != null).map(String).join(' ').toUpperCase(); - if (joined.includes(header.toUpperCase())) return true; - } - return false; -} - -/** - * Проверить, содержит ли лист столбцы транзакций - */ -function hasTransactionColumns(sheetData) { - if (!sheetData || sheetData.length === 0) return false; - // Проверяем первые 5 строк - for (let i = 0; i < Math.min(sheetData.length, 5); i++) { - const row = sheetData[i]; - if (!row) continue; - const cells = row.map(c => String(c).toUpperCase().trim()); - const hasConductor = cells.some(c => c.includes('CONDUCTOR')); - const hasDate = cells.some(c => c.includes('DATE')); - const hasTarif = cells.some(c => c.includes('TARIF_SUM') || c.includes('TARIF')); - if (hasConductor && hasDate && hasTarif) return true; - } - return false; -} - -/* ================================================================ - НОРМАЛИЗАЦИЯ ДАННЫХ - ================================================================ */ - -/** - * Нормализовать гос. номер: заменить латиницу на кириллицу, удалить лишнее - * @param {string} plate - * @returns {string} - */ -export function normalizePlate(plate) { - if (!plate) return ''; - let s = String(plate).toUpperCase().trim(); - // Замена латинских букв на кириллические - s = s.split('').map(ch => LATIN_TO_CYRILLIC[ch] || ch).join(''); - // Удаление пробелов, дефисов и непечатных символов, кроме букв и цифр - s = s.replace(/[^А-ЯЁA-Z0-9]/gi, ''); - return s; -} - -/** - * Нормализовать ФИО: верхний регистр, Ё→Е, обрезка - * @param {string} name - * @returns {string} - */ -export function normalizeName(name) { - if (!name) return ''; - return String(name) - .toUpperCase() - .trim() - .replace(/Ё/g, 'Е') - .replace(/\s+/g, ' '); -} - -/** - * Нечёткое сравнение ФИО (расстояние Левенштейна → схожесть) - * @param {string} a - * @param {string} b - * @returns {number} 0..1 - */ -export function nameSimilarity(a, b) { - if (!a || !b) return 0; - const s1 = normalizeName(a); - const s2 = normalizeName(b); - if (s1 === s2) return 1; - if (s1.length === 0 || s2.length === 0) return 0; - const dist = levenshtein(s1, s2); - const maxLen = Math.max(s1.length, s2.length); - return 1 - dist / maxLen; -} - -/** - * Расстояние Левенштейна - */ -function levenshtein(a, b) { - const m = a.length, n = b.length; - const dp = Array.from({ length: m + 1 }, () => Array(n + 1).fill(0)); - for (let i = 0; i <= m; i++) dp[i][0] = i; - for (let j = 0; j <= n; j++) dp[0][j] = j; - for (let i = 1; i <= m; i++) { - for (let j = 1; j <= n; j++) { - dp[i][j] = a[i - 1] === b[j - 1] - ? dp[i - 1][j - 1] - : 1 + Math.min(dp[i - 1][j], dp[i][j - 1], dp[i - 1][j - 1]); - } - } - return dp[m][n]; -} - -/* ================================================================ - ПАРСИНГ ВЕДОМОСТИ - ================================================================ */ - -/** - * Распарсить ведомость и вернуть структурированные данные - * @param {string[][]} sheetData - массив строк листа - * @returns {Object} - { docNumber, date, cashier, sections: { city, suburb }, drivers: [] } - */ -export function parseVedomost(sheetData) { - if (!sheetData || sheetData.length === 0) return null; - - const result = { - docNumber: '', - date: '', - cashier: '', - sections: { city: [], suburb: [] }, - drivers: [] - }; - - let currentSection = null; // 'city' | 'suburb' - let headerFound = false; - let tableStarted = false; - let tableEnded = false; - - for (let i = 0; i < sheetData.length; i++) { - const row = sheetData[i]; - if (!row || row.length === 0) continue; - - const cells = row.map(c => String(c).trim()); - - // Поиск заголовка ведомости - if (!headerFound) { - const joined = cells.join(' ').toUpperCase(); - if (joined.includes(VEDOMOST_HEADER)) { - headerFound = true; - } - continue; - } - - // Номер документа и дата - if (!result.docNumber) { - const numIdx = cells.findIndex(c => c.toUpperCase().includes('ДОКУМЕНТ')); - if (numIdx >= 0) { - result.docNumber = cells[numIdx + 1] || ''; - // Дата может быть в той же строке или следующей - const dateIdx = cells.findIndex(c => /^\d{2}[./-]\d{2}[./-]\d{4}$/.test(c)); - if (dateIdx >= 0) result.date = cells[dateIdx]; - } - } - - // Дата (если не нашли выше) - if (!result.date) { - const dateMatch = cells.find(c => /^\d{2}[./-]\d{2}[./-]\d{4}$/.test(c)); - if (dateMatch) result.date = dateMatch; - } - - // Кассир - if (!result.cashier) { - const cashIdx = cells.findIndex(c => c.toUpperCase().includes('КАССИР')); - if (cashIdx >= 0) { - // Ищем ФИО после слова "Кассир" - const nameIdx = cashIdx + 1; - if (nameIdx < cells.length && cells[nameIdx].length > 2) { - result.cashier = cells[nameIdx]; - } - } - } - - // Определение секции (Город / Пригород) - const joinedRow = cells.join(' ').toUpperCase(); - if (joinedRow.includes('ГОРОД') || joinedRow.includes('ГОР.') || (joinedRow.includes('ГОР') && !joinedRow.includes('ПРИГОР'))) { - currentSection = 'city'; - tableStarted = false; // Таблица начинается после заголовка секции - continue; - } - if (joinedRow.includes('ПРИГОРОД') || joinedRow.includes('ПРИГ.')) { - currentSection = 'suburb'; - tableStarted = false; - continue; - } - - // Пропуск пустых и служебных строк - if (cells.every(c => c === '')) { - if (tableStarted) tableEnded = true; // Пустая строка = конец таблицы - continue; - } - - // Определение начала таблицы с водителями - const headerRow = cells.join(' ').toUpperCase(); - if (headerRow.includes('ФИО') || headerRow.includes('Ф.И.О') || headerRow.includes('ВОДИТЕЛЬ') || - headerRow.includes('ФАМИЛИЯ') || (headerRow.includes('ФИО') && (headerRow.includes('СУММА') || headerRow.includes('МАРШРУТ')))) { - tableStarted = true; - tableEnded = false; - continue; - } - - // Если таблица началась, закончилась и мы снова видим непустую строку — возможно новая секция - if (tableEnded && !tableStarted) continue; - - // Парсинг строки водителя (только внутри таблицы) - if (tableStarted && !tableEnded && currentSection) { - const driver = parseDriverRow(cells, currentSection); - if (driver) { - result.drivers.push(driver); - result.sections[currentSection].push(driver); - } - } - } - - return result; -} - -/** - * Распарсить строку водителя из ведомости - * @param {string[]} cells - * @param {string} section - 'city' | 'suburb' - * @returns {Object|null} - */ -function parseDriverRow(cells, section) { - if (!cells || cells.length < 2) return null; - - // Фильтр: отбрасываем совсем пустые строки - const nonEmpty = cells.filter(c => c.trim() !== ''); - if (nonEmpty.length < 2) return null; - - // Пропускаем итоговые строки - const joined = cells.join(' ').toUpperCase(); - if (joined.includes('ИТОГО') || joined.includes('ВСЕГО') || joined.includes('ПО РАЗДЕЛУ')) { - return null; - } - - // Ищем ФИО — обычно первая колонка - const fio = cells[0] || ''; - if (fio.length < 3 || /^\d+$/.test(fio)) return null; - - // Ищем гос. номер — паттерн: буква + 3 цифры + 2 буквы + 2-3 цифры (регион) - let plate = ''; - let plateIdx = -1; - for (let j = 0; j < cells.length; j++) { - const cleaned = cells[j].replace(/[\s-]/g, '').toUpperCase(); - if (/^[А-ЯA-Z]{1}\d{3}[А-ЯA-Z]{2}\d{2,3}$/.test(cleaned) || - /^[А-ЯA-Z]{1}\d{3}[А-ЯA-Z]{2}$/.test(cleaned)) { - plate = cleaned; - plateIdx = j; - break; - } - } - - // Ищем маршрут (номер маршрута) - let route = ''; - for (let j = 1; j < cells.length; j++) { - const c = cells[j].trim(); - if (/^\d{1,3}$/.test(c) && c !== plate) { - route = c; - break; - } - } - - // Ищем сумму — последнее число или после ключевых слов - let sum = 0; - for (let j = cells.length - 1; j >= 0; j--) { - const c = cells[j].trim().replace(/\s/g, '').replace(',', '.'); - const num = parseFloat(c); - if (!isNaN(num) && num > 0) { - // Проверяем, что это не номер маршрута и не индекс - if (j !== plateIdx && j !== 0) { - sum = num; - break; - } - } - } - - return { - fio: normalizeName(fio), - plate: normalizePlate(plate), - section, - route, - sum, // "Сдано" — итого на сумму - cash: 0, // будет заполнено позже - nonCash: 0 - }; -} - -/* ================================================================ - ПАРСИНГ ТРАНЗАКЦИЙ - ================================================================ */ - -/** - * Распарсить выгрузку транзакций - * @param {string[][]} sheetData - * @returns {Object[]} - массив транзакций - */ -export function parseTransactions(sheetData) { - if (!sheetData || sheetData.length === 0) return []; - - // Найдём строку заголовков и индексы колонок - let headerRow = -1; - let colIndexes = {}; - - for (let i = 0; i < Math.min(sheetData.length, 10); i++) { - const row = sheetData[i]; - if (!row) continue; - const cells = row.map(c => String(c).toUpperCase().trim()); - const idxDate = cells.findIndex(c => c.includes('DATE')); - const idxConductor = cells.findIndex(c => c.includes('CONDUCTOR')); - const idxTarif = cells.findIndex(c => c.includes('TARIF_SUM') || (c.includes('TARIF') && !c.includes('TARIF_PAY'))); - const idxCar = cells.findIndex(c => c.includes('CAR') || c.includes('PLATE') || c.includes('AUTO')); - const idxRoute = cells.findIndex(c => c.includes('ROUTE') || c.includes('RUTE')); - const idxTarifPay = cells.findIndex(c => c.includes('TARIF_PAY') || c.includes('TARIFPAY')); - - if (idxDate >= 0 && idxConductor >= 0 && idxTarif >= 0) { - headerRow = i; - colIndexes = { - date: idxDate, - conductor: idxConductor, - tarifSum: idxTarif, - car: idxCar >= 0 ? idxCar : -1, - route: idxRoute >= 0 ? idxRoute : -1, - tarifPay: idxTarifPay >= 0 ? idxTarifPay : -1 - }; - break; - } - } - - if (headerRow < 0) return []; - - // Парсинг строк данных - const transactions = []; - for (let i = headerRow + 1; i < sheetData.length; i++) { - const row = sheetData[i]; - if (!row || row.length === 0) continue; - - const date = String(row[colIndexes.date] || '').trim(); - const conductor = String(row[colIndexes.conductor] || '').trim(); - const tarifSumStr = String(row[colIndexes.tarifSum] || '0').trim().replace(',', '.'); - const tarifSum = parseFloat(tarifSumStr); - - // Пропускаем пустые или невалидные строки - if (!date || !conductor || isNaN(tarifSum)) continue; - - // Проверка, что это строка данных, а не заголовок - if (conductor.toUpperCase().includes('CONDUCTOR')) continue; - - const car = colIndexes.car >= 0 ? String(row[colIndexes.car] || '').trim() : ''; - const route = colIndexes.route >= 0 ? String(row[colIndexes.route] || '').trim() : ''; - - transactions.push({ - date: date, - conductor: normalizeName(conductor), - plate: normalizePlate(car), - route: route, - tarifSum: tarifSum, // в копейках - tarifRub: tarifSum / 100, // в рублях - tarifPay: colIndexes.tarifPay >= 0 ? parseFloat(String(row[colIndexes.tarifPay] || '0').replace(',', '.')) : 0 - }); - } - - return transactions; -} - -/* ================================================================ - СВЕРКА - ================================================================ */ - -/** - * Выполнить сверку: сопоставить ведомости и транзакции - * @param {Object[]} vedomosti - массив распарсенных ведомостей - * @param {Object[]} transactions - массив транзакций - * @returns {Object} - результат сверки - */ -export function reconcile(vedomosti, transactions) { - // 1. Собираем уникальные даты из ведомостей - const vedomostDates = new Set(); - vedomosti.forEach(v => { - if (v.date) vedomostDates.add(normalizeDate(v.date)); - }); - - // 2. Фильтруем транзакции только по датам ведомостей - const filteredTransactions = transactions.filter(t => { - const tDate = normalizeDate(t.date); - return vedomostDates.has(tDate); - }); - - // 3. Группируем транзакции по (водитель + гос. номер) - const txByDriver = new Map(); - filteredTransactions.forEach(t => { - const key = `${t.conductor}|${t.plate || ''}`; - if (!txByDriver.has(key)) { - txByDriver.set(key, { conductor: t.conductor, plate: t.plate, totalTarif: 0, txCount: 0, dates: new Set() }); - } - const entry = txByDriver.get(key); - entry.totalTarif += t.tarifRub; - entry.txCount++; - entry.dates.add(normalizeDate(t.date)); - }); - - // 4. Собираем водителей из ведомостей и сопоставляем - const driverMap = new Map(); // ключ: нормализованное ФИО|номер - const allVedomostDrivers = []; - - vedomosti.forEach(v => { - v.drivers.forEach(d => { - const key = `${d.fio}|${d.plate}`; - if (!driverMap.has(key)) { - driverMap.set(key, { - fio: d.fio, - plate: d.plate, - route: d.route, - section: d.section, - cashier: v.cashier, - vedomostDate: v.date, - totalGiven: 0 // Сдано - }); - } - driverMap.get(key).totalGiven += d.sum; - allVedomostDrivers.push(d); - }); - }); - - // 5. Сопоставление: для каждого водителя из ведомости ищем транзакции - const results = []; - const unmatchedTx = []; // транзакции без пары в ведомости - - // Прямое сопоставление - const usedTxKeys = new Set(); - - driverMap.forEach((vd, key) => { - const [fio, plate] = key.split('|'); - - // Точное совпадение по (ФИО + номер) - const txKey = `${vd.fio}|${vd.plate}`; - let txEntry = txByDriver.get(txKey); - - // Нечёткое совпадение по ФИО, если точного нет - if (!txEntry) { - let bestSim = 0; - let bestKey = ''; - txByDriver.forEach((entry, k) => { - const [txFio, txPlate] = k.split('|'); - // Сначала проверяем совпадение номера - const plateMatch = txPlate && vd.plate && normalizePlate(txPlate) === normalizePlate(vd.plate); - const sim = nameSimilarity(vd.fio, txFio); - if (plateMatch && sim >= FUZZY_THRESHOLD && sim > bestSim) { - bestSim = sim; - bestKey = k; - } - }); - if (bestKey) { - txEntry = txByDriver.get(bestKey); - usedTxKeys.add(bestKey); - } - } else { - usedTxKeys.add(txKey); - } - - const totalCollected = txEntry ? txEntry.totalTarif : 0; - const diff = txEntry ? Math.round(totalCollected - vd.totalGiven) : -vd.totalGiven; - - results.push({ - driver: vd.fio, - car: vd.plate || '—', - route: vd.route || '—', - section: vd.section === 'city' ? 'Город' : 'Пригород', - cashier: vd.cashier || '—', - date: vd.vedomostDate || '—', - given: Math.round(vd.totalGiven), - collected: Math.round(totalCollected), - diff: diff, - txCount: txEntry ? txEntry.txCount : 0, - status: diff === 0 ? 'ok' : (Math.abs(diff) <= 10 ? 'warn' : 'err') - }); - }); - - // Не сопоставленные транзакции - txByDriver.forEach((entry, key) => { - if (!usedTxKeys.has(key)) { - unmatchedTx.push({ - driver: entry.conductor, - plate: entry.plate || '—', - totalCollected: Math.round(entry.totalTarif), - txCount: entry.txCount - }); - } - }); - - // 6. Агрегация по кассирам - const cashierMap = new Map(); - results.forEach(r => { - const key = r.cashier; - if (!cashierMap.has(key)) { - cashierMap.set(key, { cashier: key, count: 0, given: 0, collected: 0, diff: 0 }); - } - const c = cashierMap.get(key); - c.count++; - c.given += r.given; - c.collected += r.collected; - c.diff += r.diff; - }); - - // 7. Агрегация по маршрутам - const routeMap = new Map(); - results.forEach(r => { - const key = r.route || 'Без маршрута'; - if (!routeMap.has(key)) { - routeMap.set(key, { route: key, type: r.section, drivers: 0, sum: 0 }); - } - const rm = routeMap.get(key); - rm.drivers++; - rm.sum += r.given; - }); - - // 8. Выводы - const totalGiven = results.reduce((s, r) => s + r.given, 0); - const totalCollected = results.reduce((s, r) => s + r.collected, 0); - const totalDiff = results.reduce((s, r) => s + r.diff, 0); - const discrepancies = results.filter(r => r.diff !== 0); - - return { - drivers: results, - cashiers: Array.from(cashierMap.values()), - routes: Array.from(routeMap.values()), - unmatchedTx, - conclusions: { - totalDrivers: results.length, - totalGiven, - totalCollected, - totalDiff, - discrepanciesCount: discrepancies.length, - matchedPercent: results.length > 0 - ? Math.round((results.length - discrepancies.length) / results.length * 100) - : 0 - } - }; -} - -/* ================================================================ - ВСПОМОГАТЕЛЬНЫЕ ФУНКЦИИ - ================================================================ */ - -/** - * Нормализовать дату к формату YYYY-MM-DD - * @param {string} dateStr - * @returns {string} - */ -function normalizeDate(dateStr) { - if (!dateStr) return ''; - let d = String(dateStr).trim(); - // Excel serial date number - const serial = parseInt(d); - if (!isNaN(serial) && serial > 40000 && serial < 60000) { - const date = new Date((serial - 25569) * 86400 * 1000); - return date.toISOString().split('T')[0]; - } - // DD.MM.YYYY or DD/MM/YYYY - const parts = d.split(/[./-]/); - if (parts.length === 3) { - let day, month, year; - if (parts[0].length === 4) { - // YYYY-MM-DD - year = parts[0]; month = parts[1]; day = parts[2]; - } else { - // DD.MM.YYYY - day = parts[0]; month = parts[1]; year = parts[2]; - } - if (year.length === 2) year = '20' + year; - return `${year}-${month.padStart(2, '0')}-${day.padStart(2, '0')}`; - } - return d; -} \ No newline at end of file