Compare commits

..

No commits in common. "main" and "claude" have entirely different histories.
main ... claude

173 changed files with 41481 additions and 2774 deletions

3
.env Normal file
View File

@ -0,0 +1,3 @@
OPENAI_API_KEY = '"eyJhbGciOiJSUzUxMiIsInR5cCI6IkpXVCIsImtpZCI6IjFrYnhacFJNQGJSI0tSbE1xS1lqIn0.eyJ1c2VyIjoiaGI0NzQ0NCIsInR5cGUiOiJhcGlfa2V5IiwiYXBpX2tleV9pZCI6IjFmMTkxMThhLTY1N2UtNDI1Ni1hOTRhLTZjZDBmNzMwYjAzMCIsImlhdCI6MTc3NTA2MTI5Nn0.zEbSqgHTLKMt4So2eGUHTUmS4jYlB7iP1nxSpWAR177uaJfKOLJPq0E3uo0iHxldC04CClcIfx4vARJKLbPm7sR_0rB0LodTV2Qxw9k6bMm3ca77iNPFwNv518Xbo4_-BpSE8OPoP_ISjbyC4hljzBOJLaqXhEMGNymtoX7X9VfLtVFKjA76VurNnRLUxviCaGNMhqlzC9-RFb4XjTTX78JIty_2jQGhncvG3XiJe6pCIUtyp9e9N7uw0d5Mj9UcRQjqSfmeRlodquim4FY5skbmYPpUhjic6zoRG_R6dP3ya_GixGP5re4b9hjfyNEJh4vjibaqeizPQZ3J808I7vidty8D2UtxjrN05-NBRK7txy3ppgjPXwk1cZPtXxRacWrQ9rYE2Z0nOC2--v0oAKXNWS4erUtqXG862TwRM6HgfnsJj3y3XHvW7JVhAvHEL35cWfQ8YNsHyce9Ae28UhWoJifxIEcZRDePU2_fIVgcdR0S90tPhoTFkqBrUm-n'
RAG_MAX_CONTEXT_CHARS=80000
OPENAI_API_KEY = 'sk-9cfed30eb6894df5b9a3ffb4b1fb956d'

2
.gitignore vendored Normal file
View File

@ -0,0 +1,2 @@
/myenv/
/.myenv/

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View File

@ -0,0 +1 @@
{"version": 1, "answer": "{\n \"shipments\": [\n {\n \"ID_emails\": [\n \"00000000C632DB640058164AAF316BA35304963C0700E9A63B375101BC42B543D899C7BAFB5200000290329A0000E9A63B375101BC42B543D899C7BAFB5200000290507A0000\"\n ],\n \"client_name\": \"ИП Трофимов\",\n \"incoterms\": \"FOB\",\n \"cargo_ready_date\": \"25.06.26\",\n \"pickup_address\": \"Ningbo Proclean Co., Ltd., 188 Weiyi Road, Andong Industrial Zone, Qianwan New Area, Ningbo, Zhejiang Province, China\",\n \"cargo_value\": \"286000 CNY\",\n \"package_count\": \"\",\n \"total_weight_kg\": \"3330\",\n \"dimensions\": [],\n \"vehicle_dimensions\": [],\n \"total_volume_cbm\": \"25\",\n \"cargo_description\": \"Вертикальный пылесос\",\n \"hs_code\": \"\",\n \"dangerous_goods\": {\n \"batteries\": true,\n \"gases\": false,\n \"liquids\": false,\n \"dry_ice\": false,\n \"other_things\": []\n },\n \"battery_mode\": \"В составе\",\n \"stackable_with_others\": null,\n \"stackable_among_themselves\": null,\n \"msds_required\": true,\n \"batteries_packed_separately\": false,\n \"dgm_report_required\": true,\n \"brand_name\": \"\",\n \"brand_authorization_letter\": null,\n \"document_replacement_needed\": null,\n \"transshipment_with_third_country\": null,\n \"exporter_has_export_license\": null,\n \"additional_services\": [\n \"Таможенное оформление нашими силами по пути\"\n ],\n \"arrival_expediting_responsibility\": \"\",\n \"delivery_address\": \"Московская область, Долгопрудный, микрорайон Шереметьевский, Южная улица, 1с18\",\n \"special_transport_requirements\": \"\",\n \"shipment_type\": \"LCL\",\n \"container_type\": \"\",\n \"vehicle_type\": \"\",\n \"temperature_range\": \"\",\n \"customs_clearance_required\": true,\n \"customs_clearance_place_export_rf\": \"\",\n \"fumigation_on_wooden_packaging\": null,\n \"requested_shipping_type_names\": [\n \"LCL\",\n \"ЖД-сборка\"\n ],\n \"shipping_type\": \"\"\n }\n ]\n}", "structured_data": {"shipments": [{"ID_emails": ["00000000C632DB640058164AAF316BA35304963C0700E9A63B375101BC42B543D899C7BAFB5200000290329A0000E9A63B375101BC42B543D899C7BAFB5200000290507A0000"], "client_name": "ИП Трофимов", "incoterms": "FOB", "cargo_ready_date": "25.06.26", "pickup_address": "Ningbo Proclean Co., Ltd., 188 Weiyi Road, Andong Industrial Zone, Qianwan New Area, Ningbo, Zhejiang Province, China", "cargo_value": "286000 CNY", "package_count": "", "total_weight_kg": "3330", "dimensions": [], "vehicle_dimensions": [], "total_volume_cbm": "25", "cargo_description": "Вертикальный пылесос", "hs_code": "", "dangerous_goods": {"batteries": true, "gases": false, "liquids": false, "dry_ice": false, "other_things": []}, "battery_mode": "В составе", "stackable_with_others": null, "stackable_among_themselves": null, "msds_required": true, "batteries_packed_separately": false, "dgm_report_required": true, "brand_name": "", "brand_authorization_letter": null, "document_replacement_needed": null, "transshipment_with_third_country": null, "exporter_has_export_license": null, "additional_services": ["Таможенное оформление нашими силами по пути"], "arrival_expediting_responsibility": "", "delivery_address": "Московская область, Долгопрудный, микрорайон Шереметьевский, Южная улица, 1с18", "special_transport_requirements": "", "shipment_type": "LCL", "container_type": null, "vehicle_type": "", "temperature_range": "", "customs_clearance_required": null, "customs_clearance_place_export_rf": null, "fumigation_on_wooden_packaging": null, "requested_shipping_type_names": ["LCL", "ЖД-сборка"], "shipping_type": "", "documents_found": {"msds": [{"filename": "BVC-S70-3900_MSDS(2025新).pdf", "email_subject": "FW: ИП Трофимов - запрос на ЖД-сборку", "email_id": "00000000C632DB640058164AAF316BA35304963C0700E9A63B375101BC42B543D899C7BAFB5200000290329A0000E9A63B375101BC42B543D899C7BAFB5200000290507A0000"}], "dgm": [], "brand_authorization": []}, "shipping_options": [], "shipping_type_candidates": [], "shipment_direction": "unknown", "msds_status": "provided", "dgm_status": "not_provided"}], "preflight_classification": {"load_mode": "LCL", "multimodal_sea_rail": false, "shipping_type_suggestion": "", "shipment_type_if_container": "LCL", "indicative_all_modes": false}}, "total_emails_analyzed": 1}

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View File

@ -1,3 +0,0 @@
# NEKReport
Плагин для Microsoft Outlook позволяющий при помощи ии анализировать письма по грузоперевозкам и выдавать отчёт по возможным грузоперевозкам

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

169
addin/admin.html Normal file
View File

@ -0,0 +1,169 @@
<!DOCTYPE html>
<html lang="ru">
<head>
<meta charset="UTF-8">
<title>Управление типами перевозок</title>
<style>
body { font-family: Segoe UI, Arial, sans-serif; margin: 20px; background: #f5f5f5; }
.container { max-width: 1200px; margin: auto; background: white; padding: 20px; border-radius: 8px; box-shadow: 0 2px 10px rgba(0,0,0,0.1); }
h1 { margin-top: 0; color: #0078d4; }
.type-card { border: 1px solid #ccc; border-radius: 6px; padding: 15px; margin-bottom: 20px; background: #fafafa; }
.type-card h3 { margin-top: 0; cursor: pointer; color: #0078d4; }
.type-card .content { display: none; margin-top: 15px; }
.type-card.expanded .content { display: block; }
.field { margin-bottom: 10px; }
.field label { display: block; font-weight: bold; margin-bottom: 3px; }
.field input, .field textarea { width: 100%; padding: 8px; box-sizing: border-box; border: 1px solid #ccc; border-radius: 4px; }
.field textarea { min-height: 80px; }
.hint { color: #666; font-size: 12px; margin-top: 4px; }
.buttons { margin-top: 10px; }
button { padding: 8px 16px; margin-right: 10px; border: none; border-radius: 4px; cursor: pointer; font-weight: 500; }
.btn-primary { background: #0078d4; color: white; }
.btn-danger { background: #d32f2f; color: white; }
.btn-success { background: #2e7d32; color: white; }
#addBtn { margin-bottom: 20px; }
</style>
</head>
<body>
<div class="container">
<h1>⚙️ Управление типами перевозок</h1>
<button id="addBtn" class="btn-success"> Создать новый тип</button>
<div id="typesList"></div>
</div>
<script>
const API_BASE = '/llm';
async function loadTypes() {
const res = await fetch(API_BASE + '/shipping-types');
const types = await res.json();
renderTypes(types);
}
function renderTypes(types) {
const container = document.getElementById('typesList');
container.innerHTML = '';
types.forEach(type => {
const card = document.createElement('div');
card.className = 'type-card';
card.dataset.id = type.id;
const header = document.createElement('h3');
header.textContent = type.name || 'Новый тип';
header.addEventListener('click', () => card.classList.toggle('expanded'));
const contentDiv = document.createElement('div');
contentDiv.className = 'content';
contentDiv.innerHTML = `
<div class="field">
<label>Название типа</label>
<input type="text" class="name" value="${escapeHtml(type.name)}">
</div>
<div class="field">
<label>Email сотрудника (кому адресуют письма для этого типа)</label>
<input type="text" class="employee_email" value="${escapeHtml(type.employee_email || '')}">
<div class="hint">Можно несколько адресов через запятую, например: air@septem.pro, avia@septem.pro</div>
</div>
<div class="field">
<label>Критерии</label>
<textarea class="criteria">${escapeHtml(type.criteria || '')}</textarea>
</div>
<div class="field">
<label>Ключевые слова (через запятую)</label>
<input type="text" class="keywords" value="${escapeHtml((type.keywords || []).join(', '))}">
</div>
<div class="field">
<label>Шаблон письма-подтверждения</label>
<textarea class="confirmation">${escapeHtml(type.confirmation_template || '')}</textarea>
</div>
<div class="field">
<label>Шаблон письма-запроса информации</label>
<textarea class="info">${escapeHtml(type.info_request_template || '')}</textarea>
</div>
<div class="buttons">
<button class="btn-primary save">💾 Сохранить</button>
<button class="btn-danger delete">🗑️ Удалить</button>
</div>
`;
card.appendChild(header);
card.appendChild(contentDiv);
container.appendChild(card);
card.querySelector('.save').addEventListener('click', () => saveType(type.id, card));
card.querySelector('.delete').addEventListener('click', () => deleteType(type.id));
});
}
function escapeHtml(text) {
if (!text) return '';
return String(text).replace(/[&<>"]/g, function(m) {
if (m === '&') return '&amp;';
if (m === '<') return '&lt;';
if (m === '>') return '&gt;';
if (m === '"') return '&quot;';
return m;
});
}
async function saveType(id, card) {
const name = card.querySelector('.name').value.trim();
const employee_email = card.querySelector('.employee_email').value.trim();
const criteria = card.querySelector('.criteria').value;
const keywordsStr = card.querySelector('.keywords').value;
const keywords = keywordsStr.split(',').map(s => s.trim()).filter(s => s);
const confirmation_template = card.querySelector('.confirmation').value;
const info_request_template = card.querySelector('.info').value;
const data = { name, employee_email, criteria, keywords, confirmation_template, info_request_template };
const url = API_BASE + '/shipping-types/' + id;
const res = await fetch(url, {
method: 'PUT',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(data)
});
if (res.ok) {
card.querySelector('h3').textContent = name || 'Новый тип';
alert('Сохранено');
} else {
alert('Ошибка сохранения');
}
}
async function deleteType(id) {
if (!confirm('Удалить этот тип?')) return;
const res = await fetch(API_BASE + '/shipping-types/' + id, { method: 'DELETE' });
if (res.ok) {
loadTypes();
} else {
alert('Ошибка удаления');
}
}
document.getElementById('addBtn').addEventListener('click', async () => {
const newType = {
name: 'Новый тип',
employee_email: '',
criteria: '',
keywords: [],
confirmation_template: '',
info_request_template: ''
};
const res = await fetch(API_BASE + '/shipping-types', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(newType)
});
if (res.ok) {
loadTypes();
} else {
alert('Ошибка создания');
}
});
loadTypes();
</script>
</body>
</html>

285
addin/commands.html Normal file
View File

@ -0,0 +1,285 @@
<!DOCTYPE html>
<html>
<head>
<meta charset="UTF-8">
<script src="https://appsforoffice.microsoft.com/lib/1.1/hosted/office.js"></script>
</head>
<body>
<script>
Office.onReady(function () {
Office.actions.associate("analyzeSelectedEmails", analyzeSelectedEmails);
});
/* ----------------------------------------------------------
ВСТАВКА ПИСЬМА И ВЛОЖЕНИЙ В OUTLOOK
---------------------------------------------------------- */
window.addEventListener("message", async function(event) {
if (!event.data || event.data.type !== "insert_to_outlook") {
return;
}
try {
const payload = event.data.payload;
const letterText = payload.letter;
const sessionId = payload.session_id;
const item = Office.context.mailbox.item;
/* вставляем текст письма */
item.body.setAsync(
letterText,
{ coercionType: Office.CoercionType.Text }
);
/* получаем вложения */
const response = await fetch(
`https://nec.clients.septem.pro/llm/email-attachments/${sessionId}`
);
const attachments = await response.json();
for (const file of attachments) {
if (!file.content_base64) continue;
const ext = file.filename.split('.').pop().toLowerCase();
// полностью игнорируем изображения
if (['png','gif','bmp','tiff','webp'].includes(ext)) {
continue;
}
Office.context.mailbox.item.addFileAttachmentFromBase64Async(
file.content_base64,
file.filename
);
}
} catch (error) {
console.error("Insert to Outlook error:", error);
}
});
/* ----------------------------------------------------------
ПОЛУЧЕНИЕ ТЕКСТА И HTML ТЕЛА ПИСЬМА (ДОБАВЛЕНО)
---------------------------------------------------------- */
function getBodyAsync(item) {
return new Promise((resolve) => {
const result = { text: '', html: '' };
let pending = 2;
item.body.getAsync(Office.MailboxEnums.BodyType.Text, (res) => {
result.text = res.value || '';
if (--pending === 0) resolve(result);
});
item.body.getAsync(Office.MailboxEnums.BodyType.Html, (res) => {
result.html = res.value || '';
if (--pending === 0) resolve(result);
});
});
}
/* ----------------------------------------------------------
ПОЛУЧЕНИЕ ВЛОЖЕНИЙ
---------------------------------------------------------- */
function getAttachmentsWithContent(item) {
return new Promise((resolve) => {
if (!item.attachments || item.attachments.length === 0) {
resolve([]);
return;
}
const attachmentPromises = item.attachments.map(att => {
return new Promise((resolveAttachment) => {
item.getAttachmentContentAsync(att.id, function (result) {
if (result.status === Office.AsyncResultStatus.Succeeded) {
const attachmentContent = result.value;
let contentBase64 = null;
if (attachmentContent.format === Office.MailboxEnums.AttachmentContentFormat.Base64) {
contentBase64 = attachmentContent.content;
}
else if (attachmentContent.format === Office.MailboxEnums.AttachmentContentFormat.String) {
contentBase64 = btoa(attachmentContent.content);
}
resolveAttachment({
filename: att.name,
size: att.size,
content: contentBase64
});
}
else {
console.error("Failed to get attachment:", result.error);
resolveAttachment({
filename: att.name,
size: att.size,
content: null
});
}
});
});
});
Promise.all(attachmentPromises).then(resolve);
});
}
/* ----------------------------------------------------------
ОСНОВНАЯ ФУНКЦИЯ АНАЛИЗА ПИСЕМ
---------------------------------------------------------- */
async function analyzeSelectedEmails(event) {
try {
const item = Office.context.mailbox.item;
// Получаем и текст, и HTML (добавлено)
const body = await getBodyAsync(item);
const attachments = await getAttachmentsWithContent(item);
const emailData = {
id: item.itemId,
subject: item.subject,
sender: item.from?.emailAddress || "",
senderName: item.from?.displayName || "",
body: body.text, // текст
body_html: body.html, // HTML (новое поле)
receivedTime: item.dateTimeReceived?.toISOString() || "",
to: item.to?.map(t => t.emailAddress).join(", ") || "",
cc: item.cc?.map(c => c.emailAddress).join(", ") || "",
attachments: attachments
};
let sessionId = Office.context.roamingSettings.get("cargoSessionId");
const response = await fetch(
"https://nec.clients.septem.pro/llm/process-outlook-emails",
{
method: "POST",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({
emails: [emailData],
session_id: sessionId
})
}
);
if (!response.ok) {
throw new Error("Server error: " + response.status);
}
const result = await response.json();
const newSessionId = result.session_id;
Office.context.roamingSettings.set("cargoSessionId", newSessionId);
Office.context.roamingSettings.saveAsync(function (asyncResult) {
if (asyncResult.status === Office.AsyncResultStatus.Failed) {
console.error("Failed to save roamingSettings:", asyncResult.error);
}
});
Office.context.ui.displayDialogAsync(
`https://nec.clients.septem.pro/llm/addin/taskpane.html?session_id=${newSessionId}`,
{
height: 70,
width: 50,
displayInIframe: true
},
function (asyncResult) {
if (asyncResult.status === Office.AsyncResultStatus.Failed) {
console.error("Dialog failed:", asyncResult.error);
window.open(
`https://nec.clients.septem.pro?session_id=${newSessionId}`,
"_blank"
);
}
event.completed();
}
);
} catch (error) {
console.error("Unexpected error:", error);
event.completed();
}
}
</script>
</body>
</html>

164
addin/taskpane.html Normal file
View File

@ -0,0 +1,164 @@
<!DOCTYPE html>
<html>
<head>
<meta charset="UTF-8">
<meta http-equiv="Content-Security-Policy" content="connect-src https://nec.clients.septem.pro;">
<title>SEPTEM Cargo Analytics</title>
<script src="https://appsforoffice.microsoft.com/lib/1.1/hosted/office.js"></script>
<style>
body {
font-family: Segoe UI, sans-serif;
padding: 20px;
text-align: center;
background: #f7f7f7;
}
h3 {
margin-top: 0;
}
.status {
margin-top: 15px;
font-size: 14px;
}
.link {
word-break: break-all;
background: #e0e0e0;
padding: 10px;
border-radius: 6px;
margin-top: 10px;
}
.reset-btn {
margin-top: 25px;
padding: 12px 22px;
font-size: 15px;
background: #0078d4;
color: white;
border: none;
border-radius: 6px;
cursor: pointer;
transition: all 0.2s;
}
.reset-btn:hover {
background: #005fa3;
}
.reset-btn:disabled {
background: #888;
cursor: not-allowed;
}
</style>
</head>
<body>
<h3>🚚 SEPTEM Cargo Analytics</h3>
<div id="message"></div>
<div id="fallbackLink" class="link" style="display:none;"></div>
<button id="resetSession" class="reset-btn">
🔄 Начать новую сессию
</button>
<div id="resetStatus" class="status"></div>
<script>
const RESULT_URL = "https://nec.clients.septem.pro";
Office.onReady(function () {
const resetBtn = document.getElementById("resetSession");
const statusText = document.getElementById("resetStatus");
const urlParams = new URLSearchParams(window.location.search);
const sessionId = urlParams.get('session_id');
if (sessionId) {
const webUiUrl = `${RESULT_URL}?session_id=${sessionId}`;
document.getElementById('message').innerText =
"Анализ завершён. Открываю браузер...";
const newWindow = window.open(webUiUrl, '_blank');
if (!newWindow) {
document.getElementById('fallbackLink').innerHTML =
`Если браузер не открылся, <a href="${webUiUrl}" target="_blank">нажмите здесь</a>`;
document.getElementById('fallbackLink').style.display = 'block';
} else {
setTimeout(() => {
Office.context.ui.close();
}, 3000);
}
} else {
document.getElementById('message').innerText =
"Нет данных для анализа.";
}
/* ----------------------------
СБРОС СЕССИИ
---------------------------- */
resetBtn.addEventListener("click", function () {
resetBtn.disabled = true;
resetBtn.innerText = "Сбрасываю сессию...";
Office.context.roamingSettings.set("cargoSessionId", null);
Office.context.roamingSettings.saveAsync(function (result) {
if (result.status === Office.AsyncResultStatus.Succeeded) {
statusText.innerText =
"✅ Сессия сброшена. Можно начать новый анализ.";
resetBtn.innerText = "✔ Готово";
setTimeout(() => {
Office.context.ui.close();
}, 2000);
}
else {
statusText.innerText =
"❌ Ошибка сброса сессии";
resetBtn.disabled = false;
resetBtn.innerText = "🔄 Попробовать снова";
}
});
});
});
</script>
</body>
</html>

11048
cargo_rag_learning.json Normal file

File diff suppressed because one or more lines are too long

301
cargo_rag_v2.py Normal file
View File

@ -0,0 +1,301 @@
"""
Новый RAG-движок для анализа грузоперевозок.
Использует русскоязычные критерии из shipping_types.json.
Не хранит сессии, только проводит анализ.
"""
import json
import re
import logging
from pathlib import Path
from typing import List, Dict, Optional, Any
import openai
logger = logging.getLogger(__name__)
# Путь к файлу с типами перевозок (лежит рядом с этим файлом)
SHIPPING_TYPES_PATH = Path(__file__).resolve().parent / "shipping_types.json"
DEFAULT_MODEL = "deepseek-v4-flash" # или ваша модель
def load_shipping_types() -> List[Dict]:
"""Загружает список типов перевозок из JSONфайла."""
try:
with open(SHIPPING_TYPES_PATH, "r", encoding="utf-8") as f:
raw = json.load(f)
if not isinstance(raw, list):
raise ValueError("Ожидался список в shipping_types.json")
logger.info(f"Загружено {len(raw)} типов перевозок")
return raw
except Exception as e:
logger.error(f"Ошибка загрузки shipping_types.json: {e}")
return []
class CargoRAGEngineV2:
"""
Новый движок анализа грузоперевозок.
- Определяет тип перевозки по тексту писем.
- Заполняет критерии (на русском) с помощью LLM.
- Нормализует типы значений (числа, булевы).
"""
def __init__(self, openai_client: openai.OpenAI):
self.client = openai_client
self.shipping_types = load_shipping_types()
self._type_by_name = {
t.get("name", "").strip(): t for t in self.shipping_types
}
# ------------------------------------------------------------------
# Вспомогательные методы для совместимости со старыми эндпоинтами
# ------------------------------------------------------------------
def process_shipping_type_criteria(self, criteria_text: str) -> str:
"""Заглушка возвращает исходный текст, обработка не требуется."""
return criteria_text
def reload_shipping_types(self):
"""Перезагружает типы перевозок из файла."""
self.shipping_types = load_shipping_types()
self._type_by_name = {
t.get("name", "").strip(): t for t in self.shipping_types
}
def normalize_result(self, raw_data: Dict, shipping_type: Dict) -> Dict:
criteria_text = shipping_type.get("criteria", "")
criteria_list = self._criteria_text_to_list(criteria_text)
results = []
for idx, crit in enumerate(criteria_list, start=1):
raw_val = raw_data.get(crit)
normalized_val = self._normalize_value(raw_val)
results.append({
"number": idx, # ← добавили
"criterion": crit,
"value": normalized_val
})
return {
"shipping_type": shipping_type.get("name", ""),
"criteria_results": results
}
def record_cargo_learning(self, **kwargs) -> bool:
"""Заглушка для сохранения примеров (можно реализовать позже)."""
logger.info("record_cargo_learning вызван, но не реализован в V2")
return False
# ------------------------------------------------------------------
# Сборка текста писем
# ------------------------------------------------------------------
def _collect_emails_text(self, emails: List[Dict]) -> str:
parts = []
for mail in emails:
subject = mail.get("subject", "")
sender = mail.get("senderName", "") or mail.get("sender", "")
body = mail.get("body", "")
parts.append(f"Subject: {subject}\nFrom: {sender}\n\n{body}")
return "\n\n---\n\n".join(parts)
# ------------------------------------------------------------------
# Определение типа перевозки (первый проход)
# ------------------------------------------------------------------
def _detect_shipping_type(self, context_text: str, query_text: str = "") -> Optional[Dict]:
if not self.shipping_types:
return None
type_names = [t["name"] for t in self.shipping_types if t.get("name")]
names_list = "\n".join(f"- {name}" for name in type_names)
system = (
"Ты эксперт по логистике. Из списка типов перевозок выбери ОДИН, "
"наиболее подходящий для приведённых ниже писем. "
"Ответь ТОЛЬКО точным названием типа (как в списке), без пояснений."
)
user = (
f"Типы перевозок:\n{names_list}\n\n"
f"Запрос менеджера: {query_text}\n\n"
f"Письма:\n{context_text[:8000]}"
)
try:
resp = self.client.chat.completions.create(
model=DEFAULT_MODEL,
messages=[
{"role": "system", "content": system},
{"role": "user", "content": user}
],
temperature=0.0,
)
type_name = resp.choices[0].message.content.strip()
logger.info(f"LLM определил тип: '{type_name}'")
return self._type_by_name.get(type_name)
except Exception as e:
logger.error(f"Ошибка определения типа: {e}")
return None
# ------------------------------------------------------------------
# Разбор текста критериев в список ключей
# ------------------------------------------------------------------
@staticmethod
def _criteria_text_to_list(criteria_text: str) -> List[str]:
if not isinstance(criteria_text, str):
return []
lines = [ln.strip() for ln in criteria_text.splitlines() if ln.strip()]
items = []
for ln in lines:
parts = ln.split("\t", 1)
if len(parts) == 2:
text = parts[1].strip()
else:
text = re.sub(r"^\d+\.?\s*", "", ln).strip()
if text:
items.append(text)
# Убираем дубликаты, сохраняя порядок
seen = set()
unique = []
for it in items:
if it not in seen:
seen.add(it)
unique.append(it)
return unique
# ------------------------------------------------------------------
# Построение динамического JSON и запрос к LLM (второй проход)
# ------------------------------------------------------------------
def _extract_by_criteria(
self,
context_text: str,
shipping_type: Dict,
query_text: str = ""
) -> Dict[str, Any]:
criteria_text = shipping_type.get("criteria", "")
if not criteria_text:
logger.warning("У типа перевозки отсутствуют критерии")
return {}
criteria_list = self._criteria_text_to_list(criteria_text)
template = {c: None for c in criteria_list}
system = (
"Ты ассистент логиста. Твоя задача извлечь из деловой переписки информацию "
"строго по заданным критериям. Верни **только** JSON-объект, где ключи критерии "
"(на русском языке), а значения найденные данные.\n"
"Правила:\n"
"- Числа (вес, количество) возвращай как числа (не строки).\n"
"- Булевы ответы (да/нет) возвращай как true/false.\n"
"- Если по критерию ничего не найдено, оставь null.\n"
"- Не выдумывай данные, опирайся только на текст писем."
)
user = (
f"Тип перевозки: {shipping_type.get('name', '')}\n"
f"Доп. запрос менеджера: {query_text}\n\n"
f"Шаблон JSON (заполни его):\n{json.dumps(template, ensure_ascii=False, indent=2)}\n\n"
f"Текст писем:\n{context_text}"
)
try:
resp = self.client.chat.completions.create(
model=DEFAULT_MODEL,
messages=[
{"role": "system", "content": system},
{"role": "user", "content": user}
],
temperature=0.0,
)
answer = resp.choices[0].message.content
return self._parse_json_response(answer)
except Exception as e:
logger.error(f"Ошибка извлечения критериев: {e}")
return {}
def _parse_json_response(self, text: str) -> Dict:
text = re.sub(r"```json\s*", "", text)
text = re.sub(r"```", "", text)
text = text.strip()
try:
return json.loads(text)
except json.JSONDecodeError:
start = text.find("{")
end = text.rfind("}")
if start != -1 and end != -1:
try:
return json.loads(text[start:end+1])
except json.JSONDecodeError:
pass
logger.warning("Не удалось распарсить JSON-ответ LLM")
return {}
# ------------------------------------------------------------------
# Нормализация значений
# ------------------------------------------------------------------
@staticmethod
def _normalize_value(val: Any) -> Any:
if not isinstance(val, str):
return val
s = val.strip()
low = s.lower()
if low in ("да", "yes", "true", "есть", "присутствует"):
return True
if low in ("нет", "no", "false", "отсутствует"):
return False
numeric_part = re.sub(r"[^\d.,\-]", "", s).replace(",", ".")
if numeric_part:
try:
if "." in numeric_part:
return float(numeric_part)
else:
return int(numeric_part)
except ValueError:
pass
return s
def normalize_result(self, raw_data: Dict, shipping_type: Dict) -> Dict:
criteria_text = shipping_type.get("criteria", "")
criteria_list = self._criteria_text_to_list(criteria_text)
results = []
for crit in criteria_list:
raw_val = raw_data.get(crit)
normalized_val = self._normalize_value(raw_val)
results.append({
"criterion": crit,
"value": normalized_val
})
return {
"shipping_type": shipping_type.get("name", ""),
"criteria_results": results
}
# ------------------------------------------------------------------
# Основной метод анализа (вызывается из API)
# ------------------------------------------------------------------
def analyze(self, emails: List[Dict], query_text: str = "") -> Dict:
"""
Главный метод анализа писем.
Возвращает:
{
"shipping_type": "...",
"criteria_results": [{"criterion": ..., "value": ...}, ...],
"emails_count": int
}
или {"error": "..."} в случае ошибки.
"""
context = self._collect_emails_text(emails)
if not context.strip():
return {"error": "Нет текста для анализа"}
shipping_type = self._detect_shipping_type(context, query_text)
if not shipping_type:
return {"error": "Не удалось определить тип перевозки"}
raw_data = self._extract_by_criteria(context, shipping_type, query_text)
if not raw_data:
return {"error": "Не удалось извлечь данные по критериям"}
report = self.normalize_result(raw_data, shipping_type)
report["emails_count"] = len(emails)
return report

View File

@ -36,7 +36,7 @@ def iso_reference_prompt_block() -> str:
"Не выдумывай числа; при расчётах используй только эти значения.", "Не выдумывай числа; при расчётах используй только эти значения.",
"Ключи 20DV / 40DV / 40HC соответствуют формулировкам в критериях (проём 20DV, 40DV, 40HC).", "Ключи 20DV / 40DV / 40HC соответствуют формулировкам в критериях (проём 20DV, 40DV, 40HC).",
] ]
for key in ("20DV", "20HC", "20HC_PW", "40DV", "40HC", "40HC_PW", "45HC_PW_A", "45HC_PW_B"): for key in ("20DV", "20HC", "20HC_PW", "40DV", "40HC", "40DC", "40HC_PW", "45HC_PW_A", "45HC_PW_B"):
t = types.get(key) t = types.get(key)
if not isinstance(t, dict): if not isinstance(t, dict):
continue continue

View File

@ -12,74 +12,60 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "15",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки"
} }
}, },
{ {
"type": "pair", "type": "left_12",
"left": {
"num": "6", "num": "6",
"label": "Количество грузовых мест и вес, объем" "label": "Количество грузовых мест, габариты, вес, объём"
}, },
"right": { { "type": "left_12",
"num": "7", "num": "7",
"label": "Габариты грузовых мест ДхШхВ" "label": "Груз, код ТН ВЭД"
} },
{
"type": "left_12",
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "left_12",
"num": "9",
"label": "Штабелированиес другими| между собой"
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "10", "num": "10",
"label": "Штабелирвоание с другими отправками"
},
"right": {
"num": "11",
"label": "Штабелирвоание между собой"
}
},
{
"type": "pair",
"left": {
"num": "9",
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
},
"right": {
"num": "12",
"label": "Батарейки в составе груза / отдельно"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "14", "num": "11",
"label": "DGM" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "15", "num": "12",
"label": "Название бренда" "label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "16", "num": "14",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая" "label": "Замена документов"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "17", "num": "13",
"label": "Требуется ли дополнительный сервис - фитоконтроль, иное" "label": "Дополнительные сервисы (фитоконтроль и т.п.)"
}, },
"right": { "right": {
"num": "18", "num": "16",
"label": "Требуется ли замена документов" "label": "Место таможенного оформления в РФ"
} }
} }
] ]
@ -94,36 +80,35 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "16",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "6", "num": "6",
"label": "Количество и типоразмер машин" "label": "Количество ТС"
}, },
"right": { "right": {
"num": "7", "num": "7",
"label": "Количество грузовых мест и вес, объем" "label": "Вес груза и объём в машине"
} }
}, },
{ {
"type": "left_12", "type": "left_12",
"num": "8", "num": "8",
"label": "Габариты грузовых мест ДхШхВ" "label": "Габариты, вес и объём"
}, },
{ {
"type": "pair", "type": "left_12",
"left": { "num": "9",
"num": "10", "label": "Груз, код ТН ВЭД"
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
}, },
"right": { {
"num": "12", "type": "left_12",
"label": "Батарейки в составе груза / отдельно" "num": "10",
} "label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
}, },
{ {
"type": "pair", "type": "pair",
@ -132,30 +117,30 @@
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "13", "num": "12",
"label": "DGM" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "14", "num": "13",
"label": "Название бренда" "label": "Название бренда, Авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "15", "num": "15",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая" "label": "Требуется ли замена документов"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "16", "num": "14",
"label": "Требуется ли дополнительный сервис - фитоконтроль, иное" "label": "Дополнительные сервисы (фитоконтроль и т.п.)"
}, },
"right": { "right": {
"num": "17", "num": "17",
"label": "Требуется ли замена документов" "label": "Место таможенного оформления в РФ"
} }
} }
] ]
@ -170,74 +155,60 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "15",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки"
} }
}, },
{ {
"type": "pair", "type": "left_12",
"left": {
"num": "6", "num": "6",
"label": "Количество и типоразмер машин" "label": "Количество грузовых мест, габариты, вес, объём"
}, },
"right": { { "type": "left_12",
"num": "7", "num": "7",
"label": "Количество грузовых мест и вес, объем" "label": "Груз, код ТН ВЭД"
} },
{
"type": "left_12",
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "left_12",
"num": "9",
"label": "Штабелированиес другими| между собой"
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "10", "num": "10",
"label": "Штабелирвоание с другими отправками"
},
"right": {
"num": "11",
"label": "Штабелирвоание между собой"
}
},
{
"type": "pair",
"left": {
"num": "9",
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
},
"right": {
"num": "12",
"label": "Батарейки в составе груза / отдельно"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "14", "num": "11",
"label": "DGM" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "15", "num": "12",
"label": "Название бренда" "label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "16", "num": "14",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая" "label": "Замена документов"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "17", "num": "13",
"label": "Требуется ли дополнительный сервис - фитоконтроль, иное" "label": "Дополнительные сервисы (фитоконтроль и т.п.)"
}, },
"right": { "right": {
"num": "18", "num": "16",
"label": "Требуется ли замена документов" "label": "Место таможенного оформления в РФ"
} }
} }
] ]
@ -252,74 +223,60 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "15",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки"
} }
}, },
{ {
"type": "pair", "type": "left_12",
"left": {
"num": "6", "num": "6",
"label": "Количество и типоразмер машин" "label": "Количество грузовых мест, габариты, вес, объём"
}, },
"right": { { "type": "left_12",
"num": "7", "num": "7",
"label": "Количество грузовых мест и вес, объем" "label": "Груз, код ТН ВЭД"
} },
{
"type": "left_12",
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "left_12",
"num": "9",
"label": "Штабелированиес другими| между собой"
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "10", "num": "10",
"label": "Штабелирвоание с другими отправками"
},
"right": {
"num": "11",
"label": "Штабелирвоание между собой"
}
},
{
"type": "pair",
"left": {
"num": "9",
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
},
"right": {
"num": "12",
"label": "Батарейки в составе груза / отдельно"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "14", "num": "11",
"label": "DGM" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "15", "num": "12",
"label": "Название бренда" "label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "16", "num": "14",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая" "label": "Замена документов"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "17", "num": "13",
"label": "Требуется ли дополнительный сервис - фитоконтроль, иное" "label": "Дополнительные сервисы (фитоконтроль и т.п.)"
}, },
"right": { "right": {
"num": "18", "num": "16",
"label": "Требуется ли замена документов" "label": "Место таможенного оформления в РФ"
} }
} }
] ]
@ -334,74 +291,60 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "15",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки"
} }
}, },
{ {
"type": "pair", "type": "left_12",
"left": {
"num": "6", "num": "6",
"label": "Количество и типоразмер машин" "label": "Количество грузовых мест, габариты, вес, объём"
}, },
"right": { { "type": "left_12",
"num": "7", "num": "7",
"label": "Количество грузовых мест и вес, объем" "label": "Груз, код ТН ВЭД"
} },
{
"type": "left_12",
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "left_12",
"num": "9",
"label": "Штабелированиес другими| между собой"
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "10", "num": "10",
"label": "Штабелирвоание с другими отправками"
},
"right": {
"num": "11",
"label": "Штабелирвоание между собой"
}
},
{
"type": "pair",
"left": {
"num": "9",
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
},
"right": {
"num": "12",
"label": "Батарейки в составе груза / отдельно"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "14", "num": "11",
"label": "DGM" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "15", "num": "12",
"label": "Название бренда" "label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "16", "num": "14",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая" "label": "Замена документов"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "17", "num": "13",
"label": "Требуется ли дополнительный сервис - фитоконтроль, иное" "label": "Дополнительные сервисы (фитоконтроль и т.п.)"
}, },
"right": { "right": {
"num": "18", "num": "16",
"label": "Требуется ли замена документов" "label": "Место таможенного оформления в РФ"
} }
} }
] ]
@ -416,7 +359,7 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "16",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки, Место ТО в РФ"
} }
}, },
@ -428,24 +371,23 @@
}, },
"right": { "right": {
"num": "7", "num": "7",
"label": "Количество грузовых мест и вес, объем" "label": "Вес груза в контейнере и объём"
} }
}, },
{ {
"type": "left_12", "type": "left_12",
"num": "8", "num": "8",
"label": "Габариты грузовых мест ДхШхВ" "label": "Габариты, вес и объём"
}, },
{ {
"type": "pair", "type": "left_12",
"left": { "num": "9",
"num": "10", "label": "Груз, код ТН ВЭД"
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
}, },
"right": { {
"num": "12", "type": "left_12",
"label": "Батарейки в составе груза / отдельно" "num": "10",
} "label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
}, },
{ {
"type": "pair", "type": "pair",
@ -454,46 +396,40 @@
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "13", "num": "12",
"label": "Technical Description of Goods in Railway Transport" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "14", "num": "13",
"label": "Название бренда" "label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "15", "num": "15",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая"
}
},
{
"type": "pair",
"left": {
"num": "16",
"label": "Требуется ли дополнительный сервис - фитоконтроль, иное"
},
"right": {
"num": "17",
"label": "Требуется ли замена документов" "label": "Требуется ли замена документов"
} }
}, },
{
"type": "left_12",
"num": "14",
"label": "Дополнительные сервисы (крепление в контейнере, фитосанитарный контроль и т.п.)"
},
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "18", "num": "17",
"label": "Место таможенного оформления при экспорте из РФ" "label": "Место таможенного оформления при экспорте из РФ"
}, },
"right": { "right": {
"num": "19", "num": "18",
"label": "Наличие Фумигации на деревянной таре при экспорте из РФ" "label": "Фумигация деревянной тары при экспорте из РФ"
} }
}, },
{ {
"type": "left_12", "type": "left_12",
"num": "20", "num": "19",
"label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC" "label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC"
} }
] ]
@ -508,7 +444,7 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "16",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки, Место ТО в РФ"
} }
}, },
@ -520,24 +456,23 @@
}, },
"right": { "right": {
"num": "7", "num": "7",
"label": "Количество грузовых мест и вес, объем" "label": "Вес груза в контейнере и объём"
} }
}, },
{ {
"type": "left_12", "type": "left_12",
"num": "8", "num": "8",
"label": "Габариты грузовых мест ДхШхВ" "label": "Габариты, вес и объём"
}, },
{ {
"type": "pair", "type": "left_12",
"left": { "num": "9",
"num": "10", "label": "Груз, код ТН ВЭД"
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
}, },
"right": { {
"num": "12", "type": "left_12",
"label": "Батарейки в составе груза / отдельно" "num": "10",
} "label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
}, },
{ {
"type": "pair", "type": "pair",
@ -546,46 +481,40 @@
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "13", "num": "12",
"label": "Technical Description of Goods in Railway Transport" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "14", "num": "13",
"label": "Название бренда" "label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "15", "num": "15",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая"
}
},
{
"type": "pair",
"left": {
"num": "16",
"label": "Требуется ли дополнительный сервис - фитоконтроль, иное"
},
"right": {
"num": "17",
"label": "Требуется ли замена документов" "label": "Требуется ли замена документов"
} }
}, },
{
"type": "left_12",
"num": "14",
"label": "Дополнительные сервисы (крепление в контейнере, фитосанитарный контроль и т.п.)"
},
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "18", "num": "17",
"label": "Место таможенного оформления при экспорте из РФ" "label": "Место таможенного оформления при экспорте из РФ"
}, },
"right": { "right": {
"num": "19", "num": "18",
"label": "Наличие Фумигации на деревянной таре при экспорте из РФ" "label": "Фумигация деревянной тары при экспорте из РФ"
} }
}, },
{ {
"type": "left_12", "type": "left_12",
"num": "20", "num": "19",
"label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC" "label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC"
} }
] ]
@ -600,7 +529,7 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "16",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки, Место ТО в РФ"
} }
}, },
@ -612,24 +541,23 @@
}, },
"right": { "right": {
"num": "7", "num": "7",
"label": "Количество грузовых мест и вес, объем" "label": "Вес груза в контейнере и объём"
} }
}, },
{ {
"type": "left_12", "type": "left_12",
"num": "8", "num": "8",
"label": "Габариты грузовых мест ДхШхВ" "label": "Габариты, вес и объём"
}, },
{ {
"type": "pair", "type": "left_12",
"left": { "num": "9",
"num": "10", "label": "Груз, код ТН ВЭД"
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
}, },
"right": { {
"num": "12", "type": "left_12",
"label": "Батарейки в составе груза / отдельно" "num": "10",
} "label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
}, },
{ {
"type": "pair", "type": "pair",
@ -638,46 +566,40 @@
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "13", "num": "12",
"label": "Technical Description of Goods in Railway Transport" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "14", "num": "13",
"label": "Название бренда" "label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "15", "num": "15",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая"
}
},
{
"type": "pair",
"left": {
"num": "16",
"label": "Требуется ли дополнительный сервис - фитоконтроль, иное"
},
"right": {
"num": "17",
"label": "Требуется ли замена документов" "label": "Требуется ли замена документов"
} }
}, },
{
"type": "left_12",
"num": "14",
"label": "Дополнительные сервисы (крепление в контейнере, фитосанитарный контроль и т.п.)"
},
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "18", "num": "17",
"label": "Место таможенного оформления при экспорте из РФ" "label": "Место таможенного оформления при экспорте из РФ"
}, },
"right": { "right": {
"num": "19", "num": "18",
"label": "Наличие Фумигации на деревянной таре при экспорте из РФ" "label": "Фумигация деревянной тары при экспорте из РФ"
} }
}, },
{ {
"type": "left_12", "type": "left_12",
"num": "20", "num": "19",
"label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC" "label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC"
} }
] ]
@ -692,107 +614,84 @@
"label": "Страна, город, адрес забора" "label": "Страна, город, адрес забора"
}, },
"right": { "right": {
"num": "19", "num": "16",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ" "label": "Страна, город, точный адрес доставки"
} }
}, },
{
"type": "left_12",
"num": "6",
"label": "Количество грузовых мест,габариты, вес, объём"
},
{
"type":"left_12",
"num": "7",
"label": "Груз, код ТН ВЭД"
},
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "6", "num": "8",
"label": "Количество грузовых мест и вес, объем" "label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
}, },
"right": { "right": {
"num": "7", "num": "9",
"label": "Габариты грузовых мест ДхШхВ" "label": "Штабелирование с другими| между собой"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "10", "num": "10",
"label": "Штабелирвоание с другими отправками"
},
"right": {
"num": "11",
"label": "Штабелирвоание между собой"
}
},
{
"type": "pair",
"left": {
"num": "9",
"label": "Наличие батарейк, газов, жидкостей, аэрозолей"
},
"right": {
"num": "13",
"label": "Батарейки в составе груза / отдельно"
}
},
{
"type": "pair",
"left": {
"num": "12",
"label": "MSDS" "label": "MSDS"
}, },
"right": { "right": {
"num": "14", "num": "11",
"label": "DGM" "label": "DGM"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "15", "num": "12",
"label": "Название бренда" "label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
}, },
"right": { "right": {
"num": "16", "num": "13",
"label": "Авторизационное письмо/разрешение на вывоз этих брендов из Китая" "label": "Перелёт через третью страну со сменой авианакладных"
}
},
{
"type": "pair",
"left": {
"num": "14",
"label": "Экспортная лицензия у отправителя"
},
"right": {
"num": "15",
"label": "Дополнительные сервисы (фитоконтроль и т.п.)"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "17", "num": "17",
"label": "Перелет через третью страну со сменой авинаклданых" "label": "Вывоз из аэропорта — особые требования к автотранспорту"
}, },
"right": { "right": {
"num": "18", "num": "18",
"label": "Наличие экспортной лицезии у отправителя" "label": "Нужно ли таможенное оформление"
} }
}, },
{ {
"type": "pair", "type": "pair",
"left": { "left": {
"num": "19", "num": "19",
"label": "Дополнительный сервис"
},
"right": {
"num": "20",
"label": "Экспедирование в аэропорту назначения"
}
},
{
"type": "pair",
"left": {
"num": "22",
"label": "Спец требования при доставке до адреса в РФ"
},
"right": {
"num": "23",
"label": "Таможенное оформление"
}
},
{
"type": "pair",
"left": {
"num": "24",
"label": "Место таможенного оформления при экспорте из РФ" "label": "Место таможенного оформления при экспорте из РФ"
}, },
"right": { "right": {
"num": "25", "num": "20",
"label": "Наличие Фумигации на деревянной таре при экспорте из РФ" "label": "Фумигация деревянной тары при экспорте из РФ"
} }
} }
] ]

View File

@ -0,0 +1,700 @@
{
"version": 1,
"source_xlsx": "Запрос для ИИ Общий(3).xlsx",
"by_shipping_type": {
"Автомобильная перевозка (LTL)": {
"sheet": "Авто LTL интерфейс",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "15",
"label": "Страна, город, точный адрес доставки"
}
},
{
"type": "left_12",
"num": "6",
"label": "Количество грузовых мест, габариты, вес, объём"
},
{ "type": "left_12",
"num": "7",
"label": "Груз, код ТН ВЭД"
},
{
"type": "left_12",
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "left_12",
"num": "9",
"label": "Штабелированиес другими| между собой"
},
{
"type": "pair",
"left": {
"num": "10",
"label": "MSDS"
},
"right": {
"num": "11",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "12",
"label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "14",
"label": "Замена документов"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "Дополнительные сервисы (фитоконтроль и т.п.)"
},
"right": {
"num": "16",
"label": "Место таможенного оформления в РФ"
}
}
]
},
"Автомобильная перевозка (FTL)": {
"sheet": "Авто FTL интерфейс",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "16",
"label": "Страна, город, точный адрес доставки"
}
},
{
"type": "pair",
"left": {
"num": "6",
"label": "Количество ТС"
},
"right": {
"num": "7",
"label": "Вес груза и объём в машине"
}
},
{
"type": "left_12",
"num": "8",
"label": "Габариты, вес и объём"
},
{
"type": "left_12",
"num": "9",
"label": "Груз, код ТН ВЭД"
},
{
"type": "left_12",
"num": "10",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "pair",
"left": {
"num": "11",
"label": "MSDS"
},
"right": {
"num": "12",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "Название бренда, Авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "15",
"label": "Требуется ли замена документов"
}
},
{
"type": "pair",
"left": {
"num": "14",
"label": "Дополнительные сервисы (фитоконтроль и т.п.)"
},
"right": {
"num": "17",
"label": "Место таможенного оформления в РФ"
}
}
]
},
"Морская перевозка (LCL)": {
"sheet": "Море, море+жд, жд LCL интерфейc",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "15",
"label": "Страна, город, точный адрес доставки"
}
},
{
"type": "left_12",
"num": "6",
"label": "Количество грузовых мест, габариты, вес, объём"
},
{ "type": "left_12",
"num": "7",
"label": "Груз, код ТН ВЭД"
},
{
"type": "left_12",
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "left_12",
"num": "9",
"label": "Штабелированиес другими| между собой"
},
{
"type": "pair",
"left": {
"num": "10",
"label": "MSDS"
},
"right": {
"num": "11",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "12",
"label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "14",
"label": "Замена документов"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "Дополнительные сервисы (фитоконтроль и т.п.)"
},
"right": {
"num": "16",
"label": "Место таможенного оформления в РФ"
}
}
]
},
"Железнодорожная перевозка (LCL)": {
"sheet": "Море, море+жд, жд LCL интерфейc",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "15",
"label": "Страна, город, точный адрес доставки"
}
},
{
"type": "left_12",
"num": "6",
"label": "Количество грузовых мест, габариты, вес, объём"
},
{ "type": "left_12",
"num": "7",
"label": "Груз, код ТН ВЭД"
},
{
"type": "left_12",
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "left_12",
"num": "9",
"label": "Штабелированиес другими| между собой"
},
{
"type": "pair",
"left": {
"num": "10",
"label": "MSDS"
},
"right": {
"num": "11",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "12",
"label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "14",
"label": "Замена документов"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "Дополнительные сервисы (фитоконтроль и т.п.)"
},
"right": {
"num": "16",
"label": "Место таможенного оформления в РФ"
}
}
]
},
"Мультимодальная перевозка море + ж/д (LCL)": {
"sheet": "Море, море+жд, жд LCL интерфейc",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "15",
"label": "Страна, город, точный адрес доставки"
}
},
{
"type": "left_12",
"num": "6",
"label": "Количество грузовых мест, габариты, вес, объём"
},
{ "type": "left_12",
"num": "7",
"label": "Груз, код ТН ВЭД"
},
{
"type": "left_12",
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "left_12",
"num": "9",
"label": "Штабелированиес другими| между собой"
},
{
"type": "pair",
"left": {
"num": "10",
"label": "MSDS"
},
"right": {
"num": "11",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "12",
"label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "14",
"label": "Замена документов"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "Дополнительные сервисы (фитоконтроль и т.п.)"
},
"right": {
"num": "16",
"label": "Место таможенного оформления в РФ"
}
}
]
},
"Морская перевозка (FCL)": {
"sheet": "Море, море+жд, жд FCL интерфейc",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "16",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ"
}
},
{
"type": "pair",
"left": {
"num": "6",
"label": "Количество и типоразмер контейнеров"
},
"right": {
"num": "7",
"label": "Вес груза в контейнере и объём"
}
},
{
"type": "left_12",
"num": "8",
"label": "Габариты, вес и объём"
},
{
"type": "left_12",
"num": "9",
"label": "Груз, код ТН ВЭД"
},
{
"type": "left_12",
"num": "10",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "pair",
"left": {
"num": "11",
"label": "MSDS"
},
"right": {
"num": "12",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "15",
"label": "Требуется ли замена документов"
}
},
{
"type": "left_12",
"num": "14",
"label": "Дополнительные сервисы (крепление в контейнере, фитосанитарный контроль и т.п.)"
},
{
"type": "pair",
"left": {
"num": "17",
"label": "Место таможенного оформления при экспорте из РФ"
},
"right": {
"num": "18",
"label": "Фумигация деревянной тары при экспорте из РФ"
}
},
{
"type": "left_12",
"num": "19",
"label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC"
}
]
},
"Железнодорожная перевозка (FCL)": {
"sheet": "Море, море+жд, жд FCL интерфейc",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "16",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ"
}
},
{
"type": "pair",
"left": {
"num": "6",
"label": "Количество и типоразмер контейнеров"
},
"right": {
"num": "7",
"label": "Вес груза в контейнере и объём"
}
},
{
"type": "left_12",
"num": "8",
"label": "Габариты, вес и объём"
},
{
"type": "left_12",
"num": "9",
"label": "Груз, код ТН ВЭД"
},
{
"type": "left_12",
"num": "10",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "pair",
"left": {
"num": "11",
"label": "MSDS"
},
"right": {
"num": "12",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "15",
"label": "Требуется ли замена документов"
}
},
{
"type": "left_12",
"num": "14",
"label": "Дополнительные сервисы (крепление в контейнере, фитосанитарный контроль и т.п.)"
},
{
"type": "pair",
"left": {
"num": "17",
"label": "Место таможенного оформления при экспорте из РФ"
},
"right": {
"num": "18",
"label": "Фумигация деревянной тары при экспорте из РФ"
}
},
{
"type": "left_12",
"num": "19",
"label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC"
}
]
},
"Мультимодальная перевозка море + ж/д (FCL)": {
"sheet": "Море, море+жд, жд FCL интерфейc",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "16",
"label": "Страна, город, точный адрес доставки, Место ТО в РФ"
}
},
{
"type": "pair",
"left": {
"num": "6",
"label": "Количество и типоразмер контейнеров"
},
"right": {
"num": "7",
"label": "Вес груза в контейнере и объём"
}
},
{
"type": "left_12",
"num": "8",
"label": "Габариты, вес и объём"
},
{
"type": "left_12",
"num": "9",
"label": "Груз, код ТН ВЭД"
},
{
"type": "left_12",
"num": "10",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
{
"type": "pair",
"left": {
"num": "11",
"label": "MSDS"
},
"right": {
"num": "12",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "13",
"label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "15",
"label": "Требуется ли замена документов"
}
},
{
"type": "left_12",
"num": "14",
"label": "Дополнительные сервисы (крепление в контейнере, фитосанитарный контроль и т.п.)"
},
{
"type": "pair",
"left": {
"num": "17",
"label": "Место таможенного оформления при экспорте из РФ"
},
"right": {
"num": "18",
"label": "Фумигация деревянной тары при экспорте из РФ"
}
},
{
"type": "left_12",
"num": "19",
"label": "Превышение габаритов дверного проема контейнеров 20DV, 40DV, 40HC"
}
]
},
"Авиаперевозка": {
"sheet": "Авиа Интерфейс",
"blocks": [
{
"type": "pair",
"left": {
"num": "4",
"label": "Страна, город, адрес забора"
},
"right": {
"num": "16",
"label": "Страна, город, точный адрес доставки"
}
},
{
"type": "left_12",
"num": "6",
"label": "Количество грузовых мест,габариты, вес, объём"
},
{
"type":"left_12",
"num": "7",
"label": "Груз, код ТН ВЭД"
},
{
"type": "pair",
"left": {
"num": "8",
"label": "Батарейки, газы под давлением, жидкости| батарейки в составе или отдельно"
},
"right": {
"num": "9",
"label": "Штабелирование с другими| между собой"
}
},
{
"type": "pair",
"left": {
"num": "10",
"label": "MSDS"
},
"right": {
"num": "11",
"label": "DGM"
}
},
{
"type": "pair",
"left": {
"num": "12",
"label": "Бренд, авторизационное письмо/разрешение на вывоз бренда"
},
"right": {
"num": "13",
"label": "Перелёт через третью страну со сменой авианакладных"
}
},
{
"type": "pair",
"left": {
"num": "14",
"label": "Экспортная лицензия у отправителя"
},
"right": {
"num": "15",
"label": "Дополнительные сервисы (фитоконтроль и т.п.)"
}
},
{
"type": "pair",
"left": {
"num": "17",
"label": "Вывоз из аэропорта — особые требования к автотранспорту"
},
"right": {
"num": "18",
"label": "Нужно ли таможенное оформление"
}
},
{
"type": "pair",
"left": {
"num": "19",
"label": "Место таможенного оформления при экспорте из РФ"
},
"right": {
"num": "20",
"label": "Фумигация деревянной тары при экспорте из РФ"
}
}
]
}
}
}

View File

@ -1,154 +0,0 @@
from pypdf import PdfReader
from docx import Document
import openpyxl
from typing import Union
import io
import logging
from bs4 import BeautifulSoup
logger = logging.getLogger(__name__)
class DocumentProcessor:
@staticmethod
def _cell_to_text(v) -> str:
if v is None:
return ""
s = str(v).strip()
return s if s else ""
@staticmethod
def _sheet_rows_text_openpyxl(sheet) -> list[str]:
out: list[str] = []
for row in sheet.iter_rows(values_only=True):
vals = [DocumentProcessor._cell_to_text(c) for c in row]
vals = [x for x in vals if x]
if vals:
out.append(" | ".join(vals))
return out
@staticmethod
def _sheet_cols_text_openpyxl(sheet) -> list[str]:
out: list[str] = []
max_col = sheet.max_column or 0
max_row = sheet.max_row or 0
for c in range(1, max_col + 1):
vals: list[str] = []
for r in range(1, max_row + 1):
v = DocumentProcessor._cell_to_text(sheet.cell(r, c).value)
if v:
vals.append(v)
if vals:
col_letter = openpyxl.utils.get_column_letter(c)
out.append(f"COL {col_letter} TOP_DOWN: " + " || ".join(vals))
out.append(f"COL {col_letter} BOTTOM_UP: " + " || ".join(reversed(vals)))
return out
@staticmethod
def _sheet_rows_text_xlrd(sheet) -> list[str]:
out: list[str] = []
for row_idx in range(sheet.nrows):
row = sheet.row_values(row_idx)
vals = [DocumentProcessor._cell_to_text(c) for c in row]
vals = [x for x in vals if x]
if vals:
out.append(" | ".join(vals))
return out
@staticmethod
def _sheet_cols_text_xlrd(sheet) -> list[str]:
out: list[str] = []
for c in range(sheet.ncols):
vals: list[str] = []
for r in range(sheet.nrows):
v = DocumentProcessor._cell_to_text(sheet.cell_value(r, c))
if v:
vals.append(v)
if vals:
out.append(f"COL {c+1} TOP_DOWN: " + " || ".join(vals))
out.append(f"COL {c+1} BOTTOM_UP: " + " || ".join(reversed(vals)))
return out
def normalize_email_html(html_content: str) -> str:
"""
Очищает HTML письма и нормализует таблицы
"""
soup = BeautifulSoup(html_content, "html.parser")
# удаляем скрипты и стили
for tag in soup(["script", "style"]):
tag.decompose()
# нормализуем таблицы
for table in soup.find_all("table"):
table["style"] = "border-collapse: collapse; width: 100%;"
for cell in table.find_all(["td", "th"]):
cell["style"] = "border:1px solid #ccc;padding:6px;"
return str(soup)
def extract_text(self, content: bytes, filename: str) -> str:
ext = filename.lower().split('.')[-1]
# Изображения не обрабатываем
if ext in ['png', 'jpg', 'jpeg', 'gif', 'bmp', 'tiff', 'webp']:
return ""
if ext == 'pdf':
return self._extract_pdf(content)
elif ext in ['docx', 'doc']:
return self._extract_docx(content)
elif ext in ['xlsx', 'xls']:
return self._extract_excel(content)
elif ext == 'txt':
return content.decode('utf-8')
else:
raise ValueError(f"Unsupported format: {ext}")
def _extract_pdf(self, content: bytes) -> str:
pdf = PdfReader(io.BytesIO(content))
return "\n".join(page.extract_text() for page in pdf.pages)
def _extract_docx(self, content: bytes) -> str:
doc = Document(io.BytesIO(content))
return "\n".join(paragraph.text for paragraph in doc.paragraphs)
def _extract_excel(self, content: bytes) -> str:
try:
# сначала пробуем openpyxl (для .xlsx)
wb = openpyxl.load_workbook(io.BytesIO(content))
text = []
for sheet in wb.worksheets:
text.append(f"=== SHEET: {sheet.title} ===")
# Сохраняем совместимость: построчное чтение.
rows_text = self._sheet_rows_text_openpyxl(sheet)
if rows_text:
text.append("[ROWS]")
text.extend(rows_text)
# Новый режим: чтение по колонкам (общие данные часто внизу столбца).
cols_text = self._sheet_cols_text_openpyxl(sheet)
if cols_text:
text.append("[COLUMNS]")
text.extend(cols_text)
return "\n".join(text)
except Exception as e:
# если не получилось, возможно это .xls, пробуем xlrd
try:
import xlrd
workbook = xlrd.open_workbook(file_contents=content)
text = []
for sheet in workbook.sheets():
text.append(f"=== SHEET: {sheet.name} ===")
rows_text = self._sheet_rows_text_xlrd(sheet)
if rows_text:
text.append("[ROWS]")
text.extend(rows_text)
cols_text = self._sheet_cols_text_xlrd(sheet)
if cols_text:
text.append("[COLUMNS]")
text.extend(cols_text)
return "\n".join(text)
except ImportError:
logger.error("xlrd not installed, cannot parse .xls files")
return ""
except Exception as e2:
logger.error(f"Failed to parse Excel with xlrd: {e2}")
return ""

815
document_processor.py Normal file
View File

@ -0,0 +1,815 @@
import io
import os
import re
import csv
import tempfile
import subprocess
import logging
from typing import List
from bs4 import BeautifulSoup
from pypdf import PdfReader
from docx import Document
import openpyxl
import pdfplumber
from PIL import Image
import pytesseract
from striprtf.striprtf import rtf_to_text
logger = logging.getLogger(__name__)
class DocumentProcessor:
# =========================================================
# HELPERS
# =========================================================
@staticmethod
def _cleanup_text(text: str) -> str:
if not text:
return ""
text = text.replace("\xa0", " ")
text = text.replace("\t", " ")
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
@staticmethod
def _cell_to_text(v) -> str:
if v is None:
return ""
s = str(v).strip()
return s if s else ""
# =========================================================
# HTML
# =========================================================
@staticmethod
def normalize_email_html(html_content: str) -> str:
"""
Очищает HTML письма и нормализует таблицы
"""
soup = BeautifulSoup(
html_content,
"html.parser"
)
for tag in soup(["script", "style"]):
tag.decompose()
for table in soup.find_all("table"):
table["style"] = (
"border-collapse: collapse; width: 100%;"
)
for cell in table.find_all(["td", "th"]):
cell["style"] = (
"border:1px solid #ccc;padding:6px;"
)
return str(soup)
# =========================================================
# MAIN
# =========================================================
def extract_text(
self,
content: bytes,
filename: str
) -> str:
ext = filename.lower().split(".")[-1]
try:
# =================================================
# PDF
# =================================================
if ext == "pdf":
return self._extract_pdf(content)
# =================================================
# WORD
# =================================================
elif ext in ["docx", "docm"]:
return self._extract_docx(content)
elif ext == "doc":
return self._extract_doc(content)
elif ext == "rtf":
return self._extract_rtf(content)
# =================================================
# EXCEL
# =================================================
elif ext in ["xlsx", "xlsm", "xlsxm"]:
return self._extract_excel_openpyxl(content)
elif ext == "xls":
return self._extract_excel_xls(content)
elif ext == "csv":
return self._extract_csv(content)
# =================================================
# IMAGES / OCR
# =================================================
elif ext in [
"png",
"jpg",
"jpeg",
"gif",
"bmp",
"webp"
]:
return self._extract_image(content)
elif ext in ["tif", "tiff"]:
return self._extract_tiff(content)
# =================================================
# TEXT
# =================================================
elif ext == "txt":
try:
return content.decode("utf-8")
except Exception:
return content.decode(
"cp1251",
errors="ignore"
)
# =================================================
# UNKNOWN
# =================================================
logger.warning(
f"Unsupported attachment format: {ext}"
)
return ""
except Exception:
logger.exception(
f"Attachment processing failed: {filename}"
)
return ""
# =========================================================
# PDF
# =========================================================
def _extract_pdf(
self,
content: bytes
) -> str:
try:
parts = []
with pdfplumber.open(
io.BytesIO(content)
) as pdf:
for page_num, page in enumerate(
pdf.pages,
start=1
):
parts.append(
f"=== PAGE {page_num} ==="
)
# =====================================
# TEXT
# =====================================
try:
text = page.extract_text()
if text:
parts.append(text)
except Exception as e:
logger.warning(
f"PDF text extraction failed: {e}"
)
# =====================================
# TABLES
# =====================================
try:
tables = page.extract_tables()
for table_idx, table in enumerate(
tables,
start=1
):
parts.append(
f"=== TABLE {table_idx} ==="
)
for row in table:
if not row:
continue
cells = []
for cell in row:
if cell is None:
continue
txt = str(cell).strip()
txt = " ".join(
txt.split()
)
if txt:
cells.append(txt)
if cells:
parts.append(
" | ".join(cells)
)
except Exception as e:
logger.warning(
f"PDF table extraction failed: {e}"
)
text = "\n".join(parts)
return self._cleanup_text(text)
except Exception as e:
logger.error(f"PDF parse failed: {e}")
return ""
# =========================================================
# DOCX / DOCM
# =========================================================
def _extract_docx(
self,
content: bytes
) -> str:
try:
doc = Document(io.BytesIO(content))
parts = []
# =============================================
# PARAGRAPHS
# =============================================
for p in doc.paragraphs:
txt = p.text.strip()
if txt:
parts.append(txt)
# =============================================
# TABLES
# =============================================
for table_idx, table in enumerate(
doc.tables
):
parts.append(
f"=== TABLE {table_idx + 1} ==="
)
for row in table.rows:
cells = []
for cell in row.cells:
cell_parts = []
# =================================
# DIRECT CELL TEXT
# =================================
txt = "\n".join(
p.text.strip()
for p in cell.paragraphs
if p.text.strip()
)
txt = " ".join(txt.split())
if txt:
cell_parts.append(txt)
# =================================
# NESTED TABLES
# =================================
for nested_table in cell.tables:
for nested_row in nested_table.rows:
nested_vals = []
for nested_cell in nested_row.cells:
nested_txt = "\n".join(
p.text.strip()
for p in nested_cell.paragraphs
if p.text.strip()
)
nested_txt = " ".join(
nested_txt.split()
)
if nested_txt:
nested_vals.append(
nested_txt
)
if nested_vals:
cell_parts.append(
" || ".join(
nested_vals
)
)
# =================================
# FINAL CELL
# =================================
if cell_parts:
cells.append(
" <<<TABLE>>> ".join(
cell_parts
)
)
if cells:
parts.append(
" | ".join(cells)
)
text = "\n".join(parts)
return self._cleanup_text(text)
except Exception as e:
logger.error(f"DOCX parse failed: {e}")
return ""
# =========================================================
# DOC
# =========================================================
def _extract_doc(
self,
content: bytes
) -> str:
try:
with tempfile.TemporaryDirectory() as tmpdir:
input_path = os.path.join(
tmpdir,
"temp.doc"
)
with open(input_path, "wb") as f:
f.write(content)
subprocess.run(
[
"libreoffice",
"--headless",
"--convert-to",
"docx",
input_path,
"--outdir",
tmpdir,
],
check=True,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
)
converted_path = os.path.join(
tmpdir,
"temp.docx"
)
if not os.path.exists(
converted_path
):
logger.error(
"DOC conversion failed"
)
return ""
with open(
converted_path,
"rb"
) as f:
converted_content = f.read()
return self._extract_docx(
converted_content
)
except Exception as e:
logger.error(f"DOC parse failed: {e}")
return ""
# =========================================================
# RTF
# =========================================================
def _extract_rtf(
self,
content: bytes
) -> str:
try:
try:
text = content.decode("utf-8")
except Exception:
text = content.decode(
"cp1251",
errors="ignore"
)
text = rtf_to_text(text)
return self._cleanup_text(text)
except Exception as e:
logger.error(f"RTF parse failed: {e}")
return ""
# =========================================================
# XLSX
# =========================================================
def _extract_excel_openpyxl(
self,
content: bytes
) -> str:
try:
wb = openpyxl.load_workbook(
io.BytesIO(content),
data_only=True
)
text = []
for sheet in wb.worksheets:
text.append(
f"=== SHEET: {sheet.title} ==="
)
# =====================================
# MERGED CELLS
# =====================================
merged_ranges = list(
sheet.merged_cells.ranges
)
if merged_ranges:
text.append(
"=== MERGED CELLS ==="
)
for r in merged_ranges:
text.append(str(r))
# =====================================
# ROWS
# =====================================
text.append("[ROWS]")
for row in sheet.iter_rows(
values_only=True
):
vals = [
self._cell_to_text(c)
for c in row
]
vals = [x for x in vals if x]
if vals:
text.append(
" | ".join(vals)
)
# =====================================
# COLUMNS
# =====================================
text.append("[COLUMNS]")
max_col = sheet.max_column or 0
max_row = sheet.max_row or 0
for c in range(1, max_col + 1):
vals = []
for r in range(1, max_row + 1):
v = self._cell_to_text(
sheet.cell(r, c).value
)
if v:
vals.append(v)
if vals:
col_letter = (
openpyxl.utils
.get_column_letter(c)
)
text.append(
f"COL {col_letter}: "
+ " || ".join(vals)
)
return self._cleanup_text(
"\n".join(text)
)
except Exception as e:
logger.error(
f"XLSX parse failed: {e}"
)
return ""
# =========================================================
# XLS
# =========================================================
def _extract_excel_xls(
self,
content: bytes
) -> str:
try:
import xlrd
workbook = xlrd.open_workbook(
file_contents=content
)
text = []
for sheet in workbook.sheets():
text.append(
f"=== SHEET: {sheet.name} ==="
)
for row_idx in range(
sheet.nrows
):
row = sheet.row_values(
row_idx
)
vals = [
self._cell_to_text(c)
for c in row
]
vals = [x for x in vals if x]
if vals:
text.append(
" | ".join(vals)
)
return self._cleanup_text(
"\n".join(text)
)
except Exception as e:
logger.error(f"XLS parse failed: {e}")
return ""
# =========================================================
# CSV
# =========================================================
def _extract_csv(
self,
content: bytes
) -> str:
try:
try:
text = content.decode("utf-8")
except Exception:
text = content.decode(
"cp1251",
errors="ignore"
)
reader = csv.reader(
io.StringIO(text)
)
rows = []
for row in reader:
vals = [
str(x).strip()
for x in row
if str(x).strip()
]
if vals:
rows.append(
" | ".join(vals)
)
return self._cleanup_text(
"\n".join(rows)
)
except Exception as e:
logger.error(f"CSV parse failed: {e}")
return ""
# =========================================================
# OCR
# =========================================================
def _ocr_image(
self,
image: Image.Image
) -> str:
try:
text = pytesseract.image_to_string(
image,
lang="rus+eng",
config="--psm 6"
)
return self._cleanup_text(text)
except Exception as e:
logger.error(f"OCR failed: {e}")
return ""
# =========================================================
# IMAGE
# =========================================================
def _extract_image(
self,
content: bytes
) -> str:
try:
image = Image.open(
io.BytesIO(content)
)
return self._ocr_image(image)
except Exception as e:
logger.error(
f"Image parse failed: {e}"
)
return ""
# =========================================================
# TIFF
# =========================================================
def _extract_tiff(
self,
content: bytes
) -> str:
try:
image = Image.open(
io.BytesIO(content)
)
texts = []
for frame in range(
getattr(image, "n_frames", 1)
):
image.seek(frame)
texts.append(
self._ocr_image(image)
)
return self._cleanup_text(
"\n".join(texts)
)
except Exception as e:
logger.error(
f"TIFF parse failed: {e}"
)
return ""

285
email_cleaner.py Normal file
View File

@ -0,0 +1,285 @@
# email_cleaner.py
import re
from typing import List, Optional, Set
# ------------------------------------------------------------
# Шаблоны конфиденциальности (русские + английские)
# ------------------------------------------------------------
DISCLAIMER_PHRASES = [
r"IMPORTANT NOTICE:\s*This e?mail.+?intended\s+only.+",
r"This e?mail (?:message )?is confidential.+?intended\s+only\s+for.+?(?=\n\s*\n|\Z)",
r"The information transmitted is intended only for the person or entity to which it is addressed",
r"If you have received this email in error, please notify the sender immediately",
r"This message contains confidential information and is intended only for the individual named",
r"Please consider the environment before printing this email",
r"This email has been scanned by the .*? antivirus",
r"Warning: Although .*? has taken reasonable precautions",
r"The content of this email is confidential and intended for the recipient specified in message only\..*",
r"The content of this email is confidential and intended for the recipient specified in message only. It is strictly forbidden to share any part of this message with any third party, without a written consent of the sender. If you received this message by mistake, please reply to this message and follow with its deletion, so that we can ensure such a mistake does not occur in the future.",
# Русские варианты
r"Содержание этого письма конфиденциально и предназначено только для указанного получателя.*",
r"Данное сообщение может содержать конфиденциальную информацию.*",
r"Настоящее сообщение содержит конфиденциальную информацию.*",
r"Это сообщение содержит информацию, являющуюся коммерческой тайной.*",
r"Сообщение предназначено исключительно для указанного адресата.*",
r"Если вы не являетесь указанным получателем, пожалуйста, удалите это сообщение.*",
r"Любое распространение, копирование или использование этого сообщения.*",
r"Это письмо содержит конфиденциальную информацию.*",
# Автоответы
r"\bAUTO-REPLY\b|automatic reply|out of office",
r"Автоматическое уведомление",
]
# ------------------------------------------------------------
# Строки-разделители подписей
# ------------------------------------------------------------
SIGNATURE_STARTERS = [
r"^\s*--\s*$", # стандартный разделитель подписи
r"^\s*Thanks\s*[,.]?\s*$",
r"^\s*Thank you\s*[,.]?\s*$",
r"^\s*С уважением\s*[,.]?\s*$",
r"^\s*Best regards\s*[,.]?\s*$",
r"^\s*Kind regards\s*[,.]?\s*$",
r"^\s*Regards\s*[,.]?\s*$",
r"^\s*Здравствуйте\s*$",
r"^\s*Привет\s*$",
r"^\s*Добрый день\s*$",
r"^\s*Sincerely\s*[,.]?\s*$",
r"^\s*С наилучшими пожеланиями\s*$",
r"^\s*Спасибо\s*$",
r"^\s*Благодарю\s*$",
r"^\s*С ув\.\s*$",
r"^\s*С уважением и надеждой на сотрудничество\s*$",
]
# ------------------------------------------------------------
# Разделители цепочек пересылки (НЕ трогаем From/To/Subject внутри)
# ------------------------------------------------------------
FORWARD_SEPARATORS = [
r"^-{5,}\s*(Forwarded message|Пересланное сообщение|Пересылаемое сообщение|Переадресованное сообщение)\s*-{5,}",
r"^Begin forwarded message:",
r"^Начало переадресованного сообщения:",
r"^--- Исходное сообщение ---",
r"^-{3,}\s*Original Message\s*-{3,}",
]
# ------------------------------------------------------------
# Ключевые слова, предотвращающие удаление подписи/контактов
# ------------------------------------------------------------
TRANSPORT_KEYWORDS = [
"груз", "паллет", "вес", "адрес", "инвойс", "забор", "доставка",
"тн вэд", "контейнер", "отправка", "перевозка", "машина", "фура",
"ставка", "тариф", "расписание", "договор", "контракт", "заказ",
"invoice", "tracking", "delivery", "shipment", "cargo",
]
# ------------------------------------------------------------
# Основной класс очистки
# ------------------------------------------------------------
class EmailContextCleaner:
def __init__(
self,
remove_disclaimers: bool = True,
remove_signatures: bool = True,
remove_forward_chain: bool = True,
remove_contact_blocks: bool = True,
collapse_repeats: bool = True,
remove_empty_lines: bool = True,
remove_quote_lines: bool = True,
short_lines_only: int = 0,
max_url_length: int = 60,
remove_timestamps: bool = True,
custom_phrases: Optional[List[str]] = None,
):
self.remove_disclaimers = remove_disclaimers
self.remove_signatures = remove_signatures
self.remove_forward_chain = remove_forward_chain
self.remove_contact_blocks = remove_contact_blocks
self.collapse_repeats = collapse_repeats
self.remove_empty_lines = remove_empty_lines
self.remove_quote_lines = remove_quote_lines
self.short_lines_only = short_lines_only
self.max_url_length = max_url_length
self.remove_timestamps = remove_timestamps
# Компиляция шаблонов
self._disclaimer_patterns = [
re.compile(p, re.IGNORECASE | re.DOTALL | re.MULTILINE)
for p in DISCLAIMER_PHRASES
]
if custom_phrases:
self._disclaimer_patterns += [
re.compile(p, re.IGNORECASE | re.DOTALL) for p in custom_phrases
]
self._signature_starters = [re.compile(p) for p in SIGNATURE_STARTERS]
self._forward_separators = [
re.compile(p, re.IGNORECASE | re.MULTILINE) for p in FORWARD_SEPARATORS
]
def clean(self, text: str) -> str:
if not text:
return text
# 1. Дисклеймеры
if self.remove_disclaimers:
for pat in self._disclaimer_patterns:
text = pat.sub(" ", text)
# 2. Цепочки пересылок (оставить только первую часть)
if self.remove_forward_chain:
min_pos = len(text)
for pat in self._forward_separators:
match = pat.search(text)
if match:
pos = match.start()
if pos > 0 and pos < min_pos:
min_pos = pos
if min_pos > 0 and min_pos < len(text):
before = text[:min_pos]
# Удаляем хвост, только если до разделителя было осмысленного текста > 50 символов
meaningful = re.sub(r"\s+", "", before)
if len(meaningful) > 50:
text = before.strip()
# 3. Удалить строки подписей (после "Best regards" и т.п.)
if self.remove_signatures:
lines = text.splitlines()
result_lines = []
for i, line in enumerate(lines):
stripped = line.strip()
if any(pat.match(stripped) for pat in self._signature_starters):
remaining = lines[i+1:]
# Проверяем, что дальше не идёт важная информация
is_signature = all(
len(l.strip()) < 80 and not any(
kw in l.lower() for kw in TRANSPORT_KEYWORDS
)
for l in remaining if l.strip()
)
if is_signature:
break
result_lines.append(line)
text = "\n".join(result_lines)
# 4. Удалить строки цитирования
if self.remove_quote_lines:
text = re.sub(r"^(>\s*)+.*$", "", text, flags=re.MULTILINE)
# 5. Удалить очень короткие строки
if self.short_lines_only > 0:
text = "\n".join(
line for line in text.splitlines()
if len(line.strip()) >= self.short_lines_only
)
# 6. Обработка URL
if self.max_url_length > 0 or self.max_url_length == 0:
url_pattern = re.compile(r"https?://\S+|ftp://\S+")
def replace_url(m):
url = m.group(0)
if self.max_url_length == 0:
return ""
if len(url) > self.max_url_length:
short = url[:self.max_url_length] + "..."
return short
return url
text = url_pattern.sub(replace_url, text)
# 7. Временные метки
if self.remove_timestamps:
text = re.sub(
r"\b\d{1,2}[./-]\d{1,2}[./-]\d{2,4}\s+\d{1,2}:\d{2}(?::\d{2})?\s*(?:AM|PM)?\s*(?:GMT[+-]\d+)?\b",
"",
text,
flags=re.IGNORECASE,
)
# 8. Пустые строки
if self.remove_empty_lines:
text = re.sub(r"\n{3,}", "\n\n", text)
# 9. Дублирующиеся абзацы
if self.collapse_repeats:
text = self._deduplicate_paragraphs(text)
# 10. Блоки контактов (подписи без явного приветствия)
if self.remove_contact_blocks:
text = self._remove_contact_blocks(text)
# Финальная обрезка пробелов в строках
text = "\n".join(line.strip() for line in text.splitlines())
return text.strip()
def _deduplicate_paragraphs(self, text: str) -> str:
parts = re.split(r"(\n\s*\n)", text)
seen: Set[str] = set()
out = []
for i, part in enumerate(parts):
if i % 2 == 1:
out.append(part)
continue
key = part.strip()
if len(key) < 50:
out.append(part)
continue
if key in seen:
continue
seen.add(key)
out.append(part)
return "".join(out)
def _remove_contact_blocks(self, text: str) -> str:
"""
Удаляет блоки, похожие на подпись: Имя Фамилия, должность, телефоны, email, сайт,
которые идут после пустой строки и не содержат ключевых слов перевозки.
"""
lines = text.splitlines()
out = []
i = 0
while i < len(lines):
line = lines[i].strip()
if self._is_contact_line(line):
# Проверим, что предыдущая строка пустая или это начало текста
if i == 0 or (i > 0 and lines[i-1].strip() == ""):
# Собираем блок, пока строки похожи на контакты или короткие и без ключевых слов
j = i
while j < len(lines):
l = lines[j].strip()
if l == "" or self._is_contact_line(l) or (len(l) < 60 and not any(
kw in l.lower() for kw in TRANSPORT_KEYWORDS
)):
j += 1
else:
break
block_text = "\n".join(lines[i:j])
# Не удаляем, если есть транспортные ключевые слова
if not any(kw in block_text.lower() for kw in TRANSPORT_KEYWORDS):
i = j
continue
out.append(line)
i += 1
return "\n".join(out)
def _is_contact_line(self, line: str) -> bool:
"""Определяет, является ли строка элементом контактной информации."""
if re.search(r"\+?\d[\d\s\(\)\-]{6,}\d", line): # телефон
return True
if re.search(r"\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b", line): # email
return True
if re.search(r"www\.", line): # сайт
return True
if re.search(r"^\s*[A-Z][a-z]+\s+[A-Z][a-z]+\s*$", line): # Имя Фамилия (упрощённо)
return True
if re.search(r"Business Development Manager|COO|Operations dept\.|Leading Sales", line, re.IGNORECASE):
return True
return False
# ------------------------------------------------------------
# Быстрая функция для совместимости
# ------------------------------------------------------------
_default_cleaner = EmailContextCleaner()
def clean_email_context(text: str) -> str:
"""Очищает текст письма от мусора перед отправкой в LLM."""
return _default_cleaner.clean(text)

232
email_cleaner.py.save Normal file
View File

@ -0,0 +1,232 @@
import re
from typing import List, Optional, Set
# ------------------------------------------------------------
# Шаблоны для удаления (можно вынести в конфиг или JSON)
# ------------------------------------------------------------
# Фразы конфиденциальности / автоответы / экологические призывы (английские + русские)
DISCLAIMER_PHRASES = [
r"IMPORTANT NOTICE:\s*This e?mail.+?intended\s+only.+",
r"This e?mail (?:message )?is confidential.+?intended\s+only\s+for.+?(?=\n\s*\n|\Z)",
r"The information transmitted is intended only for the person or entity to which it is addressed",
r"If you have received this email in error, please notify the sender immediately",
r"This message contains confidential information and is intended only for the individual named",
r"Please consider the environment before printing this email",
r"This email has been scanned by the .*? antivirus",
r"Warning: Although .*? has taken reasonable precautions",
r"Sent from my iPhone",
r"Sent from my iPad",
r"Sent from my Android device",
r"Отправлено из приложения Яндекс.Почты",
r"Отправлено с iPhone",
r"Это письмо содержит конфиденциальную информацию",
r"Если вы получили это письмо по ошибке, пожалуйста, сообщите отправителю",
r"Данное сообщение может содержать сведения, составляющие коммерческую тайну",
r"Подумайте о природе, прежде чем распечатывать это письмо",
r"Автоматическое уведомление",
r"AUTO-REPLY|automatic reply|out of office",
]
# Шаблоны подписей (начало)
SIGNATURE_STARTERS = [
r"^\s*--\s*$", # стандартный разделитель подписи
r"^\s*Best regards\s*$",
r"^\s*Kind regards\s*$",
r"^\s*Sincerely\s*$",
r"^\s*Thanks\s*$",
r"^\s*Regards\s*$",
r"^\s*С уважением\s*$",
r"^\s*С наилучшими пожеланиями\s*$",
r"^\s*Спасибо\s*$",
r"^\s*Благодарю\s*$",
r"^\s*С ув.\s*$",
]
# Строки-разделители пересланных сообщений (нужно оставить только первое или удалить всё после)
FORWARD_SEPARATORS = [
r"^-{5,} Forwarded message -{5,}",
r"^-{5,} Пересылаемое сообщение -{5,}",
r"^-{5,} Пересланное сообщение -{5,}",
r"^From:.*?\nSent:.*?\nTo:.*?\nSubject:.*?\n", # Заголовок пересылки Outlook
r"^Begin forwarded message:",
r"^Начало переадресованного сообщения:",
]
# Лишние префиксы тем, которые можно упростить
RE_FWD_PREFIX = re.compile(
r"(?:RE|FW|FWD|Ответ|Пересланное|Пересылаемое|Перенаправлено|ПЕРЕСЛ)\s*:\s*",
re.IGNORECASE,
)
# ------------------------------------------------------------
# Основной класс очистки
# ------------------------------------------------------------
class EmailContextCleaner:
def __init__(
self,
remove_disclaimers: bool = True,
remove_signatures: bool = True,
collapse_repeats: bool = True,
remove_forward_chain: bool = True, # удалять цепочки пересылок (оставит только первое)
remove_empty_lines: bool = True,
remove_quote_lines: bool = True, # строки цитирования "> "
short_lines_only: int = 0, # удалять строки короче N символов (0 - нет)
max_url_length: int = 60, # обрезать URL длиннее N (0 - удалять полностью)
remove_timestamps: bool = True,
custom_phrases: Optional[List[str]] = None,
):
self.remove_disclaimers = remove_disclaimers
self.remove_signatures = remove_signatures
self.collapse_repeats = collapse_repeats
self.remove_forward_chain = remove_forward_chain
self.remove_empty_lines = remove_empty_lines
self.remove_quote_lines = remove_quote_lines
self.short_lines_only = short_lines_only
self.max_url_length = max_url_length
self.remove_timestamps = remove_timestamps
# Компиляция шаблонов
self._disclaimer_patterns = [
re.compile(p, re.IGNORECASE | re.DOTALL | re.MULTILINE)
for p in DISCLAIMER_PHRASES
]
if custom_phrases:
self._disclaimer_patterns += [
re.compile(p, re.IGNORECASE | re.DOTALL) for p in custom_phrases
]
self._signature_starters = [
re.compile(p) for p in SIGNATURE_STARTERS
]
self._forward_separators = [
re.compile(p, re.IGNORECASE | re.MULTILINE) for p in FORWARD_SEPARATORS
]
def clean(self, text: str) -> str:
if not text:
return text
# 1. Убрать дисклеймеры
if self.remove_disclaimers:
for pat in self._disclaimer_patterns:
# Заменяем на пробел, чтобы не склеить слова
text = pat.sub(" ", text)
# 2. Удалить цепочки пересылок (оставить только первую часть)
if self.remove_forward_chain:
# Ищем первый разделитель пересылки; всё после него удаляем,
# но только если перед этим уже был основной текст (чтобы не потерять единственное содержимое)
min_pos = len(text)
for pat in self._forward_separators:
match = pat.search(text)
if match:
pos = match.start()
if pos > 0 and pos < min_pos:
min_pos = pos
if min_pos > 0 and min_pos < len(text):
# Удаляем хвост, только если до разделителя было больше 50 символов осмысленного текста (не только заголовки)
before = text[:min_pos]
meaningful = re.sub(r"\s+", "", before)
if len(meaningful) > 50:
text = before.strip()
# 3. Удалить строки подписей (после "Best regards" и т.п.)
if self.remove_signatures:
lines = text.splitlines()
result_lines = []
for i, line in enumerate(lines):
stripped = line.strip()
if any(pat.match(stripped) for pat in self._signature_starters):
# Проверяем, не является ли строка частью основного текста
# Если дальше идут только пустые строки или короткие строки (типичная подпись), обрезаем
remaining = lines[i+1:]
# Подпись обычно не содержит длинных предложений без пунктуации
is_signature = all(
len(l.strip()) < 80 and not re.search(r"\b(договор|контракт|заказ|invoice|tracking|отгрузка|поставка|delivery)\b", l, re.IGNORECASE)
for l in remaining if l.strip()
)
if is_signature:
break
result_lines.append(line)
text = "\n".join(result_lines)
# 4. Удалить строки с цитированием (начинаются с ">")
if self.remove_quote_lines:
text = re.sub(r"^>.*$", "", text, flags=re.MULTILINE)
# Также удалить многоуровневое цитирование ">>"
text = re.sub(r"^(>\s*)+.*$", "", text, flags=re.MULTILINE)
# 5. Удалить очень короткие строки (если включено)
if self.short_lines_only > 0:
text = "\n".join(
line for line in text.splitlines()
if len(line.strip()) >= self.short_lines_only
)
# 6. Обработка URL
if self.max_url_length > 0 or self.max_url_length == 0:
url_pattern = re.compile(r"https?://\S+|ftp://\S+")
def replace_url(m):
url = m.group(0)
if self.max_url_length == 0:
return "" # удалить полностью
# Обрезать до max_url_length символов, сохранить протокол
if len(url) > self.max_url_length:
# Можно оставить домен + ... или просто обрезать
short = url[:self.max_url_length] + "..."
return short
return url
text = url_pattern.sub(replace_url, text)
# 7. Удалить временные метки (например, "2025-08-15 14:23 GMT+3")
if self.remove_timestamps:
text = re.sub(
r"\b\d{1,2}[./-]\d{1,2}[./-]\d{2,4}\s+\d{1,2}:\d{2}(?::\d{2})?\s*(?:AM|PM)?\s*(?:GMT[+-]\d+)?\b",
"",
text,
flags=re.IGNORECASE,
)
# 8. Удалить дублирующиеся пустые строки
if self.remove_empty_lines:
text = re.sub(r"\n\s*\n\s*\n+", "\n\n", text)
# Удалить одиночные пустые строки, если их больше двух подряд
text = re.sub(r"\n{3,}", "\n\n", text)
# 9. Схлопнуть повторяющиеся абзацы (уже было, но можно усилить)
if self.collapse_repeats:
text = self._deduplicate_paragraphs(text)
# 10. Убрать лишние пробелы в начале/конце строк
text = "\n".join(line.strip() for line in text.splitlines())
text = text.strip()
return text
def _deduplicate_paragraphs(self, text: str) -> str:
"""Удалить дословно повторяющиеся абзацы (как в RAGEngine)."""
parts = re.split(r"(\n\s*\n)", text)
seen: Set[str] = set()
out = []
for i, part in enumerate(parts):
if i % 2 == 1:
out.append(part)
continue
key = part.strip()
if len(key) < 50: # короткие фрагменты не дубликаты
out.append(part)
continue
if key in seen:
continue
seen.add(key)
out.append(part)
return "".join(out)
# ------------------------------------------------------------
# Быстрая функция для совместимости
# ------------------------------------------------------------
_default_cleaner = EmailContextCleaner()
def clean_email_context(text: str) -> str:
"""Очищает текст письма от мусора перед отправкой в LLM."""
return _default_cleaner.clean(text)

View File

@ -39,6 +39,16 @@ def _row_cells(ws, r: int, max_c: int = 12) -> list[tuple[int, str]]:
return out return out
def _is_num_like(s: str) -> bool:
if not isinstance(s, str):
return False
t = s.strip()
if not t:
return False
# 1, 1., 1.0
return t.isdigit() or t.replace(".", "", 1).isdigit()
def parse_interface_sheet(ws) -> list[dict]: def parse_interface_sheet(ws) -> list[dict]:
"""Разбор листа интерфейса в последовательность визуальных блоков.""" """Разбор листа интерфейса в последовательность визуальных блоков."""
blocks: list[dict] = [] blocks: list[dict] = []
@ -50,19 +60,21 @@ def parse_interface_sheet(ws) -> list[dict]:
r += 1 r += 1
continue continue
# Заголовок: 7 ячеек в колонках 1,2,3,4,5,7,9 и все значения — «номера пунктов» # Заголовок верхнего ряда:
if ( # в новых версиях файла набор колонок может отличаться, поэтому
len(cells) == 7 # распознаём динамически: строка с номерами + следующая строка с подписями
and {c[0] for c in cells} == {1, 2, 3, 4, 5, 7, 9} # в тех же колонках.
and all(str(c[1]).strip().isdigit() or c[1].strip().replace(".", "").isdigit() for c in cells) cols = [c[0] for c in cells]
): vals = [c[1] for c in cells]
nums = [c[1] for c in sorted(cells, key=lambda x: (x[0] not in (1, 2, 3, 4, 5, 7, 9), x[0]))] if len(cells) >= 4 and all(_is_num_like(v) for v in vals):
# порядок как в Excel слева направо по колонкам
nums = [c[1] for c in sorted(cells, key=lambda x: x[0])]
r2 = r + 1 r2 = r + 1
cells2 = _row_cells(ws, r2) cells2 = _row_cells(ws, r2)
if len(cells2) == 7 and {c[0] for c in cells2} == {1, 2, 3, 4, 5, 7, 9}: map2 = {c: v for c, v in cells2}
labels = [c[1] for c in sorted(cells2, key=lambda x: x[0])] if all(c in map2 and str(map2[c]).strip() for c in cols):
labels = [str(map2[c]).strip() for c in cols]
# Если в следующей строке снова только числа — это не шапка.
if not all(_is_num_like(v) for v in labels):
nums = [str(v).strip() for v in vals]
blocks.append({"type": "header", "nums": nums, "labels": labels}) blocks.append({"type": "header", "nums": nums, "labels": labels})
r += 2 r += 2
continue continue

View File

@ -1,3 +1,5 @@
from starlette.middleware.base import BaseHTTPMiddleware
from starlette.requests import Request
from fastapi.staticfiles import StaticFiles from fastapi.staticfiles import StaticFiles
from fastapi.responses import FileResponse, Response from fastapi.responses import FileResponse, Response
from typing import List, Optional, Dict from typing import List, Optional, Dict
@ -5,6 +7,7 @@ from fastapi import FastAPI, UploadFile, File, HTTPException, Form, Body
from fastapi.responses import JSONResponse from fastapi.responses import JSONResponse
from pydantic import BaseModel from pydantic import BaseModel
import uvicorn import uvicorn
#from cargo_rag_v2 import CargoRAGEngineV2
from rag_engine_gemini import RAGEngineGemini from rag_engine_gemini import RAGEngineGemini
import logging import logging
from prometheus_fastapi_instrumentator import Instrumentator from prometheus_fastapi_instrumentator import Instrumentator
@ -20,7 +23,7 @@ from urllib.parse import quote
logging.basicConfig(level=logging.INFO) logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__) logger = logging.getLogger(__name__)
os.makedirs("/opt/rag-gemini/app/reports", exist_ok=True) os.makedirs("/opt/nek/app/reports", exist_ok=True)
SHIPPING_TYPES_FILE = os.path.join(os.path.dirname(__file__), "shipping_types.json") SHIPPING_TYPES_FILE = os.path.join(os.path.dirname(__file__), "shipping_types.json")
PROCESSED_SHIPPING_TYPES_FILE = os.path.join( PROCESSED_SHIPPING_TYPES_FILE = os.path.join(
os.path.dirname(__file__), "shipping_types_processed.json" os.path.dirname(__file__), "shipping_types_processed.json"
@ -150,7 +153,7 @@ async def get_commands():
else: else:
return {"error": "commands.html not found"} return {"error": "commands.html not found"}
#rag = CargoRAGEngineV2()
rag = RAGEngineGemini() rag = RAGEngineGemini()
Instrumentator().instrument(app).expose(app) Instrumentator().instrument(app).expose(app)
@ -167,6 +170,22 @@ app.add_middleware(
expose_headers=["Content-Disposition", "Content-Length"] expose_headers=["Content-Disposition", "Content-Length"]
) )
class CollapseDuplicatePathSlashesMiddleware(BaseHTTPMiddleware):
"""Схлопывает повторяющиеся слэши в path (например POST //process-outlook-emails → /process-outlook-emails)."""
async def dispatch(self, request: Request, call_next):
path = request.scope.get("path") or ""
if "//" in path:
collapsed = path
while "//" in collapsed:
collapsed = collapsed.replace("//", "/")
request.scope["path"] = collapsed
return await call_next(request)
app.add_middleware(CollapseDuplicatePathSlashesMiddleware)
class EmailAttachment(BaseModel): class EmailAttachment(BaseModel):
filename: str filename: str
@ -207,15 +226,12 @@ class CargoQueryResponse(BaseModel):
class AnalyzeCargoRequest(BaseModel): class AnalyzeCargoRequest(BaseModel):
email_ids: List[str] email_ids: List[str]
class CargoLearningRequest(BaseModel): class CargoLearningRequest(BaseModel):
"""Сохранение примера для обучения: переписка + структурированный ответ (как в отчёте).""" """Сохранение примера для обучения: переписка + структурированный ответ (как в отчёте)."""
structured_data: dict structured_data: dict
session_id: Optional[str] = None session_id: Optional[str] = None
context_preview: Optional[str] = None context_preview: Optional[str] = None
notes: Optional[str] = None notes: Optional[str] = None
@app.post("/record-cargo-learning") @app.post("/record-cargo-learning")
async def record_cargo_learning(req: CargoLearningRequest): async def record_cargo_learning(req: CargoLearningRequest):
""" """

586
main_v2.py Normal file
View File

@ -0,0 +1,586 @@
import os
import time
import json
import base64
import hashlib
import logging
from typing import List, Optional, Dict
from pathlib import Path
import uvicorn
import openai
from fastapi import FastAPI, UploadFile, File, HTTPException, Form, Body
from fastapi.responses import JSONResponse, FileResponse, Response
from fastapi.middleware.cors import CORSMiddleware
from fastapi.staticfiles import StaticFiles
from starlette.requests import Request
from starlette.middleware.base import BaseHTTPMiddleware
from pydantic import BaseModel
from prometheus_fastapi_instrumentator import Instrumentator
from urllib.parse import quote
# Импорт нового движка
from cargo_rag_v2 import CargoRAGEngineV2
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
# ----------------------------- Конфигурация -----------------------------
SHIPPING_TYPES_FILE = os.path.join(os.path.dirname(__file__), "shipping_types.json")
PROCESSED_SHIPPING_TYPES_FILE = os.path.join(
os.path.dirname(__file__), "shipping_types_processed.json"
)
# OpenAI клиент
#openai_api_key = os.getenv("sk-9cfed30eb6894df5b9a3ffb4b1fb956d")
openai_api_key = "sk-9cfed30eb6894df5b9a3ffb4b1fb956d"
if not openai_api_key:
raise RuntimeError("OPENAI_API_KEY is not set")
openai_client = openai.OpenAI(
api_key=openai_api_key,
base_url=os.getenv("OPENAI_BASE_URL", "http://localhost:8090/v1"),
)
# Инициализация движка
rag = CargoRAGEngineV2(openai_client)
# ------------------ Хранилище сессий (писем) --------------------------
# Ключ: session_id, значение: список писем (каждое письмо словарь)
_session_emails: Dict[str, List[Dict]] = {}
# ------------------ Pydantic модели ------------------
class EmailAttachment(BaseModel):
filename: str
size: int
content: Optional[str] = None # base64
class OutlookEmail(BaseModel):
id: str
subject: str
sender: str
senderName: Optional[str] = None
body: str
body_html: Optional[str] = None
receivedTime: Optional[str] = None
to: Optional[str] = None
cc: Optional[str] = None
attachments: List[EmailAttachment] = []
class OutlookEmailsRequest(BaseModel):
emails: List[OutlookEmail]
session_id: Optional[str] = None
class CargoQueryRequest(BaseModel):
query: str
session_id: Optional[str] = None
top_k: int = 10
class CargoQueryResponse(BaseModel):
answer: str
structured_data: dict
sources: list
total_emails_analyzed: int
class AnalyzeCargoRequest(BaseModel):
email_ids: List[str]
class CargoLearningRequest(BaseModel):
structured_data: dict
session_id: Optional[str] = None
context_preview: Optional[str] = None
notes: Optional[str] = None
class ShippingTypeBase(BaseModel):
name: str
criteria: str = ""
keywords: List[str] = []
employee_email: str = ""
confirmation_template: str = ""
info_request_template: str = ""
class ShippingTypeCreate(ShippingTypeBase):
pass
class ShippingTypeUpdate(ShippingTypeBase):
pass
class ShippingType(ShippingTypeBase):
id: int
# ------------------ Middleware и приложение ------------------
app = FastAPI(title="SEPTEM Cargo RAG System V2")
app.mount("/addin", StaticFiles(directory=os.path.join(os.getcwd(), "addin"), html=True), name="addin")
OLD_HOST = "nec.septem.pro"
NEW_HOST = "nec.clients.septem.pro"
@app.middleware("http")
async def redirect_legacy_host(request, call_next):
host = request.headers.get("host", "").split(":")[0].lower()
if host == OLD_HOST:
target_url = f"https://{NEW_HOST}{request.url.path}"
if request.url.query:
target_url = f"{target_url}?{request.url.query}"
return JSONResponse(
status_code=308,
content={"detail": "Permanent Redirect"},
headers={"Location": target_url},
)
return await call_next(request)
class CollapseDuplicatePathSlashesMiddleware(BaseHTTPMiddleware):
async def dispatch(self, request: Request, call_next):
path = request.scope.get("path") or ""
if "//" in path:
collapsed = path
while "//" in collapsed:
collapsed = collapsed.replace("//", "/")
request.scope["path"] = collapsed
return await call_next(request)
app.add_middleware(CollapseDuplicatePathSlashesMiddleware)
app.add_middleware(
CORSMiddleware,
allow_origins=[
"https://nec.clients.septem.pro",
"https://localhost:3000",
"http://localhost:8501",
],
allow_credentials=True,
allow_methods=["*"],
allow_headers=["*"],
expose_headers=["Content-Disposition", "Content-Length"],
)
Instrumentator().instrument(app).expose(app)
# ------------------ Вспомогательные функции ------------------
def load_shipping_types():
if not os.path.exists(SHIPPING_TYPES_FILE):
return []
with open(SHIPPING_TYPES_FILE, "r", encoding="utf-8") as f:
types = json.load(f)
normalized_types = []
for t in types:
normalized = {}
for key, value in t.items():
clean_key = key.strip()
if clean_key == "keywords" and isinstance(value, list):
normalized[clean_key] = [kw.strip().strip('"').strip("'") for kw in value if kw.strip()]
elif isinstance(value, str):
normalized[clean_key] = value.strip()
else:
normalized[clean_key] = value
normalized_types.append(normalized)
return normalized_types
def save_shipping_types(types):
with open(SHIPPING_TYPES_FILE, "w", encoding="utf-8") as f:
json.dump(types, f, ensure_ascii=False, indent=2)
def load_processed_shipping_criteria() -> Dict[str, str]:
if not os.path.exists(PROCESSED_SHIPPING_TYPES_FILE):
return {}
try:
with open(PROCESSED_SHIPPING_TYPES_FILE, "r", encoding="utf-8") as f:
data = json.load(f)
if isinstance(data, dict):
return {str(k): str(v) for k, v in data.items() if v is not None}
return {}
except Exception as e:
logger.warning(f"Failed to load processed shipping criteria: {e}")
return {}
def save_processed_shipping_criteria(mapping: Dict[str, str]):
with open(PROCESSED_SHIPPING_TYPES_FILE, "w", encoding="utf-8") as f:
json.dump(mapping, f, ensure_ascii=False, indent=2)
def upsert_processed_shipping_criteria(type_name: str, processed_criteria: str):
if not isinstance(type_name, str) or not type_name.strip():
return
processed_criteria = processed_criteria if isinstance(processed_criteria, str) else ""
mapping = load_processed_shipping_criteria()
mapping[type_name] = processed_criteria
save_processed_shipping_criteria(mapping)
# ------------------ Эндпоинты ------------------
@app.get("/addin/taskpane.html")
async def get_taskpane():
if os.path.exists("addin/taskpane.html"):
return FileResponse("addin/taskpane.html")
elif os.path.exists("frontend/taskpane.html"):
return FileResponse("frontend/taskpane.html")
else:
return {"error": "taskpane.html not found"}
@app.get("/addin/commands.html")
async def get_commands():
if os.path.exists("addin/commands.html"):
return FileResponse("addin/commands.html")
elif os.path.exists("frontend/commands.html"):
return FileResponse("frontend/commands.html")
else:
return {"error": "commands.html not found"}
@app.post("/process-outlook-emails")
async def process_outlook_emails(request: OutlookEmailsRequest):
try:
# Генерируем session_id, если не передан
session_id = request.session_id or hashlib.md5(
f"{time.time()}{str(request.emails)}".encode()
).hexdigest()[:16]
# Сохраняем письма (со всеми полями) в хранилище
emails_list = [email.model_dump() for email in request.emails]
_session_emails[session_id] = emails_list
logger.info(f"Сессия {session_id}: сохранено {len(emails_list)} писем")
return {
"status": "success",
"session_id": session_id,
"emails_processed": len(request.emails),
}
except Exception as e:
logger.error(f"Process Outlook emails error: {e}")
raise HTTPException(status_code=500, detail=str(e))
@app.post("/query-cargo", response_model=CargoQueryResponse)
async def query_cargo(request: CargoQueryRequest):
try:
emails = _session_emails.get(request.session_id, [])
if not emails:
raise HTTPException(status_code=400, detail="No emails found for this session")
# Вызываем анализ
result = rag.analyze(emails, request.query)
if "error" in result:
raise HTTPException(status_code=500, detail=result["error"])
# Формируем текстовый ответ (answer) из критериев
answer_lines = []
for item in result.get("criteria_results", []):
val = item["value"]
if val is None:
val_str = "Не указано"
else:
val_str = str(val)
answer_lines.append(f"{item['criterion']}: {val_str}")
answer = "\n".join(answer_lines)
# Структурированные данные для UI
structured_data = {
"shipments": [
{
"shipping_type": result.get("shipping_type"),
"criteria_results": result.get("criteria_results"),
}
]
}
# Источники оригинальные письма (без вложений, только метаданные)
sources = []
for email in emails:
src = {
"id": email.get("id"),
"subject": email.get("subject"),
"sender": email.get("sender"),
"senderName": email.get("senderName"),
"receivedTime": email.get("receivedTime"),
"to": email.get("to"),
"cc": email.get("cc"),
}
sources.append(src)
return CargoQueryResponse(
answer=answer,
structured_data=structured_data,
sources=sources,
total_emails_analyzed=len(emails),
)
except HTTPException:
raise
except Exception as e:
logger.error(f"Cargo query error: {e}")
raise HTTPException(status_code=500, detail=str(e))
@app.post("/generate-cargo-report")
async def generate_cargo_report(session_id: str = Body(..., embed=True)):
try:
emails = _session_emails.get(session_id, [])
if not emails:
raise HTTPException(status_code=400, detail="No emails found for this session")
result = rag.analyze(emails)
if "error" in result:
raise HTTPException(status_code=500, detail=result["error"])
# Подготавливаем sources — все письма с вложениями (content_base64 обязательно!)
sources = []
for idx, email in enumerate(emails):
src = {
"id": email.get("id"),
"subject": email.get("subject", ""),
"sender": email.get("sender", ""),
"senderName": email.get("senderName", ""),
"receivedTime": email.get("receivedTime", ""),
"to": email.get("to", ""),
"cc": email.get("cc", ""),
"body": email.get("body", ""),
"body_html": email.get("body_html", ""),
"attachments": []
}
for att in email.get("attachments", []):
src["attachments"].append({
"filename": att.get("filename", ""),
"size": att.get("size", 0),
"text": att.get("text", ""),
"content_base64": att.get("content", "")
})
sources.append(src)
return {
"shipping_type": result.get("shipping_type", ""),
"criteria_results": result.get("criteria_results", []),
"emails_count": len(emails),
"sources": sources
}
except HTTPException:
raise
except Exception as e:
logger.error(f"Generate report error: {e}")
raise HTTPException(status_code=500, detail=str(e))
@app.get("/email-sessions/{session_id}")
async def get_session_info(session_id: str):
emails = _session_emails.get(session_id, [])
return {
"session_id": session_id,
"created_at": datetime.now().isoformat(),
"emails_count": len(emails),
}
@app.post("/upload", include_in_schema=False)
async def upload_document(file: UploadFile = File(...)):
try:
content = await file.read()
doc_id = hashlib.md5(file.filename.encode()).hexdigest()
# Здесь можно сохранить файл, но пока заглушка
return {"status": "processing", "document_id": doc_id}
except Exception as e:
logger.error(f"Upload error: {e}")
raise HTTPException(status_code=500, detail=str(e))
@app.get("/health")
async def health():
return {"status": "healthy"}
# ------------------ Вложения (старые методы, адаптированные под хранилище) ------------------
@app.get("/attachments/{session_id}/{email_index}/{attachment_index}")
async def get_attachment(session_id: str, email_index: int, attachment_index: int):
emails = _session_emails.get(session_id, [])
if email_index < 0 or email_index >= len(emails):
raise HTTPException(status_code=404, detail="Email not found")
attachments = emails[email_index].get("attachments", [])
if attachment_index < 0 or attachment_index >= len(attachments):
raise HTTPException(status_code=404, detail="Attachment not found")
att = attachments[attachment_index]
filename = att.get("filename", "attachment")
content_base64 = att.get("content")
def make_content_disposition(filename: str, disposition: str = "inline") -> str:
ascii_filename = filename.encode('ascii', 'ignore').decode('ascii') or 'attachment'
utf8_filename = quote(filename, safe='')
return f'{disposition}; filename="{ascii_filename}"; filename*=UTF-8\'\'{utf8_filename}'
if not content_base64:
# Если нет контента, возвращаем пустой ответ
return Response(
content="",
media_type="text/plain",
headers={
"Content-Disposition": make_content_disposition(filename + ".txt"),
"Access-Control-Expose-Headers": "Content-Disposition, Content-Length",
},
)
try:
file_content = base64.b64decode(content_base64)
ext = filename.lower().split('.')[-1] if '.' in filename else ''
mime_types = {
'pdf': 'application/pdf',
'doc': 'application/msword',
'docx': 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
'xls': 'application/vnd.ms-excel',
'xlsx': 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet',
'txt': 'text/plain',
'csv': 'text/csv',
'png': 'image/png',
'jpg': 'image/jpeg',
'jpeg': 'image/jpeg',
'gif': 'image/gif',
'bmp': 'image/bmp',
'zip': 'application/zip',
'rar': 'application/vnd.rar',
}
media_type = mime_types.get(ext, 'application/octet-stream')
return Response(
content=file_content,
media_type=media_type,
headers={
"Content-Disposition": make_content_disposition(filename),
"Content-Length": str(len(file_content)),
"Access-Control-Expose-Headers": "Content-Disposition, Content-Length",
},
)
except Exception as e:
logger.error(f"Error serving attachment: {e}")
raise HTTPException(status_code=500, detail=f"Error processing file: {str(e)}")
@app.get("/email-attachments/{session_id}")
async def get_email_attachments(session_id: str):
emails = _session_emails.get(session_id, [])
files = []
for email in emails:
for att in email.get("attachments", []):
filename = att.get("filename", "")
ext = filename.split(".")[-1].lower() if "." in filename else ""
if ext in ("png", "jpg", "jpeg", "gif", "bmp", "tiff", "webp"):
continue
if att.get("content"):
files.append({
"filename": filename,
"content_base64": att["content"],
})
return files
# ------------------ CRUD для типов перевозок ------------------
@app.get("/shipping-types", response_model=List[ShippingType])
async def get_shipping_types():
return load_shipping_types()
@app.post("/shipping-types", response_model=ShippingType)
async def create_shipping_type(item: ShippingTypeCreate):
types = load_shipping_types()
new_id = max([t["id"] for t in types], default=0) + 1
new_item = item.model_dump()
new_item["id"] = new_id
types.append(new_item)
save_shipping_types(types)
try:
type_name = new_item.get("name", "")
criteria_text = new_item.get("criteria", "") or ""
processed = rag.process_shipping_type_criteria(criteria_text)
upsert_processed_shipping_criteria(type_name, processed)
rag.reload_shipping_types()
except Exception as e:
logger.warning(f"AI criteria processing failed on create: {e}")
return new_item
@app.put("/shipping-types/{item_id}", response_model=ShippingType)
async def update_shipping_type(item_id: int, item: ShippingTypeUpdate):
types = load_shipping_types()
for t in types:
if t["id"] == item_id:
old_name = t.get("name", "")
t.update(item.model_dump())
save_shipping_types(types)
try:
type_name = t.get("name", "")
criteria_text = t.get("criteria", "") or ""
processed = rag.process_shipping_type_criteria(criteria_text)
if isinstance(old_name, str) and old_name.strip() and old_name != type_name:
mapping = load_processed_shipping_criteria()
if old_name in mapping:
mapping.pop(old_name, None)
save_processed_shipping_criteria(mapping)
upsert_processed_shipping_criteria(type_name, processed)
rag.reload_shipping_types()
except Exception as e:
logger.warning(f"AI criteria processing failed on update: {e}")
return t
raise HTTPException(status_code=404, detail="Type not found")
@app.delete("/shipping-types/{item_id}")
async def delete_shipping_type(item_id: int):
types = load_shipping_types()
removed = None
for t in types:
if t.get("id") == item_id:
removed = t
break
new_types = [t for t in types if t["id"] != item_id]
if len(new_types) == len(types):
raise HTTPException(status_code=404, detail="Type not found")
save_shipping_types(new_types)
try:
if removed:
type_name = removed.get("name", "")
mapping = load_processed_shipping_criteria()
if type_name in mapping:
mapping.pop(type_name, None)
save_processed_shipping_criteria(mapping)
rag.reload_shipping_types()
except Exception as e:
logger.warning(f"Failed to delete processed criteria on delete: {e}")
return {"ok": True}
@app.post("/record-cargo-learning")
async def record_cargo_learning(req: CargoLearningRequest):
try:
ok = rag.record_cargo_learning(
structured_data=req.structured_data,
session_id=req.session_id,
context_preview=req.context_preview,
notes=req.notes,
)
return {"status": "ok" if ok else "skipped", "stored": ok}
except Exception as e:
logger.error(f"record-cargo-learning error: {e}")
raise HTTPException(status_code=500, detail=str(e))
if __name__ == "__main__":
uvicorn.run(app, host="0.0.0.0", port=8000)

0
precheck/__init__.py Normal file
View File

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

View File

@ -0,0 +1,230 @@
import re
from typing import List
def extract_additional_services(text: str) -> List[str]:
"""
Извлекает дополнительные услуги ТОЛЬКО из тела письма (без вложений).
Услуга добавляется, если:
- упоминается сама услуга,
- рядом (в пределах ~120 символов) есть слово-запрос,
- отсутствует отрицание.
"""
if not text or not text.strip():
return []
text_lower = text.lower().strip()
# ----- вспомогательные функции -----
def _has_negation_near(regex: str, text_scope: str, window: int = 60) -> bool:
for m in re.finditer(regex, text_scope):
start = max(0, m.start() - window)
end = min(len(text_scope), m.end() + window)
ctx = text_scope[start:end]
if re.search(
r"(?:не\s+(?:нуж(?:ен|на|но|ны)|требу(?:ется|ются)|треб(?:уется|уются)"
r"|предоставл(?:яется|яются)|план(?:ируется|ируем)|будет|надо)"
r"|без\s|нет\s+необходимост[и]|не\s+требуется)",
ctx,
re.IGNORECASE,
):
return True
return False
def _has_request_near(regex: str, text_scope: str, window: int = 120) -> bool:
for m in re.finditer(regex, text_scope):
start = max(0, m.start() - window)
end = min(len(text_scope), m.end() + window)
ctx = text_scope[start:end]
if re.search(
r"(?:необход[иа]м[оы]?|требу[ею]тся?|нуж(?:ен|на|но|ны)|"
r"прош[уу]|предусмотр[ие]те?|организу[ей]те?|зака[жз]ите?|"
r"сдела[ей]те?|запросите?|обеспечьте?|просим|пожалуйста|"
r"просьба|required|please\s+(?:provide|arrange)|need|necessary|"
r"provide|arrange|request|ensure)",
ctx,
re.IGNORECASE,
):
return True
return False
# ----- Единый список услуг (упоминание + запрос) -----
services = {
# упаковка и тара
"упаковка / переупаковка": [
r"переупаковк[аи]", r"упаковк[аи]\s*груза", r"паллет(?:иров|изаци)",
r"обреш[её]тк[аи]", r"ж[её]стк[аи]я\s+(?:упаковка|тара)",
r"деревянн[аи]я\s+(?:тара|упаковка)", r"вакуумн[аи]я\s+упаковк[аи]",
r"креплени[ея]\s+груза", r"страпы|упорные\s+брусья",
],
"маркировка / этикетирование": [
r"маркировк[аи]", r"этикетировани[ея]",
r"нанесени[ея]\s+(?:маркировки|знаков\s+опасности)",
r"штрихкод", r"стикер[а-я]*",
],
"страхование груза": [
r"страхов(?:ани[ея]|к[аи])\s+груз[а-я]*", r"застраховать\s+груз",
r"страхов(?:ой|ого)\s+полис", r"cargo\s+insurance",
],
"таможенное оформление / брокерские услуги": [
r"таможенн(?:ое|ая|ый)\s+оформлени[ея]", r"таможн[а-я]+\s+очистк[аи]",
r"таможенн[а-я]+\s+брокер", r"customs\s+clearance", r"customs\s+broker",
],
"сертификация / разрешительные документы": [
r"сертификац[а-я]*", r"сертификат\s+соответствия",
r"декларац[а-я]*\s+соответствия", r"фитосанитарн[а-я]+\s+сертификац",
r"ветеринарн[а-я]+\s+сертификац", r"карантинн[а-я]+\s+сертификац",
r"радиационн[а-я]+\s+контрол[а-я]*", r"сертификат\s+происхождения",
],
"фитосанитарный / карантинный контроль": [
r"фитосанитарн[а-я]+\s+контрол[а-я]*", r"фитоконтрол[а-я]*",
r"карантинн[а-я]+\s+контрол[а-я]*",
],
"фумигация": [
r"фумигац[а-я]*", r"fumigation",
],
"погрузо-разгрузочные работы (ПРР)": [
r"погрузочн[а-я]*\s*[-]\s*разгрузочн[а-я]*", r"погрузк[аи]\s*/\s*разгрузк[аи]",
r"погрузочн[а-я]*\s+работ[а-я]*", r"разгрузочн[а-я]*\s+работ[а-я]*",
r"грузов[а-я]*\s+работ[а-я]*", r"тальманск[а-я]+\s+услуг",
r"сюрвейерск[а-я]+\s+услуг",
],
"складские услуги / хранение": [
r"складск[а-я]+\s+(?:услуг|обработк|логистик)", r"ответственное\s+хранение",
r"временное\s+хранение", r"бондов[а-я]+\s+склад",
r"консолидац[а-я]+\s+груз[а-я]*", r"кросс[-\s]?докинг",
r"управлени[ея]\s+запасами", r"pick\s*(?:&|and)\s*pack",
],
"доставка до двери / last mile": [
r"доставк[аи]\s+до\s+двер[ией]", r"от\s+двери\s+до\s+двери",
r"last\s*mile\s*delivery", r"плечев[а-я]+\s+доставк[аи]",
r"door[-\s]to[-\s]door",
],
"отслеживание / мониторинг груза": [
r"отслеживани[ея]\s+груз[а-я]*", r"трекинг\s+груз[а-я]*",
r"мониторинг\s+груз[а-я]*", r"gps[-\s]?слежени[ея]",
r"датчик[а-я]*\s+(?:удар[а-я]*|вскрыти[ея]|температур)",
r"cargo\s+tracking",
],
"охрана / сопровождение груза": [
r"охран[а-я]*\s+груз[а-я]*", r"сопровождени[ея]\s+груз[а-я]*",
r"вооруж[её]нн[а-я]+\s+охран[а-я]*", r"бронированн[а-я]+\s+машин",
r"security\s+escort",
],
"фотоотчёт / осмотр груза": [
r"фотоотч[её]т", r"фотографировани[ея]\s+груза",
r"осмотр[а-я]*\s+груз[а-я]*", r"инспекц[а-я]*\s+груз[а-я]*",
],
"утилизация / вывоз отходов": [
r"утилизац[а-я]*", r"уничтожени[ея]\s+(?:груз[а-я]*|отход)",
r"вывоз\s+отходов", r"disposal",
],
"переезды / relocation": [
r"переезд[а-я]*\s+(?:офис|завод|квартир)", r"перевозк[аи]\s+вещей",
r"relocation\s+service",
],
"консолидация / сборка грузов": [
r"консолидац[а-я]*\s+(?:груз[а-я]*|парти[ей])",
r"сбор[а-я]+\s+груз[а-я]*", r"сборн[а-я]+\s+отправк[аи]",
],
"обратная логистика / возврат": [
r"обратн[а-я]+\s+логистик[аи]", r"возврат\s+(?:товар[а-я]*|тары)",
r"reverse\s+logistics",
],
"фрахтовый аудит": [
r"фрахтов[а-я]+\s+аудит", r"проверк[аи]\s+счетов",
],
"консалтинг / оптимизация цепей поставок": [
r"консалтинг\s+(?:логистик|цеп[ей]|поставок)",
r"оптимизац[а-я]*\s+(?:логистик|маршрут)", r"supply\s+chain\s+consulting",
],
"сюрвейерские услуги": [
r"сюрвей[её]рск[а-я]*", r"независим[а-я]+\s+экспертиз[аи]",
],
"термобокс / термоконтейнер": [
r"термобокс[а-я]*", r"термоконтейнер[а-я]*",
r"температурн[а-я]+\s+упаковк[аи]", r"изотермическ[а-я]+\s+упаковк[аи]",
],
"хладагенты / холодоэлементы": [
r"хладагент[а-я]*", r"хладоэлемент[а-я]*", r"холодоэлемент[а-я]*",
r"сух[а-я]*\s+л[её]д", r"dry\s*ice",
],
"температурный логгер": [
r"логгер[а-я]*\s*(?:температур|считывани)", r"температурн[а-я]+\s+логгер",
r"датчик[а-я]*\s*температур", r"частота\s+считывани[ея]\s+каждые\s+\d+\s+минут",
],
"спецтранспорт / негабарит": [
r"негабаритн[а-я]+\s+перевозк[аи]", r"тяжеловесн[а-я]+\s+перевозк[аи]",
r"инжиниринг\s+маршрут[а-я]*", r"спецразрешени[ея]",
r"трал[а-я]*", r"низкорамн[а-я]+\s+платформ[а-я]*",
r"танк[-\s]?контейнер[а-я]*", r"рефрижератор[а-я]*",
r"изотермическ[а-я]+\s+фура|jumbo[-\s]?фура",
r"автовоз[а-я]*", r"зерновоз[а-я]*", r"муковоз[а-я]*",
],
"опасные грузы (ADRs)": [
r"опасн[а-я]+\s+груз[а-я]*", r"классификац[а-я]+\s+опасност[и]",
r"допог\s*/\s*adr", r"imdg", r"iata\s+dgr",
r"упаковк[аи]\s+опасн[а-я]*", r"перевозк[аи]\s+опасн[а-я]*",
],
"живые животные": [
r"жив[а-я]+\s+животн[а-я]*", r"перевозк[аи]\s+животных",
r"iata\s+live\s+animals", r"ветеринарн[а-я]+\s+сопровождени[ея]",
],
"ценные / VIP грузы": [
r"ценн[а-я]+\s+груз[а-я]*", r"vip\s+груз[а-я]*",
r"бронированн[а-я]+\s+машин[а-я]*", r"вооруж[её]нн[а-я]+\s+охран[а-я]*",
r"перевозк[аи]\s+искусств[а-я]*", r"музейн[а-я]+\s+экспонат",
r"климатическ[а-я]+\s+контейнер[а-я]*",
],
"гуманитарные / военные грузы": [
r"гуманитарн[а-я]+\s+груз[а-я]*", r"военн[а-я]+\s+груз[а-я]*",
r"бронированн[а-я]+\s+колонн[а-я]*",
],
"логистика мероприятий": [
r"логистик[аи]\s+мероприяти[йя]", r"доставк[аи]\s+оборудовани[яй]\s+для\s+(?:концерт|выстав|олимпиад)",
],
}
found = []
for service_name, patterns in services.items():
# Есть ли упоминание услуги?
mentioned = any(re.search(p, text_lower) for p in patterns)
if not mentioned:
continue
# Есть ли отрицание рядом?
if any(_has_negation_near(p, text_lower) for p in patterns):
continue
# Есть ли запросное слово рядом?
if any(_has_request_near(p, text_lower) for p in patterns):
found.append(service_name)
# Удаление дубликатов
unique = list(dict.fromkeys(found))
# Фильтрация не-услуг (фразы, которые не должны попасть в список, даже если рядом есть запрос)
stop_phrases = [
r"\btotal\b",
r"можно\s+кантовать",
r"штабелир",
r"non\s*stackable",
r"силами\s*NEC",
r"оплат[аи]\s+терминальн",
r"терминальн[а-я]+\s+расход",
r"экспедирован(?:ие)?\s*силами",
r"экспедирование\s*силами",
r"запрашиваем\s+текущие\s+ставки",
r"просят\s+индикатив",
r"нужны\s+dgm", # отдельный пункт
r"страхов(?:ани[ея]|к[аи])\s+силами\s*NEC\s*[-]\s*нет",
r"ftl|lcl|fcl", # типы перевозок
r"авиа\s*перевозк[аи]|жд\s*перевозк[аи]|морск[а-я]+\s*перевозк[аи]",
]
filtered = []
for srv in unique:
if any(re.search(ph, srv, re.IGNORECASE) for ph in stop_phrases):
continue
if len(srv.strip()) <= 3 or srv.strip().endswith(":"):
continue
filtered.append(srv.strip())
return filtered

130
precheck/prompt_builder.py Normal file
View File

@ -0,0 +1,130 @@
# precheck/prompt_builder.py
import json
from typing import Dict, List, Optional
from .scanner import ScanReport, FieldStatus
from .registry import FIELD_REGISTRY, FieldDefinition
class DynamicPromptBuilder:
"""
На основе ScanReport создаёт промпт для LLM, запрашивая только недостающие поля.
"""
def __init__(self, registry: List[FieldDefinition] = None):
self.registry = registry or FIELD_REGISTRY
def build(self, report: ScanReport,
shipping_type_name: Optional[str] = None,
context_text: str = "",
query_text: str = "") -> Dict:
"""
Возвращает словарь с ключами:
- system_prompt: str
- user_prompt: str
- json_schema: dict (схема ожидаемого ответа)
"""
# Определим, какие поля нужно запросить:
# 1) Все поля со статусом NOT_FOUND или UNCERTAIN, которые required_by_default
# 2) Дополнительно можно запросить UNCERTAIN даже если не required
fields_to_request: List[FieldDefinition] = []
for field_def in self.registry:
status = report.field_statuses.get(field_def.name)
if not status:
continue
if status.status == "NOT_FOUND" or status.status == "UNCERTAIN":
fields_to_request.append(field_def)
if not fields_to_request:
return {
"system_prompt": "",
"user_prompt": "",
"json_schema": {},
}
# Строим JSON-схему с обёрткой shipments
schema = {
"type": "object",
"properties": {
"shipments": {
"type": "array",
"items": {
"type": "object",
"properties": {},
"additionalProperties": False
}
}
},
"required": ["shipments"],
"additionalProperties": False
}
shipment_props = schema["properties"]["shipments"]["items"]["properties"]
shipment_required = []
for fd in fields_to_request:
prop = {}
if fd.type == "string":
prop["type"] = "string"
elif fd.type == "number":
prop["type"] = "number"
elif fd.type == "boolean":
prop["type"] = "boolean"
elif fd.type == "array":
prop["type"] = "array"
if fd.name == "dimensions":
prop["items"] = {
"type": "object",
"properties": {
"length_cm": {"type": "number"},
"width_cm": {"type": "number"},
"height_cm": {"type": "number"},
"weight_kg": {"type": "number"},
"volume_cbm": {"type": "number"}
}
}
elif fd.name == "vehicle_dimensions":
prop["items"] = {
"type": "object",
"properties": {
"length_cm": {"type": "number"},
"width_cm": {"type": "number"},
"height_cm": {"type": "number"}
}
}
else:
prop["items"] = {"type": "string"}
elif fd.type == "object":
prop["type"] = "object"
if fd.name == "dangerous_goods":
prop["properties"] = {
"batteries": {"type": "boolean"},
"gases": {"type": "boolean"},
"liquids": {"type": "boolean"},
"dry_ice": {"type": "boolean"}
}
shipment_props[fd.name] = prop
if fd.required_by_default:
shipment_required.append(fd.name)
if shipment_required:
schema["properties"]["shipments"]["items"]["required"] = shipment_required
# Системный промпт
system_prompt = (
"Ты — эксперт по логистике. Из предоставленного текста письма извлеки все описанные грузоперевозки. "
"Для каждой перевозки заполни только указанные ниже поля. "
"Если в письме несколько независимых партий/маршрутов — верни массив с несколькими объектами. "
"Если поле не найдено, оставь null. Ответь строго в формате JSON-объекта с единственным ключом \"shipments\". "
"Не добавляй лишних данных."
)
# Пользовательский промпт
user_prompt = (
f"Контекст письма:\n{context_text}\n\n"
f"Дополнительный запрос: {query_text}\n\n"
f"Ожидаемая JSON-схема ответа:\n{json.dumps(schema, ensure_ascii=False, indent=2)}"
)
return {
"system_prompt": system_prompt,
"user_prompt": user_prompt,
"json_schema": schema
}

265
precheck/registry.py Normal file
View File

@ -0,0 +1,265 @@
# precheck/registry.py
from dataclasses import dataclass, field
from typing import Callable, Optional, Any, List, Dict
from .extractors.weights import extract_preferred_weight_kg, extract_quantities_from_text
from .extractors.dimensions import extract_dimensions_from_text, extract_vehicle_dimensions_from_text
from .extractors.containers import extract_container_mentions
from .extractors.dangerous import extract_dangerous_phrases
@dataclass
class FieldDefinition:
name: str # ключ в JSON shipment
label: str # человекочитаемое название
type: str # "string", "number", "boolean", "array", "object"
required_by_default: bool = True # обязательно для любого типа перевозки?
category: str = "general" # для группировки в промпте
extractor: Optional[Callable[[str, Optional[Dict]], Any]] = None # функция(text, shipment=None) -> Any
validator: Optional[Callable[[Any], bool]] = None
dependencies: List[str] = field(default_factory=list) # имена полей, от которых зависит
default_value: Any = None # значение по умолчанию, если не найдено и не обязательно
def _is_non_empty_string(val: Any) -> bool:
return isinstance(val, str) and bool(val.strip())
def _is_positive_number(val: Any) -> bool:
if val is None:
return False
if isinstance(val, (int, float)):
return val > 0
return False
def _is_valid_dimensions(val: Any) -> bool:
if not isinstance(val, list):
return False
for d in val:
if not isinstance(d, dict):
return False
if not all(isinstance(d.get(k), (int, float)) and d.get(k, 0) > 0 for k in ("length_cm", "width_cm", "height_cm")):
return False
return True
def _is_valid_dangerous_goods(val: Any) -> bool:
if not isinstance(val, dict):
return False
for k in ("batteries", "gases", "liquids", "dry_ice"):
if k in val and val[k] not in (True, False, None):
return False
return True
# Простые экстракторы-обёртки для использования в реестре
def _extract_total_weight(text: str, shipment: Optional[Dict] = None) -> Optional[float]:
res = extract_preferred_weight_kg(text, shipment)
return res.get("value") if res and res.get("value") else None
def _extract_package_count(text: str, shipment: Optional[Dict] = None) -> Optional[int]:
quant = extract_quantities_from_text(text)
# берём из отдельной суммы или триплетов
cnt = quant.get("separate_sum", {}).get("package_count")
if cnt is None and quant.get("triplet_sum"):
cnt = quant["triplet_sum"].get("package_count")
return int(cnt) if cnt is not None else None
def _extract_total_volume(text: str, shipment: Optional[Dict] = None) -> Optional[float]:
quant = extract_quantities_from_text(text)
vol = quant.get("separate_sum", {}).get("total_volume_cbm")
if vol is None and quant.get("triplet_sum"):
vol = quant["triplet_sum"].get("total_volume_cbm")
return vol
def _extract_dimensions(text: str, shipment: Optional[Dict] = None) -> Optional[List[Dict]]:
dims = extract_dimensions_from_text(text)
return dims if dims else None
def _extract_vehicle_dimensions(text: str, shipment: Optional[Dict] = None) -> Optional[List[Dict]]:
vdims = extract_vehicle_dimensions_from_text(text)
return vdims if vdims else None
def _extract_dangerous_goods(text: str, shipment: Optional[Dict] = None) -> Optional[Dict]:
phrases = extract_dangerous_phrases(text)
if not phrases:
return None
# грубая классификация по ключевым словам
result = {"batteries": None, "gases": None, "liquids": None, "dry_ice": None}
text_low = text.lower()
if any(w in text_low for w in ["батаре", "batter", "lithium"]):
result["batteries"] = True
if any(w in text_low for w in ["газ", "gas", "пропан", "бутан"]):
result["gases"] = True
if any(w in text_low for w in ["жидкост", "liquid", "растворител", "краск"]):
result["liquids"] = True
if any(w in text_low for w in ["сухой лёд", "dry ice"]):
result["dry_ice"] = True
return result
# Реестр всех полей
FIELD_REGISTRY: List[FieldDefinition] = [
FieldDefinition(
name="client_name",
label="Клиент",
type="string",
required_by_default=True,
category="general",
extractor=None, # пока нет надёжного экстрактора, оставим None
validator=_is_non_empty_string
),
FieldDefinition(
name="incoterms",
label="Условия поставки Incoterms",
type="string",
required_by_default=False,
category="general",
extractor=None,
),
FieldDefinition(
name="cargo_ready_date",
label="Дата готовности груза",
type="string",
required_by_default=False,
category="general",
extractor=None,
),
FieldDefinition(
name="pickup_address",
label="Адрес забора груза",
type="string",
required_by_default=True,
category="route",
extractor=None,
),
FieldDefinition(
name="delivery_address",
label="Адрес доставки",
type="string",
required_by_default=True,
category="route",
extractor=None,
),
FieldDefinition(
name="cargo_value",
label="Стоимость груза",
type="string",
required_by_default=False,
category="general",
extractor=None,
),
FieldDefinition(
name="package_count",
label="Количество грузовых мест",
type="number",
required_by_default=True,
category="cargo",
extractor=_extract_package_count,
validator=_is_positive_number,
),
FieldDefinition(
name="total_weight_kg",
label="Вес груза (кг)",
type="number",
required_by_default=True,
category="cargo",
extractor=_extract_total_weight,
validator=_is_positive_number,
),
FieldDefinition(
name="dimensions",
label="Габариты грузовых мест",
type="array",
required_by_default=False,
category="cargo",
extractor=_extract_dimensions,
validator=_is_valid_dimensions,
),
FieldDefinition(
name="total_volume_cbm",
label="Общий объём (м³)",
type="number",
required_by_default=False,
category="cargo",
extractor=_extract_total_volume,
validator=_is_positive_number,
dependencies=["dimensions"], # может быть вычислен из размеров
),
FieldDefinition(
name="cargo_description",
label="Характер груза",
type="string",
required_by_default=True,
category="cargo",
extractor=None,
),
FieldDefinition(
name="hs_code",
label="Код ТН ВЭД",
type="string",
required_by_default=False,
category="cargo",
extractor=None,
),
FieldDefinition(
name="dangerous_goods",
label="Опасные свойства",
type="object",
required_by_default=False,
category="cargo",
extractor=_extract_dangerous_goods,
validator=_is_valid_dangerous_goods,
),
FieldDefinition(
name="msds_required",
label="Требуется MSDS",
type="boolean",
required_by_default=False,
category="documents",
extractor=None,
),
FieldDefinition(
name="dgm_report_required",
label="Требуется DGM",
type="boolean",
required_by_default=False,
category="documents",
extractor=None,
),
FieldDefinition(
name="brand_name",
label="Бренд",
type="string",
required_by_default=False,
category="general",
extractor=None,
),
FieldDefinition(
name="container_type",
label="Тип контейнера",
type="string",
required_by_default=False,
category="transport",
extractor=lambda text, sh: ", ".join(extract_container_mentions(text)) if extract_container_mentions(text) else None,
),
FieldDefinition(
name="vehicle_type",
label="Тип транспорта",
type="string",
required_by_default=False,
category="transport",
extractor=None,
),
FieldDefinition(
name="vehicle_dimensions",
label="Габариты машины",
type="array",
required_by_default=False,
category="transport",
extractor=_extract_vehicle_dimensions,
validator=_is_valid_dimensions,
),
FieldDefinition(
name="temperature_range",
label="Температурный режим",
type="string",
required_by_default=False,
category="transport",
extractor=None,
),
# ... можно добавить остальные поля по мере необходимости
]

101
precheck/scanner.py Normal file
View File

@ -0,0 +1,101 @@
# precheck/scanner.py
from typing import Dict, List, Optional, Any, Tuple
from dataclasses import dataclass, field
from .registry import FIELD_REGISTRY, FieldDefinition
@dataclass
class FieldStatus:
status: str # "FOUND", "NOT_FOUND", "UNCERTAIN"
value: Any = None
confidence: float = 1.0
message: str = ""
@dataclass
class ScanReport:
field_statuses: Dict[str, FieldStatus] = field(default_factory=dict)
extracted_values: Dict[str, Any] = field(default_factory=dict)
overall_confidence: float = 0.0
missing_required: List[str] = field(default_factory=list)
class PreCheckScanner:
"""
Сканирует текст письма (и вложений) на наличие ключевых полей,
используя детерминированные экстракторы из реестра.
"""
def __init__(self, registry: List[FieldDefinition] = None):
self.registry = registry or FIELD_REGISTRY
def scan(self, context_text: str, sources: Optional[List[Dict]] = None,
shipping_type_name: Optional[str] = None) -> ScanReport:
"""
Основной метод сканирования.
:param context_text: полный текст письма и вложений
:param sources: исходные письма (для возможной привязки к ID)
:param shipping_type_name: имя типа перевозки для учёта дополнительных обязательных полей
:return: ScanReport
"""
report = ScanReport()
found_count = 0
total_required = 0
missing_required = []
for field_def in self.registry:
# Проверяем, является ли поле обязательным для данного типа перевозки (упрощённо: если required_by_default)
# В будущем можно учесть shipping_type.mandatory_fields
if field_def.required_by_default:
total_required += 1
# Если нет экстрактора, считаем поле не найденным
if field_def.extractor is None:
status = FieldStatus(status="NOT_FOUND", confidence=0.0,
message="Нет детерминированного экстрактора")
else:
try:
# Вызываем экстрактор с текстом
value = field_def.extractor(context_text, None)
if value is not None and (field_def.validator is None or field_def.validator(value)):
status = FieldStatus(status="FOUND", value=value, confidence=0.9)
report.extracted_values[field_def.name] = value
found_count += 1
else:
# Если есть значение, но не прошло валидацию - UNCERTAIN, иначе NOT_FOUND
if value is not None:
status = FieldStatus(status="UNCERTAIN", value=value, confidence=0.4,
message="Значение не прошло валидацию")
else:
status = FieldStatus(status="NOT_FOUND", confidence=0.0)
except Exception as e:
status = FieldStatus(status="NOT_FOUND", confidence=0.0, message=str(e))
report.field_statuses[field_def.name] = status
if field_def.required_by_default and status.status != "FOUND":
missing_required.append(field_def.name)
report.missing_required = missing_required
total_fields = len(self.registry)
report.overall_confidence = found_count / total_fields if total_fields > 0 else 0.0
# Учёт зависимостей: если dimension найдены, попробуем вычислить total_volume_cbm
if "dimensions" in report.extracted_values and "total_volume_cbm" not in report.extracted_values:
dims = report.extracted_values["dimensions"]
try:
vol = sum(
(d.get("length_cm", 0) * d.get("width_cm", 0) * d.get("height_cm", 0)) / 1_000_000
for d in dims
)
if vol > 0:
report.extracted_values["total_volume_cbm"] = round(vol, 6)
# обновим статус
if "total_volume_cbm" in report.field_statuses:
report.field_statuses["total_volume_cbm"] = FieldStatus(
status="FOUND", value=vol, confidence=0.7,
message="Вычислено из габаритов"
)
# если было в missing_required, убрать
if "total_volume_cbm" in report.missing_required:
report.missing_required.remove("total_volume_cbm")
except Exception:
pass
return report

View File

@ -1,92 +1,41 @@
import logging
import os
from fastapi import FastAPI, Request from fastapi import FastAPI, Request
from fastapi.responses import JSONResponse, Response from fastapi.responses import JSONResponse, Response
import httpx import httpx
try:
from dotenv import load_dotenv
load_dotenv()
except ImportError:
pass
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
app = FastAPI() app = FastAPI()
UPSTREAM_URL = os.getenv( OPENROUTER_KEY = "sk-9cfed30eb6894df5b9a3ffb4b1fb956d"
"LLM_UPSTREAM_URL",
"https://llm.corp.septem.pro/v1/chat/completions",
).strip()
OPENROUTER_KEY = os.getenv("OPENROUTER_API_KEY", "").strip()
def _client_timeout() -> httpx.Timeout:
total = float(os.getenv("LLM_PROXY_TIMEOUT_SECONDS", "120"))
connect = float(os.getenv("LLM_PROXY_CONNECT_TIMEOUT", "15"))
return httpx.Timeout(total, connect=connect)
@app.post("/v1/chat/completions") @app.post("/v1/chat/completions")
async def proxy(req: Request): async def proxy(req: Request):
if not OPENROUTER_KEY:
logger.error("OPENROUTER_API_KEY is not set (env or .env)")
return JSONResponse(
status_code=503,
content={
"error": "Service unavailable",
"details": "Set OPENROUTER_API_KEY for the LLM proxy (or add it to .env).",
},
)
try:
body = await req.json() body = await req.json()
except Exception as e:
return JSONResponse(
status_code=400,
content={"error": "Invalid JSON", "details": str(e)},
)
verify = os.getenv("HTTPX_VERIFY", "true").lower() not in ("0", "false", "no") async with httpx.AsyncClient() as client:
async with httpx.AsyncClient(verify=verify) as client:
try: try:
r = await client.post( r = await client.post(
UPSTREAM_URL, "https://api.deepseek.com/chat/completions",
headers={ headers={
"Authorization": f"Bearer {OPENROUTER_KEY}", "Authorization": f"Bearer {OPENROUTER_KEY}",
"HTTP-Referer": os.getenv("LLM_HTTP_REFERER", "http://localhost"), "HTTP-Referer": "http://localhost",
"X-Title": os.getenv("LLM_PROXY_X_TITLE", "rag-engine"), "model": "deepseek-v4-flash",
"X-Title": "rag-engine"
}, },
json=body, json=body,
timeout=_client_timeout(), timeout=300,
)
except httpx.RequestError as e:
logger.error(
"LLM upstream unreachable (%s): %s",
UPSTREAM_URL,
e,
exc_info=True,
) )
except httpx.HTTPError as e:
# Ошибка сети / таймаут
return JSONResponse( return JSONResponse(
status_code=502, status_code=502,
content={ content={"error": "Bad gateway", "details": str(e)},
"error": "Bad gateway",
"details": str(e),
"upstream": UPSTREAM_URL,
"hint": "Проверьте VPN, DNS, доступность хоста и переменную LLM_UPSTREAM_URL.",
},
) )
if r.status_code >= 500: print(r.text)
logger.warning(
"LLM upstream HTTP %s, body (prefix): %s",
r.status_code,
(r.text or "")[:800],
)
# Если бэкенд всегда отвечает JSON:
# return JSONResponse(status_code=r.status_code, content=r.json())
# Универсальный вариант — пробрасываем «как есть»
return Response( return Response(
content=r.content, content=r.content,
status_code=r.status_code, status_code=r.status_code,

5371
rag_engine_gemini.py Normal file

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

Some files were not shown because too many files have changed in this diff Show More