Импорт .docx удалён: не использовался (0 кейсов), парсер не совместим с русскими спецдонесениями, LLM-распознавание вырезано
- admin.py: /parse-doc + _extract_docx_text/_extract_kv_pairs/_coerce_preview/ _parse_bool удалены; импорты почищены. - schemas.py: ParseDocResponse удалена. - AdminDashboard: секция «Импорт отчёта .docx» + все upload/parse handlers и state удалены (компонент стал на ~160 строк короче).
This commit is contained in:
+2
-101
@@ -1,107 +1,19 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from io import BytesIO
|
||||
import re
|
||||
import xml.etree.ElementTree as ET
|
||||
import zipfile
|
||||
|
||||
from datetime import datetime, timezone
|
||||
import logging
|
||||
|
||||
from fastapi import APIRouter, Depends, File, HTTPException, Query, UploadFile
|
||||
from fastapi import APIRouter, Depends, HTTPException, Query
|
||||
|
||||
from backend.database import db
|
||||
from backend.routers.auth import require_roles
|
||||
from backend.schemas import CaseListResponse, CaseResponse, CaseUpdate, DashboardResponse, ParseDocResponse
|
||||
from backend.schemas import CaseListResponse, CaseResponse, CaseUpdate, DashboardResponse
|
||||
|
||||
log = logging.getLogger('vector.admin')
|
||||
|
||||
router = APIRouter(prefix='/api/v1/admin', tags=['admin'], dependencies=[Depends(require_roles(['admin']))])
|
||||
|
||||
|
||||
def _parse_bool(value: str | None) -> bool | None:
|
||||
if value is None:
|
||||
return None
|
||||
normalized = value.strip().lower()
|
||||
if normalized in {'1', 'true', 'yes', 'y', 'да'}:
|
||||
return True
|
||||
if normalized in {'0', 'false', 'no', 'n', 'нет'}:
|
||||
return False
|
||||
return None
|
||||
|
||||
|
||||
def _extract_docx_text(content: bytes) -> str:
|
||||
try:
|
||||
with zipfile.ZipFile(BytesIO(content)) as archive:
|
||||
xml = archive.read('word/document.xml')
|
||||
root = ET.fromstring(xml)
|
||||
except (zipfile.BadZipFile, KeyError, ET.ParseError) as exc:
|
||||
raise ValueError('Не удалось прочитать .docx: файл повреждён или имеет неверный формат') from exc
|
||||
ns = {'w': 'http://schemas.openxmlformats.org/wordprocessingml/2006/main'}
|
||||
paragraphs: list[str] = []
|
||||
for paragraph in root.findall('.//w:body/w:p', ns):
|
||||
texts = [node.text for node in paragraph.findall('.//w:t', ns) if node.text]
|
||||
if texts:
|
||||
paragraphs.append(''.join(texts).strip())
|
||||
return '\n'.join(paragraphs).strip()
|
||||
|
||||
|
||||
def _extract_kv_pairs(raw_text: str) -> dict[str, str]:
|
||||
parsed: dict[str, str] = {}
|
||||
for line in raw_text.splitlines():
|
||||
if ':' not in line:
|
||||
continue
|
||||
key, value = line.split(':', 1)
|
||||
parsed[key.strip().lower()] = value.strip()
|
||||
return parsed
|
||||
|
||||
|
||||
def _coerce_preview(raw_text: str) -> dict:
|
||||
pairs = _extract_kv_pairs(raw_text)
|
||||
case: dict[str, object] = {}
|
||||
|
||||
if 'age' in pairs:
|
||||
try:
|
||||
case['age'] = int(pairs['age'])
|
||||
except ValueError:
|
||||
pass
|
||||
if 'gender' in pairs:
|
||||
case['gender'] = pairs['gender']
|
||||
if 'status' in pairs:
|
||||
case['status'] = pairs['status']
|
||||
if 'found alive' in pairs:
|
||||
case['found_alive'] = _parse_bool(pairs['found alive'])
|
||||
if 'found distance km' in pairs:
|
||||
try:
|
||||
case['found_distance_km'] = float(pairs['found distance km'])
|
||||
except ValueError:
|
||||
pass
|
||||
if 'found lat' in pairs:
|
||||
try:
|
||||
case['found_lat'] = float(pairs['found lat'])
|
||||
except ValueError:
|
||||
pass
|
||||
if 'found lon' in pairs:
|
||||
try:
|
||||
case['found_lon'] = float(pairs['found lon'])
|
||||
except ValueError:
|
||||
pass
|
||||
if 'who found' in pairs:
|
||||
case['who_found'] = pairs['who found']
|
||||
if 'notes' in pairs:
|
||||
case['notes'] = pairs['notes']
|
||||
elif 'note' in pairs:
|
||||
case['notes'] = pairs['note']
|
||||
|
||||
case_id = None
|
||||
if 'case id' in pairs:
|
||||
match = re.search(r'\d+', pairs['case id'])
|
||||
if match:
|
||||
case_id = match.group(0)
|
||||
|
||||
return {'raw_text': raw_text, 'case_id': case_id, 'case': case}
|
||||
|
||||
|
||||
@router.get('/cases', response_model=CaseListResponse)
|
||||
def admin_cases(
|
||||
page: int = Query(default=1, ge=1),
|
||||
@@ -157,14 +69,3 @@ def admin_update_case(case_id: str, payload: CaseUpdate) -> dict:
|
||||
@router.get('/dashboard', response_model=DashboardResponse)
|
||||
def admin_dashboard() -> dict:
|
||||
return db.stats()
|
||||
|
||||
|
||||
@router.post('/parse-doc', response_model=ParseDocResponse)
|
||||
async def admin_parse_doc(file: UploadFile = File(...)) -> dict:
|
||||
content = await file.read()
|
||||
try:
|
||||
raw_text = _extract_docx_text(content)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=400, detail=str(exc)) from exc
|
||||
preview = _coerce_preview(raw_text)
|
||||
return {'filename': file.filename, 'parsed': True, 'preview': preview, 'raw_text': raw_text}
|
||||
|
||||
Reference in New Issue
Block a user