A2-A6: logging instead of print, claude-sonnet-5 model id, timezone-aware datetime, FastAPI lifespan, .docx parse errors -> HTTP 400
This commit is contained in:
@@ -26,9 +26,12 @@ def _parse_bool(value: str | None) -> bool | None:
|
||||
|
||||
|
||||
def _extract_docx_text(content: bytes) -> str:
|
||||
with zipfile.ZipFile(BytesIO(content)) as archive:
|
||||
xml = archive.read('word/document.xml')
|
||||
root = ET.fromstring(xml)
|
||||
try:
|
||||
with zipfile.ZipFile(BytesIO(content)) as archive:
|
||||
xml = archive.read('word/document.xml')
|
||||
root = ET.fromstring(xml)
|
||||
except (zipfile.BadZipFile, KeyError, ET.ParseError) as exc:
|
||||
raise ValueError('Не удалось прочитать .docx: файл повреждён или имеет неверный формат') from exc
|
||||
ns = {'w': 'http://schemas.openxmlformats.org/wordprocessingml/2006/main'}
|
||||
paragraphs: list[str] = []
|
||||
for paragraph in root.findall('.//w:body/w:p', ns):
|
||||
@@ -134,6 +137,9 @@ def admin_dashboard() -> dict:
|
||||
@router.post('/parse-doc', response_model=ParseDocResponse)
|
||||
async def admin_parse_doc(file: UploadFile = File(...)) -> dict:
|
||||
content = await file.read()
|
||||
raw_text = _extract_docx_text(content)
|
||||
try:
|
||||
raw_text = _extract_docx_text(content)
|
||||
except ValueError as exc:
|
||||
raise HTTPException(status_code=400, detail=str(exc)) from exc
|
||||
preview = _coerce_preview(raw_text)
|
||||
return {'filename': file.filename, 'parsed': True, 'preview': preview, 'raw_text': raw_text}
|
||||
|
||||
Reference in New Issue
Block a user