06d7e922bd
A paper is a document written against a template. Its structure is never copied into it: `paper` points at a template and the outline is read live on every render, so switching `template_id` re-shapes the whole document in one write. Sentences are addressed by *position* rather than by a template row id: `paper_sentence.paper_template_filed_sort` holds the sort of the placement the sentence belongs to, and `sort` holds its place inside that paragraph. That indirection is what makes a template switch non-destructive — a sentence that remembers "position 7" lands on whatever the new template puts at position 7 — and it is why a paragraph is any position either the template or the content mentions: the structure survives with no content, and content whose position the template does not define is still rendered, in order, under 未设定. Citations are a table rather than a column, since one sentence may quote several references. `quote` is required — a citation that does not say what it quotes is refused with 422 — while `reference_id` is a plain nullable integer with no foreign key, because the reference library does not exist yet. Deleting a template a paper is written against is refused with 409 and a count, matching how the field library refuses to drop a field still in use. scripts/smoke_papers.py walks the whole loop — create, empty structure, write a paragraph with citations, switch templates, keep unmatched positions, move a paragraph, delete — in 40 checks, and cleans up after itself.
378 lines
12 KiB
Python
378 lines
12 KiB
Python
"""Request/response schemas for papers (论文).
|
||
|
||
Three shapes of payload live here:
|
||
|
||
* the **paper** itself — title, author, status, keywords, target journal and
|
||
the template it is written against;
|
||
* **paragraph** payloads — the writer's unit of work. A paragraph is edited by
|
||
reading every sentence of it and writing the whole set back, so the update
|
||
schema is a full replacement;
|
||
* the **document** — what the reader sees: the template's paragraphs in order,
|
||
each carrying the sentences stored at its position, plus the flat citation
|
||
list for the 参考文献 block.
|
||
|
||
Text normalisation is deliberately strict, because the two sides of the wire
|
||
have to agree on what counts as empty:
|
||
|
||
* a *sentence* is one line — leading/trailing whitespace goes, and any run of
|
||
whitespace inside (including a pasted newline or an ideographic space) folds
|
||
to a single space. An all-whitespace sentence becomes the empty string, which
|
||
is legal and renders as a blank line;
|
||
* a *title* must not be blank once trimmed;
|
||
* a *citation* must quote something — non-blank ``quote`` — or it is rejected.
|
||
"""
|
||
|
||
from datetime import datetime
|
||
from typing import Literal
|
||
|
||
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
||
|
||
#: Writing states, mirroring :data:`app.models.paper.PAPER_STATUSES`. Kept as a
|
||
#: literal so the generated OpenAPI document describes the enum for clients.
|
||
PaperStatus = Literal["draft", "writing", "done"]
|
||
|
||
#: Characters that separate keywords in the stored string. Half-width and
|
||
#: full-width commas, the ideographic comma, and semicolons — a user types
|
||
#: whichever one their input method offers.
|
||
KEYWORD_SEPARATORS = ",,、;;"
|
||
|
||
|
||
def fold_whitespace(value: str) -> str:
|
||
"""Collapse every whitespace run in ``value`` to one space and trim it.
|
||
|
||
``str.split`` with no argument splits on Unicode whitespace, which covers
|
||
the ideographic space (U+3000) a Chinese input method produces, so a pasted
|
||
line cannot smuggle an invisible full-width blank into a "trimmed" field.
|
||
"""
|
||
return " ".join(value.split())
|
||
|
||
|
||
def normalize_keywords(value: str | None) -> str | None:
|
||
"""Canonicalise the keyword string: trimmed, de-duplicated, ``,``-joined.
|
||
|
||
Returns ``None`` for a value that carries no keywords, so "cleared" and
|
||
"never set" are stored the same way.
|
||
"""
|
||
if value is None:
|
||
return None
|
||
|
||
parts: list[str] = []
|
||
for chunk in value.replace(";", ";").replace(",", ",").replace("、", ",").split(";"):
|
||
for item in chunk.split(","):
|
||
word = fold_whitespace(item)
|
||
if word and word not in parts:
|
||
parts.append(word)
|
||
return ", ".join(parts) or None
|
||
|
||
|
||
class CitationInput(BaseModel):
|
||
"""One citation of one sentence.
|
||
|
||
``quote`` is required: the API refuses a citation that does not say what it
|
||
quotes. ``reference_id`` is optional, because the reference library is
|
||
still to come — a citation may be written now and linked later.
|
||
"""
|
||
|
||
reference_id: int | None = Field(default=None, ge=1)
|
||
quote: str = Field(default="", max_length=4000)
|
||
|
||
@field_validator("quote")
|
||
@classmethod
|
||
def _require_quote(cls, value: str) -> str:
|
||
text = value.strip()
|
||
if not text:
|
||
raise ValueError("引用的内容不能为空")
|
||
return text
|
||
|
||
|
||
class SentenceInput(BaseModel):
|
||
"""One sentence of a paragraph, as the editor sends it.
|
||
|
||
``sort`` is optional: omit it and the server numbers the sentences from 1
|
||
in the order they arrive, which is what the editor does by default. Send it
|
||
to keep a deliberate numbering (10, 20, 30 …) or a tie.
|
||
"""
|
||
|
||
sort: int | None = Field(default=None, ge=-1_000_000, le=1_000_000)
|
||
content: str = ""
|
||
citations: list[CitationInput] = Field(default_factory=list)
|
||
|
||
@field_validator("content")
|
||
@classmethod
|
||
def _normalise_content(cls, value: str) -> str:
|
||
return fold_whitespace(value)
|
||
|
||
|
||
class ParagraphUpdate(BaseModel):
|
||
"""Payload for ``PUT /papers/{id}/paragraphs/{sort}``.
|
||
|
||
A full replacement of the paragraph's sentences. ``target_sort`` moves the
|
||
whole paragraph: the sentences are written into another paragraph's
|
||
position, appended after whatever that paragraph already holds. That is the
|
||
manual form of what a template switch does by itself.
|
||
"""
|
||
|
||
sentences: list[SentenceInput] = Field(default_factory=list)
|
||
target_sort: int | None = Field(default=None, ge=-1_000_000, le=1_000_000)
|
||
|
||
|
||
class SentenceCreate(SentenceInput):
|
||
"""Payload for ``POST /papers/{id}/sentences`` — one new sentence."""
|
||
|
||
paper_template_filed_sort: int = Field(ge=-1_000_000, le=1_000_000)
|
||
|
||
|
||
class SentenceUpdate(BaseModel):
|
||
"""Payload for ``PATCH /papers/{id}/sentences/{sentence_id}``.
|
||
|
||
Every part is optional. ``citations`` is a full replacement when present.
|
||
"""
|
||
|
||
sort: int | None = Field(default=None, ge=-1_000_000, le=1_000_000)
|
||
paper_template_filed_sort: int | None = Field(
|
||
default=None, ge=-1_000_000, le=1_000_000
|
||
)
|
||
content: str | None = None
|
||
citations: list[CitationInput] | None = None
|
||
|
||
@field_validator("content")
|
||
@classmethod
|
||
def _normalise_content(cls, value: str | None) -> str | None:
|
||
return None if value is None else fold_whitespace(value)
|
||
|
||
|
||
# --- reads ------------------------------------------------------------------
|
||
|
||
|
||
class CitationRead(BaseModel):
|
||
"""A stored citation."""
|
||
|
||
model_config = ConfigDict(from_attributes=True)
|
||
|
||
id: int
|
||
reference_id: int | None
|
||
quote: str
|
||
sort: int
|
||
|
||
|
||
class SentenceRead(BaseModel):
|
||
"""A stored sentence with its citations."""
|
||
|
||
model_config = ConfigDict(from_attributes=True)
|
||
|
||
id: int
|
||
paper_id: int
|
||
template_id: int | None
|
||
paper_template_filed_sort: int
|
||
sort: int
|
||
content: str
|
||
citations: list[CitationRead]
|
||
created_at: datetime
|
||
updated_at: datetime
|
||
|
||
|
||
class ParagraphRead(BaseModel):
|
||
"""One paragraph of a paper: its heading, and the sentences under it.
|
||
|
||
``name`` is ``None`` when the paper's template has no placement at this
|
||
position — the case that makes a template switch non-destructive. The
|
||
heading is then rendered as *未设定* while the sentences keep their place in
|
||
the document; ``matched`` says which of the two happened.
|
||
"""
|
||
|
||
paper_template_filed_sort: int
|
||
#: The placement this paragraph resolved to, when there was one.
|
||
template_field_id: int | None = None
|
||
field_id: int | None = None
|
||
#: Heading text as written in the template — the numbering is part of it.
|
||
name: str | None = None
|
||
level: int = 1
|
||
font_size: float | None = None
|
||
font_color: str | None = None
|
||
#: ``True`` when the template has a placement at this position.
|
||
matched: bool = False
|
||
sentences: list[SentenceRead]
|
||
|
||
|
||
class PaperCitationRead(BaseModel):
|
||
"""One citation as it appears in the paper's 参考文献 list.
|
||
|
||
``index`` is the number shown in the text as ``[index]``; it counts
|
||
citations in reading order across the whole document, so the list and the
|
||
markers cannot disagree.
|
||
"""
|
||
|
||
index: int
|
||
id: int
|
||
reference_id: int | None
|
||
quote: str
|
||
sentence_id: int
|
||
sentence_content: str
|
||
paper_template_filed_sort: int
|
||
paragraph_name: str | None = None
|
||
|
||
|
||
class PaperBase(BaseModel):
|
||
"""Shared body of the paper create payload."""
|
||
|
||
title: str = Field(min_length=1, max_length=255)
|
||
template_id: int | None = None
|
||
abstract: str | None = None
|
||
author: str | None = Field(default=None, max_length=255)
|
||
status: PaperStatus = "draft"
|
||
keywords: str | None = Field(default=None, max_length=255)
|
||
target_journal: str | None = Field(default=None, max_length=255)
|
||
|
||
@field_validator("title")
|
||
@classmethod
|
||
def _require_title(cls, value: str) -> str:
|
||
text = value.strip()
|
||
if not text:
|
||
raise ValueError("论文标题不能为空")
|
||
return text
|
||
|
||
@field_validator("author", "target_journal")
|
||
@classmethod
|
||
def _blank_to_none(cls, value: str | None) -> str | None:
|
||
"""Treat a whitespace-only author or journal as "not set"."""
|
||
if value is None:
|
||
return None
|
||
return value.strip() or None
|
||
|
||
@field_validator("keywords")
|
||
@classmethod
|
||
def _normalise_keywords(cls, value: str | None) -> str | None:
|
||
return normalize_keywords(value)
|
||
|
||
|
||
class PaperCreate(PaperBase):
|
||
"""Payload for ``POST /papers``.
|
||
|
||
Creation is metadata only: the structure comes from the chosen template,
|
||
read live at render time, so there is nothing to copy and no empty rows to
|
||
materialise. The paper appears with its whole outline and no content.
|
||
"""
|
||
|
||
|
||
class PaperUpdate(BaseModel):
|
||
"""Payload for ``PATCH /papers/{id}`` — every part optional.
|
||
|
||
Changing ``template_id`` is the template switch: it re-shapes the document
|
||
in one write. Sentence rows are untouched, which is why switching back
|
||
restores the previous layout exactly.
|
||
"""
|
||
|
||
title: str | None = Field(default=None, min_length=1, max_length=255)
|
||
template_id: int | None = None
|
||
abstract: str | None = None
|
||
author: str | None = Field(default=None, max_length=255)
|
||
status: PaperStatus | None = None
|
||
keywords: str | None = Field(default=None, max_length=255)
|
||
target_journal: str | None = Field(default=None, max_length=255)
|
||
|
||
@field_validator("title")
|
||
@classmethod
|
||
def _require_title(cls, value: str | None) -> str | None:
|
||
if value is None:
|
||
return None
|
||
text = value.strip()
|
||
if not text:
|
||
raise ValueError("论文标题不能为空")
|
||
return text
|
||
|
||
@field_validator("author", "target_journal")
|
||
@classmethod
|
||
def _blank_to_none(cls, value: str | None) -> str | None:
|
||
if value is None:
|
||
return None
|
||
return value.strip() or None
|
||
|
||
@field_validator("keywords")
|
||
@classmethod
|
||
def _normalise_keywords(cls, value: str | None) -> str | None:
|
||
return normalize_keywords(value)
|
||
|
||
|
||
class PaperListItem(BaseModel):
|
||
"""A paper as it appears in the table, without its content."""
|
||
|
||
id: int
|
||
title: str
|
||
template_id: int | None
|
||
#: The template's name, or ``None`` when no template is chosen.
|
||
template_name: str | None
|
||
author: str | None
|
||
status: PaperStatus
|
||
keywords: str | None
|
||
target_journal: str | None
|
||
#: How many sentences have been written.
|
||
sentence_count: int
|
||
#: How many distinct paragraphs those sentences occupy.
|
||
paragraph_count: int
|
||
#: How many paragraphs the current template defines — the denominator of
|
||
#: the progress read-out. 0 when no template is chosen.
|
||
template_paragraph_count: int
|
||
created_at: datetime
|
||
updated_at: datetime
|
||
|
||
|
||
class PaperRead(PaperListItem):
|
||
"""A paper with everything the table does not need."""
|
||
|
||
abstract: str | None
|
||
|
||
|
||
class PaperDocumentRead(BaseModel):
|
||
"""The whole paper as it reads: structure, content, citations, warnings.
|
||
|
||
Assembled by :func:`app.crud.paper.build_document` in a single pass, so a
|
||
client renders from one response and never has to stitch arrays together.
|
||
"""
|
||
|
||
paper: PaperRead
|
||
#: Paragraphs in ascending position order. Every position that either the
|
||
#: template or a sentence mentions appears exactly once, so an empty
|
||
#: paragraph is still a paragraph — the structure survives with no content.
|
||
paragraphs: list[ParagraphRead]
|
||
#: Citations in reading order, numbered to match the in-text markers.
|
||
citations: list[PaperCitationRead]
|
||
#: Anything the reader should be told about the document as a whole, e.g.
|
||
#: a template that places two fields at the same position.
|
||
warnings: list[str] = Field(default_factory=list)
|
||
|
||
|
||
class ParagraphListRead(BaseModel):
|
||
"""Response of a paragraph read: the paragraph plus the paper it belongs to.
|
||
|
||
The editor needs to know which paper it is editing and where else it could
|
||
move the paragraph to, and both are already loaded by the document query.
|
||
"""
|
||
|
||
paper_id: int
|
||
paper_title: str
|
||
template_id: int | None
|
||
paragraph: ParagraphRead
|
||
|
||
|
||
__all__ = [
|
||
"CitationInput",
|
||
"CitationRead",
|
||
"KEYWORD_SEPARATORS",
|
||
"PaperBase",
|
||
"PaperCitationRead",
|
||
"PaperCreate",
|
||
"PaperDocumentRead",
|
||
"PaperListItem",
|
||
"PaperRead",
|
||
"PaperStatus",
|
||
"PaperUpdate",
|
||
"ParagraphListRead",
|
||
"ParagraphRead",
|
||
"ParagraphUpdate",
|
||
"SentenceCreate",
|
||
"SentenceRead",
|
||
"SentenceUpdate",
|
||
"fold_whitespace",
|
||
"normalize_keywords",
|
||
]
|