diff --git a/backend/alembic/versions/c41d7b09e5af_create_paper_authoring_schema.py b/backend/alembic/versions/c41d7b09e5af_create_paper_authoring_schema.py new file mode 100644 index 0000000..15561a2 --- /dev/null +++ b/backend/alembic/versions/c41d7b09e5af_create_paper_authoring_schema.py @@ -0,0 +1,165 @@ +"""create paper authoring schema + +Adds the three tables behind the 论文 feature — writing a paper against a +template: + +``paper`` the document: title, author, status, keywords, + target journal, and the template it is written + against. +``paper_sentence`` one row per sentence, addressed by the *position* + (``paper_template_filed_sort``) of the paragraph it + belongs to rather than by a foreign key to a + ``template_field`` row, plus its own ``sort`` + inside that paragraph. +``paper_sentence_reference`` citations of one sentence, many per sentence. + +Why the paragraph is addressed by position +------------------------------------------ +Two templates have no rows in common; what they can share is a position. A +sentence that remembers "I sit at position 7" therefore survives a template +swap: it moves to whatever the new template puts at position 7, and a position +the new template does not have is still rendered, in order, under an unset +heading. Nothing is deleted by a swap and nothing is lost — see +``app.crud.paper.build_document``. + +Notes on the foreign keys, as elsewhere in this schema: TiDB parses +``FOREIGN KEY`` for compatibility and then ignores it. They are declared to +document the relationships and the integrity is enforced in the application +layer. ``paper_sentence_reference.reference_id`` deliberately has **no** +foreign key: the reference library table does not exist yet. + +Revision ID: c41d7b09e5af +Revises: 5030e3939a26 +Create Date: 2026-09-18 + +""" + +from collections.abc import Sequence + +import sqlalchemy as sa +from alembic import op + +# revision identifiers, used by Alembic. +revision: str = "c41d7b09e5af" +down_revision: str | None = "5030e3939a26" +branch_labels: str | Sequence[str] | None = None +depends_on: str | Sequence[str] | None = None + +#: ``CURRENT_TIMESTAMP`` rather than ``now()``: the spelling MySQL and TiDB +#: both accept as a DATETIME column default without expression parentheses. +_NOW = sa.text("CURRENT_TIMESTAMP") + + +def upgrade() -> None: + op.create_table( + "paper", + sa.Column("id", sa.Integer(), autoincrement=True, nullable=False), + sa.Column("title", sa.String(length=255), nullable=False), + # Nullable: a paper may exist before it has chosen a template. + sa.Column("template_id", sa.Integer(), nullable=True), + sa.Column("abstract", sa.Text(), nullable=True), + sa.Column("author", sa.String(length=255), nullable=True), + # 草稿 / 撰写中 / 已完成, stored as the short token. + sa.Column( + "status", + sa.String(length=16), + server_default=sa.text("'draft'"), + nullable=False, + ), + # 关键词 as one comma-separated string. + sa.Column("keywords", sa.String(length=255), nullable=True), + sa.Column("target_journal", sa.String(length=255), nullable=True), + sa.Column("created_at", sa.DateTime(), server_default=_NOW, nullable=False), + sa.Column("updated_at", sa.DateTime(), server_default=_NOW, nullable=False), + sa.ForeignKeyConstraint( + ["template_id"], ["paper_template.id"], ondelete="RESTRICT" + ), + sa.PrimaryKeyConstraint("id"), + ) + op.create_index("ix_paper_template_id", "paper", ["template_id"], unique=False) + + op.create_table( + "paper_sentence", + sa.Column("id", sa.Integer(), autoincrement=True, nullable=False), + sa.Column("paper_id", sa.Integer(), nullable=False), + # Provenance: the template this sentence was written against. Cleared, + # not cascaded, when that template is deleted. + sa.Column("template_id", sa.Integer(), nullable=True), + # The paragraph, as the template placement's `sort` — not a field id. + sa.Column("paper_template_filed_sort", sa.Integer(), nullable=False), + # This sentence's position inside that paragraph. + sa.Column("sort", sa.Integer(), nullable=False), + # TEXT: a "sentence" is whatever the writer treats as one line, and a + # line can be long. No server default — MySQL and TiDB cannot default a + # TEXT column, and the ORM always supplies a value. + sa.Column("content", sa.Text(), nullable=False), + sa.Column("created_at", sa.DateTime(), server_default=_NOW, nullable=False), + sa.Column("updated_at", sa.DateTime(), server_default=_NOW, nullable=False), + sa.ForeignKeyConstraint(["paper_id"], ["paper.id"], ondelete="CASCADE"), + sa.ForeignKeyConstraint( + ["template_id"], ["paper_template.id"], ondelete="SET NULL" + ), + sa.PrimaryKeyConstraint("id"), + ) + # One index for the document read: every sentence of a paper, in paragraph + # then sentence order. + op.create_index( + "ix_paper_sentence_paper_paragraph_sort", + "paper_sentence", + ["paper_id", "paper_template_filed_sort", "sort"], + unique=False, + ) + op.create_index( + "ix_paper_sentence_template_id", "paper_sentence", ["template_id"], unique=False + ) + + op.create_table( + "paper_sentence_reference", + sa.Column("id", sa.Integer(), autoincrement=True, nullable=False), + sa.Column("sentence_id", sa.Integer(), nullable=False), + # No foreign key: the reference library does not exist yet. Plain + # integer, nullable, so a citation can be written before it is linked. + sa.Column("reference_id", sa.Integer(), nullable=True), + # 引用内容 — required. A citation that does not say what it quotes is + # rejected by the API. + sa.Column("quote", sa.Text(), nullable=False), + sa.Column("sort", sa.Integer(), nullable=False), + sa.ForeignKeyConstraint( + ["sentence_id"], ["paper_sentence.id"], ondelete="CASCADE" + ), + sa.PrimaryKeyConstraint("id"), + ) + op.create_index( + "ix_paper_sentence_reference_sentence_sort", + "paper_sentence_reference", + ["sentence_id", "sort"], + unique=False, + ) + op.create_index( + "ix_paper_sentence_reference_reference_id", + "paper_sentence_reference", + ["reference_id"], + unique=False, + ) + + +def downgrade() -> None: + # Reverse dependency order throughout. + op.drop_index( + "ix_paper_sentence_reference_reference_id", + table_name="paper_sentence_reference", + ) + op.drop_index( + "ix_paper_sentence_reference_sentence_sort", + table_name="paper_sentence_reference", + ) + op.drop_table("paper_sentence_reference") + + op.drop_index("ix_paper_sentence_template_id", table_name="paper_sentence") + op.drop_index( + "ix_paper_sentence_paper_paragraph_sort", table_name="paper_sentence" + ) + op.drop_table("paper_sentence") + + op.drop_index("ix_paper_template_id", table_name="paper") + op.drop_table("paper") diff --git a/backend/app/api/router.py b/backend/app/api/router.py index 79b673c..8031352 100644 --- a/backend/app/api/router.py +++ b/backend/app/api/router.py @@ -2,9 +2,10 @@ from fastapi import APIRouter -from app.api.routes import health, section_fields, templates +from app.api.routes import health, papers, section_fields, templates api_router = APIRouter() api_router.include_router(health.router) +api_router.include_router(papers.router) api_router.include_router(section_fields.router) api_router.include_router(templates.router) diff --git a/backend/app/api/routes/papers.py b/backend/app/api/routes/papers.py new file mode 100644 index 0000000..d11e94b --- /dev/null +++ b/backend/app/api/routes/papers.py @@ -0,0 +1,304 @@ +"""Paper endpoints (论文) — the writing surface of the application. + +The shape of the API follows the shape of the work: + +* ``/papers`` is the library: create, list, edit, delete. +* ``/papers/{id}/document`` is the paper as it reads — the template's + paragraphs in order, each with the sentences stored at its position. It is + one request, because a client should never have to stitch the structure and + the content together and risk ordering them differently than the server does. +* ``/papers/{id}/paragraphs/{sort}`` is the writer's unit of work: read one + paragraph, write it back whole. + +Two rules are enforced here rather than in the schema, because they reference +rows in other tables: a paper may only point at a template that exists, and a +template may not be deleted while a paper is written against it. +""" + +from fastapi import APIRouter, Depends, HTTPException, Query, Response, status +from sqlalchemy.orm import Session + +from app.crud import paper as crud +from app.db.session import get_db +from app.models import Paper, PaperSentence, PaperTemplate +from app.schemas.common import BatchDeleteRequest, BatchDeleteResult, PageResult +from app.schemas.paper import ( + PaperCreate, + PaperDocumentRead, + PaperListItem, + PaperRead, + PaperUpdate, + ParagraphListRead, + ParagraphUpdate, + SentenceCreate, + SentenceRead, + SentenceUpdate, +) + +router = APIRouter(prefix="/papers", tags=["papers"]) + + +def _get_or_404(db: Session, paper_id: int) -> Paper: + paper = crud.get(db, paper_id) + if paper is None: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"论文 {paper_id} 不存在", + ) + return paper + + +def _get_sentence_or_404(db: Session, paper: Paper, sentence_id: int) -> PaperSentence: + """Find a sentence *within* this paper. + + Scoping the lookup to the paper is what stops a sentence id from one paper + being edited through another paper's URL. + """ + for sentence in paper.sentences: + if sentence.id == sentence_id: + return sentence + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"句子 {sentence_id} 不属于论文 {paper.id}", + ) + + +def _assert_template_exists(db: Session, template_id: int | None) -> None: + """Reject a paper pointing at a template that is not there. + + Checked in Python rather than by a foreign key, because TiDB parses but + does not enforce ``FOREIGN KEY`` — an unchecked write would happily leave a + paper with no structure and no way to notice. + """ + if template_id is None: + return + if db.get(PaperTemplate, template_id) is None: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail=f"模板 {template_id} 不存在", + ) + + +@router.get("", response_model=PageResult[PaperListItem], summary="List papers") +def list_papers( + db: Session = Depends(get_db), + keyword: str | None = Query(default=None, description="按标题、作者或关键词模糊搜索"), + status_filter: str | None = Query( + default=None, + alias="status", + description="按状态过滤:draft / writing / done", + ), + template_id: int | None = Query(default=None, description="按所用模板过滤"), + page: int = Query(default=1, ge=1), + page_size: int = Query(default=20, ge=1, le=200), +) -> PageResult[PaperListItem]: + """Browse papers, most recently edited first.""" + items, total = crud.list_papers( + db, + keyword=keyword, + status=status_filter, + template_id=template_id, + page=page, + page_size=page_size, + ) + return PageResult.build(items=items, total=total, page=page, page_size=page_size) + + +@router.post( + "", + response_model=PaperRead, + status_code=status.HTTP_201_CREATED, + summary="Create a paper", +) +def create_paper(payload: PaperCreate, db: Session = Depends(get_db)) -> PaperRead: + """Create a paper against a template. + + Nothing is written into the sentence table: the outline is the template's, + read live on every render, so a new paper is already the right shape with + every paragraph empty. + """ + _assert_template_exists(db, payload.template_id) + return crud.read(crud.create(db, payload)) + + +@router.post( + "/batch-delete", + response_model=BatchDeleteResult, + summary="Delete several papers", +) +def batch_delete_papers( + payload: BatchDeleteRequest, + db: Session = Depends(get_db), +) -> BatchDeleteResult: + """Delete the given papers, with all of their sentences and citations.""" + return BatchDeleteResult(deleted=crud.delete_many(db, payload.ids)) + + +@router.get("/{paper_id}", response_model=PaperRead, summary="Fetch one paper") +def get_paper(paper_id: int, db: Session = Depends(get_db)) -> PaperRead: + """Return a paper's metadata and its written/defined paragraph counts.""" + return crud.read(_get_or_404(db, paper_id)) + + +@router.get( + "/{paper_id}/document", + response_model=PaperDocumentRead, + summary="Fetch one paper as a document", +) +def get_paper_document( + paper_id: int, db: Session = Depends(get_db) +) -> PaperDocumentRead: + """Return the whole paper: headings, sentences, citations, warnings. + + Paragraphs arrive in ascending position order. Positions the template does + not define are included when content exists there, with ``matched: false`` + and no heading — the client renders those as 未设定 rather than dropping + them. + """ + return crud.build_document(db, _get_or_404(db, paper_id)) + + +@router.patch("/{paper_id}", response_model=PaperRead, summary="Update a paper") +def update_paper( + paper_id: int, + payload: PaperUpdate, + db: Session = Depends(get_db), +) -> PaperRead: + """Update the paper's metadata, including which template it uses. + + Changing ``template_id`` is the template switch. Sentences keep their + positions and gain the new template id, so the same content reappears under + the new template's headings — and switching back restores the old layout + exactly. + """ + paper = _get_or_404(db, paper_id) + + if "template_id" in payload.model_fields_set: + _assert_template_exists(db, payload.template_id) + + # Only the keys the client actually sent are applied: an omitted field must + # not be mistaken for an explicit null. + values = payload.model_dump(exclude_unset=True) + template_changed = ( + "template_id" in values and values["template_id"] != paper.template_id + ) + + return crud.read(crud.update(db, paper, values=values, template_changed=template_changed)) + + +@router.delete( + "/{paper_id}", + status_code=status.HTTP_204_NO_CONTENT, + summary="Delete a paper", +) +def delete_paper(paper_id: int, db: Session = Depends(get_db)) -> Response: + """Delete a paper with its sentences and citations.""" + crud.delete(db, _get_or_404(db, paper_id)) + return Response(status_code=status.HTTP_204_NO_CONTENT) + + +@router.get( + "/{paper_id}/paragraphs/{field_sort}", + response_model=ParagraphListRead, + summary="Read one paragraph", +) +def get_paragraph( + paper_id: int, + field_sort: int, + db: Session = Depends(get_db), +) -> ParagraphListRead: + """Return one paragraph, assembled exactly as the document renders it. + + A position that neither the template nor the content knows about is a 404: + there is no paragraph there to edit. + """ + paper = _get_or_404(db, paper_id) + paragraph = crud.get_paragraph(db, paper, field_sort) + if paragraph is None: + raise HTTPException( + status_code=status.HTTP_404_NOT_FOUND, + detail=f"论文 {paper_id} 没有第 {field_sort} 段", + ) + return ParagraphListRead( + paper_id=paper.id, + paper_title=paper.title, + template_id=paper.template_id, + paragraph=paragraph, + ) + + +@router.put( + "/{paper_id}/paragraphs/{field_sort}", + response_model=PaperDocumentRead, + summary="Replace one paragraph", +) +def replace_paragraph( + paper_id: int, + field_sort: int, + payload: ParagraphUpdate, + db: Session = Depends(get_db), +) -> PaperDocumentRead: + """Write a paragraph whole, and return the refreshed document. + + The whole document comes back so the reader view updates in place: after a + move, the paragraph the editor had open no longer exists at that position, + and re-deriving the page from a fresh document is simpler than patching a + stale copy. + """ + paper = _get_or_404(db, paper_id) + crud.replace_paragraph(db, paper, field_sort, payload) + return crud.build_document(db, paper) + + +@router.post( + "/{paper_id}/sentences", + response_model=SentenceRead, + status_code=status.HTTP_201_CREATED, + summary="Append one sentence", +) +def create_sentence( + paper_id: int, + payload: SentenceCreate, + db: Session = Depends(get_db), +) -> SentenceRead: + """Add a single sentence to a paragraph, at the end unless told otherwise. + + The paragraph does not have to exist in the template: a position the + template does not define is exactly the case the renderer already handles. + """ + paper = _get_or_404(db, paper_id) + return SentenceRead.model_validate(crud.append_sentence(db, paper, payload)) + + +@router.patch( + "/{paper_id}/sentences/{sentence_id}", + response_model=SentenceRead, + summary="Update one sentence", +) +def update_sentence( + paper_id: int, + sentence_id: int, + payload: SentenceUpdate, + db: Session = Depends(get_db), +) -> SentenceRead: + """Edit one sentence's text, its position, or its citations.""" + paper = _get_or_404(db, paper_id) + sentence = _get_sentence_or_404(db, paper, sentence_id) + updated = crud.update_sentence(db, paper, sentence, payload) + return SentenceRead.model_validate(updated) + + +@router.delete( + "/{paper_id}/sentences/{sentence_id}", + status_code=status.HTTP_204_NO_CONTENT, + summary="Delete one sentence", +) +def delete_sentence( + paper_id: int, + sentence_id: int, + db: Session = Depends(get_db), +) -> Response: + """Delete one sentence and the citations attached to it.""" + paper = _get_or_404(db, paper_id) + crud.delete_sentence(db, paper, _get_sentence_or_404(db, paper, sentence_id)) + return Response(status_code=status.HTTP_204_NO_CONTENT) diff --git a/backend/app/api/routes/templates.py b/backend/app/api/routes/templates.py index be63fad..88a75ab 100644 --- a/backend/app/api/routes/templates.py +++ b/backend/app/api/routes/templates.py @@ -4,11 +4,17 @@ A template is created from two pieces of free text plus a free selection of library fields. Nothing constrains the selection: the same field may be picked twice, the picks may arrive in any order, and the only thing that decides how the outline reads is ``sort``. + +A template that a paper is written against is also *structure* for that paper, +not only configuration: deleting it would leave the paper with nothing to +render. That delete is therefore refused, with the same 409 the field library +uses for a field still placed in a template. """ from fastapi import APIRouter, Depends, HTTPException, Query, Response, status from sqlalchemy.orm import Session +from app.crud import paper as paper_crud from app.crud import paper_template as crud from app.db.session import get_db from app.models import PaperTemplate @@ -41,6 +47,32 @@ def _assert_name_free(db: Session, name: str, *, exclude_id: int | None = None) ) +def _assert_not_used_by_papers(db: Session, template_ids: list[int]) -> None: + """Refuse to delete templates that papers are written against. + + All offenders are reported at once, so a batch delete does not turn into + trial and error. The fix is a template switch on the paper, which is one + click in the paper's own view — the message says so. + """ + counts = paper_crud.template_paper_counts(db, template_ids) + if not counts: + return + + blockers = [] + for template_id, count in sorted(counts.items()): + template = db.get(PaperTemplate, template_id) + name = template.name if template is not None else template_id + blockers.append(f"“{name}”({count} 篇论文)") + + raise HTTPException( + status_code=status.HTTP_409_CONFLICT, + detail=( + "以下模板正被论文使用,请先在论文里切换模板:" + + "、".join(blockers) + ), + ) + + def _assert_fields_exist(db: Session, field_ids: list[int]) -> None: """Reject a selection that references fields the library does not have. @@ -108,6 +140,7 @@ def batch_delete_templates( db: Session = Depends(get_db), ) -> BatchDeleteResult: """Delete the given templates and all of their placement rows.""" + _assert_not_used_by_papers(db, list(payload.ids)) return BatchDeleteResult(deleted=crud.delete_many(db, payload.ids)) @@ -162,6 +195,13 @@ def update_template( summary="Delete a paper template", ) def delete_template(template_id: int, db: Session = Depends(get_db)) -> Response: - """Delete a template. Library fields it referenced are left alone.""" - crud.delete(db, _get_or_404(db, template_id)) + """Delete a template. Library fields it referenced are left alone. + + Refused while a paper is written against the template: the outline is the + paper's structure, and removing it would empty the paper rather than tidy + up configuration. + """ + template = _get_or_404(db, template_id) + _assert_not_used_by_papers(db, [template_id]) + crud.delete(db, template) return Response(status_code=status.HTTP_204_NO_CONTENT) diff --git a/backend/app/crud/__init__.py b/backend/app/crud/__init__.py index 38f82e4..77d0e7d 100644 --- a/backend/app/crud/__init__.py +++ b/backend/app/crud/__init__.py @@ -5,6 +5,6 @@ models. They own the commit: a route calls one function and gets back either a persisted object or ``None``. """ -from app.crud import paper_template, section_field +from app.crud import paper, paper_template, section_field -__all__ = ["paper_template", "section_field"] +__all__ = ["paper", "paper_template", "section_field"] diff --git a/backend/app/crud/paper.py b/backend/app/crud/paper.py new file mode 100644 index 0000000..1690764 --- /dev/null +++ b/backend/app/crud/paper.py @@ -0,0 +1,608 @@ +"""Data access for papers: metadata, sentences, and the document they assemble into. + +The heart of the module is :func:`build_document`. Nothing stores a paper's +final shape; it is recomputed on every read from two sources: + +1. the **template** the paper points at — its placements, ordered by ``sort``, + which supply the headings and the empty structure; +2. the **sentences** — ordered by ``paper_template_filed_sort`` then ``sort``, + which supply the content. + +A paragraph exists if either side mentions its position. That single rule buys +all three behaviours the feature needs: + +* a paragraph with no content still renders (structure survives); +* a sentence whose position the template does not define still renders, under + an unset heading, in the right place in the order (nothing is lost when the + template changes); +* swapping the template changes nothing but the headings, because sentences are + addressed by position and never by a template row id. + +Deleting is done in Python rather than with ``ON DELETE CASCADE`` because TiDB +parses foreign keys and then ignores them; the ORM's cascades are the only +thing that actually removes dependent rows. +""" + +from collections import defaultdict +from collections.abc import Iterable, Sequence +from typing import Any + +from sqlalchemy import func, or_, select +from sqlalchemy import update as sa_update +from sqlalchemy.orm import Session + +from app.crud.filters import LIKE_ESCAPE, like_pattern +from app.models import ( + Paper, + PaperSentence, + PaperSentenceReference, + PaperTemplate, + TemplateField, +) +from app.schemas.paper import ( + CitationInput, + PaperCitationRead, + PaperCreate, + PaperDocumentRead, + PaperListItem, + PaperRead, + ParagraphRead, + ParagraphUpdate, + SentenceCreate, + SentenceRead, + SentenceUpdate, +) + + +# --- list / read ------------------------------------------------------------- + + +def _conditions( + keyword: str | None, + status: str | None, + template_id: int | None, +) -> list: + """Translate the list filters into SQLAlchemy predicates.""" + conditions = [] + if keyword: + pattern = like_pattern(keyword) + conditions.append( + or_( + Paper.title.like(pattern, escape=LIKE_ESCAPE), + Paper.author.like(pattern, escape=LIKE_ESCAPE), + Paper.keywords.like(pattern, escape=LIKE_ESCAPE), + ) + ) + if status: + conditions.append(Paper.status == status) + if template_id is not None: + conditions.append(Paper.template_id == template_id) + return conditions + + +def _sentence_count_column(): + """Correlated count of the paper's sentences.""" + return ( + select(func.count(PaperSentence.id)) + .where(PaperSentence.paper_id == Paper.id) + .correlate(Paper) + .scalar_subquery() + ) + + +def _paragraph_count_column(): + """Correlated count of the *distinct paragraphs* the paper has content in. + + ``DISTINCT`` rather than a plain count: this is "how many paragraphs have + been written", so ten sentences in one paragraph read as 1, not 10. + """ + return ( + select(func.count(func.distinct(PaperSentence.paper_template_filed_sort))) + .where(PaperSentence.paper_id == Paper.id) + .correlate(Paper) + .scalar_subquery() + ) + + +def _template_paragraph_count_column(): + """Correlated count of the paragraphs the paper's template defines. + + The denominator of the progress read-out. ``DISTINCT`` for the same reason + ties are legal in a template: two placements may share a position, and that + is one paragraph, not two. A paper with no template counts 0. + """ + return ( + select(func.count(func.distinct(TemplateField.sort))) + .where(TemplateField.template_id == Paper.template_id) + .correlate(Paper) + .scalar_subquery() + ) + + +def _list_item( + paper: Paper, + *, + sentence_count: int, + paragraph_count: int, + template_paragraph_count: int, +) -> PaperListItem: + """Assemble one table row from a paper and its three counts.""" + return PaperListItem( + id=paper.id, + title=paper.title, + template_id=paper.template_id, + # ``Paper.template`` is eagerly joined, so this costs nothing here. + template_name=paper.template.name if paper.template is not None else None, + author=paper.author, + status=paper.status, + keywords=paper.keywords, + target_journal=paper.target_journal, + sentence_count=sentence_count, + paragraph_count=paragraph_count, + template_paragraph_count=template_paragraph_count, + created_at=paper.created_at, + updated_at=paper.updated_at, + ) + + +def list_papers( + db: Session, + *, + keyword: str | None = None, + status: str | None = None, + template_id: int | None = None, + page: int = 1, + page_size: int = 20, +) -> tuple[list[PaperListItem], int]: + """Return one page of papers, most recently edited first. + + The three counts arrive as correlated subqueries rather than as loads of + the sentences themselves: a table row needs "3 / 12 段", not three hundred + sentence rows. + """ + conditions = _conditions(keyword, status, template_id) + + total = db.scalar(select(func.count(Paper.id)).where(*conditions)) or 0 + + stmt = ( + select( + Paper, + _sentence_count_column().label("sentence_count"), + _paragraph_count_column().label("paragraph_count"), + _template_paragraph_count_column().label("template_paragraph_count"), + ) + .where(*conditions) + .order_by(Paper.updated_at.desc(), Paper.id.desc()) + .offset((page - 1) * page_size) + .limit(page_size) + ) + + items = [ + _list_item( + paper, + sentence_count=sentence_count, + paragraph_count=paragraph_count, + template_paragraph_count=template_paragraph_count, + ) + for paper, sentence_count, paragraph_count, template_paragraph_count in ( + db.execute(stmt).all() + ) + ] + return items, total + + +def get(db: Session, paper_id: int) -> Paper | None: + """Return one paper with its sentences and template loaded, or ``None``. + + Both relationships are configured for eager loading, so this is a fixed + number of queries rather than one per sentence. + """ + return db.get(Paper, paper_id) + + +def get_many(db: Session, paper_ids: Sequence[int]) -> list[Paper]: + """Return every paper whose id is in ``paper_ids`` (missing ids ignored).""" + if not paper_ids: + return [] + return list(db.scalars(select(Paper).where(Paper.id.in_(list(paper_ids)))).all()) + + +def read(paper: Paper) -> PaperRead: + """Build the detail payload for one already-loaded paper. + + The counts are computed in Python from the eager-loaded relationships, so + they always agree with the document the same request renders — a SQL count + and a rendered list can never drift apart this way. + """ + item = _list_item( + paper, + sentence_count=len(paper.sentences), + paragraph_count=len({s.paper_template_filed_sort for s in paper.sentences}), + template_paragraph_count=( + len({placement.sort for placement in paper.template.items}) + if paper.template is not None + else 0 + ), + ) + return PaperRead(**item.model_dump(), abstract=paper.abstract) + + +# --- document assembly ------------------------------------------------------- + + +def build_document(db: Session, paper: Paper) -> PaperDocumentRead: + """Assemble the whole paper: structure, content, citations, warnings. + + One pass over the placements and one over the sentences. Positions are + unioned, so a position only one of the two sides knows about still appears: + an empty paragraph from the template, an unmatched paragraph from the + content. + """ + warnings: list[str] = [] + + # --- 1. the template's side: which heading sits at which position ------- + # + # Ties are legal in a template (that is deliberate upstream), so the first + # placement at a position — the relationship hands them over ordered by + # ``sort, id`` — is the one that renders the heading. The reader is told, + # because two fields sharing a position is far more often a mistake than a + # plan. + placement_by_sort: dict[int, TemplateField] = {} + tied: dict[int, list[str]] = defaultdict(list) + for item in paper.template.items if paper.template is not None else []: + if item.sort in placement_by_sort: + tied[item.sort].append(item.field.name) + else: + placement_by_sort[item.sort] = item + for sort, names in sorted(tied.items()): + head = placement_by_sort[sort].field.name + warnings.append( + f"模板在第 {sort} 段放了多个字段({head}、{'、'.join(names)})," + "正文按第一条显示。" + ) + + # --- 2. the content's side: sentences grouped by position --------------- + sentences_by_sort: dict[int, list[PaperSentence]] = defaultdict(list) + for sentence in paper.sentences: + sentences_by_sort[sentence.paper_template_filed_sort].append(sentence) + + if paper.template is None and paper.sentences: + warnings.append("这篇论文还没有选择模板,段落没有对应标题,全部按未设定显示。") + if paper.template is not None and not paper.template.items: + warnings.append("所用模板还没有任何段落字段,论文暂时没有结构。") + + # --- 3. the union, in ascending position order -------------------------- + positions = sorted(set(placement_by_sort) | set(sentences_by_sort)) + + paragraphs: list[ParagraphRead] = [] + citations: list[PaperCitationRead] = [] + citation_index = 0 + + for position in positions: + placement = placement_by_sort.get(position) + field = placement.field if placement is not None else None + + # Re-sorted rather than trusted: the relationship already orders rows, + # but ordering is a contract of this response, so it is stated here too. + rows = sorted( + sentences_by_sort.get(position, []), + key=lambda item: (item.sort, item.id), + ) + sentences = [SentenceRead.model_validate(row) for row in rows] + + paragraph = ParagraphRead( + paper_template_filed_sort=position, + template_field_id=placement.id if placement is not None else None, + field_id=field.id if field is not None else None, + name=field.name if field is not None else None, + level=field.level if field is not None else 1, + font_size=float(field.font_size) if field is not None else None, + font_color=field.font_color if field is not None else None, + matched=placement is not None, + sentences=sentences, + ) + + # Citations are numbered in reading order, so a marker in the text and + # its entry in the 参考文献 list cannot disagree. + for sentence in sentences: + for citation in sentence.citations: + citation_index += 1 + citations.append( + PaperCitationRead( + index=citation_index, + id=citation.id, + reference_id=citation.reference_id, + quote=citation.quote, + sentence_id=sentence.id, + sentence_content=sentence.content, + paper_template_filed_sort=position, + paragraph_name=paragraph.name, + ) + ) + + paragraphs.append(paragraph) + + return PaperDocumentRead( + paper=read(paper), + paragraphs=paragraphs, + citations=citations, + warnings=warnings, + ) + + +def find_paragraph( + document: PaperDocumentRead, field_sort: int +) -> ParagraphRead | None: + """Pick one paragraph out of an assembled document.""" + for paragraph in document.paragraphs: + if paragraph.paper_template_filed_sort == field_sort: + return paragraph + return None + + +def get_paragraph(db: Session, paper: Paper, field_sort: int) -> ParagraphRead | None: + """Return one paragraph, assembled exactly as the document renders it. + + Assembling the whole document to return one paragraph is not waste: it is + what guarantees the editor shows precisely what the reader will see, rather + than a second implementation of the same ordering rules. + """ + return find_paragraph(build_document(db, paper), field_sort) + + +# --- writes ------------------------------------------------------------------ + + +def _touch(db: Session, paper: Paper) -> None: + """Mark the paper as edited, on the database clock. + + Writing content does not itself change any column of ``paper``, so the + ``onupdate`` on ``updated_at`` would not fire and the list — which sorts by + it — would keep showing the creation time of a paper whose text changed an + hour ago. ``func.now()`` rather than Python's clock, so timestamps do not + depend on which host the API runs on. + """ + paper.updated_at = func.now() # type: ignore[assignment] + + +def _commit_touch(db: Session, paper: Paper) -> None: + """Commit a content change and refresh the paper's timestamps. + + The refresh is what turns the ``func.now()`` expression above back into a + real datetime for the response — MySQL and TiDB have no ``RETURNING``, so + the value only exists after a re-read. + """ + _touch(db, paper) + db.commit() + db.refresh(paper) + + +def _build_citations(items: Sequence[CitationInput]) -> list[PaperSentenceReference]: + """Materialise the citation list of one sentence, numbering it from 1.""" + return [ + PaperSentenceReference( + reference_id=item.reference_id, + quote=item.quote, + sort=index, + ) + for index, item in enumerate(items, start=1) + ] + + +def _max_sort(db: Session, paper: Paper, field_sort: int) -> int: + """The highest sentence sort currently used in one paragraph (0 if empty).""" + value = db.scalar( + select(func.coalesce(func.max(PaperSentence.sort), 0)).where( + PaperSentence.paper_id == paper.id, + PaperSentence.paper_template_filed_sort == field_sort, + ) + ) + return int(value or 0) + + +def create(db: Session, data: PaperCreate) -> Paper: + """Insert a paper. No rows are created for its paragraphs. + + The structure is read from the template on every render, so materialising + an empty sentence per paragraph would create rows that exist only to say + nothing — and that would then have to be kept in step with the template. + """ + paper = Paper(**data.model_dump()) + db.add(paper) + db.commit() + db.refresh(paper) + return paper + + +def update( + db: Session, + paper: Paper, + *, + values: dict[str, Any], + template_changed: bool, +) -> Paper: + """Apply a partial update. + + ``values`` holds only the keys the client actually sent — the route derives + it from ``model_fields_set``, which is the only way to tell "clear the + abstract" apart from "leave it alone"; both arrive as ``None``. + + When the template changes, every sentence is re-stamped with the new + template id. The sentences themselves keep their positions, which is what + makes a switch non-destructive — and it keeps the pair + ``(template_id, paper_template_filed_sort)`` an honest description of which + paragraph a sentence belongs to. + """ + for key, value in values.items(): + setattr(paper, key, value) + + if template_changed: + db.flush() + db.execute( + sa_update(PaperSentence) + .where(PaperSentence.paper_id == paper.id) + .values(template_id=paper.template_id) + ) + + _touch(db, paper) + db.commit() + db.refresh(paper) + return paper + + +def delete(db: Session, paper: Paper) -> None: + """Delete a paper and, through the ORM cascade, all of its sentences.""" + db.delete(paper) + db.commit() + + +def delete_many(db: Session, paper_ids: Sequence[int]) -> int: + """Delete several papers, returning how many actually existed.""" + papers = get_many(db, paper_ids) + for paper in papers: + db.delete(paper) + db.commit() + return len(papers) + + +# --- paragraphs and sentences ------------------------------------------------ + + +def replace_paragraph( + db: Session, + paper: Paper, + field_sort: int, + payload: ParagraphUpdate, +) -> None: + """Rewrite one paragraph's sentences, optionally moving the paragraph. + + The editor sends the paragraph whole — every line it shows — so this is a + replacement, not a merge. ``target_sort`` is the manual form of what a + template switch does by itself: the sentences are written at another + position and appended after whatever that paragraph already holds, which is + why the numbering is computed from the target's current maximum rather than + from 1. + + Blank lines are kept if the client sends them: an empty sentence is a + legitimate placeholder and renders as a blank line rather than vanishing. + """ + target = payload.target_sort if payload.target_sort is not None else field_sort + + # Clear the source paragraph first, so that a move onto itself does not + # count its own rows when working out where the appended block starts. + existing = list( + db.scalars( + select(PaperSentence).where( + PaperSentence.paper_id == paper.id, + PaperSentence.paper_template_filed_sort == field_sort, + ) + ).all() + ) + for row in existing: + db.delete(row) + db.flush() + + base = _max_sort(db, paper, target) + + for offset, item in enumerate(payload.sentences, start=1): + db.add( + PaperSentence( + paper_id=paper.id, + # Kept in step with the paper's template: the pair + # (template_id, position) is what names the paragraph. + template_id=paper.template_id, + paper_template_filed_sort=target, + sort=item.sort if item.sort is not None else base + offset, + content=item.content, + citations=_build_citations(item.citations), + ) + ) + + _commit_touch(db, paper) + + +def append_sentence(db: Session, paper: Paper, data: SentenceCreate) -> PaperSentence: + """Add one sentence to a paragraph, at the end unless told otherwise.""" + sort = ( + data.sort + if data.sort is not None + else _max_sort(db, paper, data.paper_template_filed_sort) + 1 + ) + sentence = PaperSentence( + paper_id=paper.id, + template_id=paper.template_id, + paper_template_filed_sort=data.paper_template_filed_sort, + sort=sort, + content=data.content, + citations=_build_citations(data.citations), + ) + db.add(sentence) + _commit_touch(db, paper) + db.refresh(sentence) + return sentence + + +def update_sentence( + db: Session, + paper: Paper, + sentence: PaperSentence, + data: SentenceUpdate, +) -> PaperSentence: + """Apply a partial update to one sentence. + + ``citations`` is a full replacement when present: the sentence's citation + list is short, and replacing it wholesale is the only way to be sure that a + deleted citation really is gone. + """ + if data.sort is not None: + sentence.sort = data.sort + if data.paper_template_filed_sort is not None: + sentence.paper_template_filed_sort = data.paper_template_filed_sort + if data.content is not None: + sentence.content = data.content + if data.citations is not None: + sentence.citations.clear() + db.flush() + sentence.citations.extend(_build_citations(data.citations)) + + sentence.template_id = paper.template_id + + _commit_touch(db, paper) + db.refresh(sentence) + return sentence + + +def delete_sentence(db: Session, paper: Paper, sentence: PaperSentence) -> None: + """Delete one sentence and its citations.""" + db.delete(sentence) + _commit_touch(db, paper) + + +# --- template housekeeping --------------------------------------------------- + + +def _paper_counts_by_template( + db: Session, template_ids: Iterable[int] +) -> dict[int, int]: + """How many papers use each template id.""" + wanted = list({template_id for template_id in template_ids}) + if not wanted: + return {} + stmt = ( + select(Paper.template_id, func.count(Paper.id)) + .where(Paper.template_id.in_(wanted)) + .group_by(Paper.template_id) + ) + return {template_id: count for template_id, count in db.execute(stmt).all()} + + +def template_paper_counts( + db: Session, template_ids: Sequence[int] +) -> dict[int, int]: + """Public form of :func:`_paper_counts_by_template`. + + Used by the template endpoints: deleting a template a paper is written + against would leave that paper with no structure at all, so it is refused + with the count that explains why. + """ + return _paper_counts_by_template(db, template_ids) diff --git a/backend/app/models/__init__.py b/backend/app/models/__init__.py index 2dfc885..b6f1573 100644 --- a/backend/app/models/__init__.py +++ b/backend/app/models/__init__.py @@ -3,14 +3,34 @@ Importing this package registers every model on ``Base.metadata``, which is what ``alembic revision --autogenerate`` inspects. A new model therefore has to be added to the imports below, not only to its own module. + +The import order here is for readability only: relationships are declared by +name and resolved through the SQLAlchemy registry after every module has been +imported, so a cycle between two model modules is not a problem. """ from app.models.mixins import TimestampMixin +from app.models.paper import ( + PAPER_STATUSES, + STATUS_DONE, + STATUS_DRAFT, + STATUS_WRITING, + Paper, +) +from app.models.paper_sentence import PaperSentence +from app.models.paper_sentence_reference import PaperSentenceReference from app.models.paper_template import PaperTemplate from app.models.section_field import SectionField from app.models.template_field import TemplateField __all__ = [ + "PAPER_STATUSES", + "STATUS_DONE", + "STATUS_DRAFT", + "STATUS_WRITING", + "Paper", + "PaperSentence", + "PaperSentenceReference", "PaperTemplate", "SectionField", "TemplateField", diff --git a/backend/app/models/paper.py b/backend/app/models/paper.py new file mode 100644 index 0000000..c846543 --- /dev/null +++ b/backend/app/models/paper.py @@ -0,0 +1,116 @@ +"""Papers (论文表) — the document this tool exists to produce. + +A paper is a bag of metadata plus two things that are deliberately kept apart: + +* its **structure**, which is not stored here at all. Every render reads the + :class:`~app.models.paper_template.PaperTemplate` the paper points at, live. + Nothing is copied, so switching ``template_id`` re-shapes the whole document + in one write. +* its **content**, which lives in + :class:`~app.models.paper_sentence.PaperSentence` — one row per sentence, + addressed by the paragraph's ``sort`` inside that template rather than by a + foreign key to a ``template_field`` row. + +That second choice is what makes the promised workflow work. Because a sentence +remembers "I belong at position 7", not "I belong to placement #42", swapping +the template moves every sentence to whatever the new template has at position +7. Sentences whose position the new template does not have are not lost and not +hidden: they are still rendered in position order, under an *unset* heading. +See :func:`app.crud.paper.build_document`. + +``status`` is one of :data:`PAPER_STATUSES`. It is stored as the short token +(``draft``), not as the Chinese label, so the wording in the UI can change +without a migration. +""" + +from typing import TYPE_CHECKING + +from sqlalchemy import ForeignKey, Integer, String, Text, text +from sqlalchemy.orm import Mapped, mapped_column, relationship + +from app.db.base import Base +from app.models.mixins import TimestampMixin + +if TYPE_CHECKING: # pragma: no cover - typing only + from app.models.paper_sentence import PaperSentence + from app.models.paper_template import PaperTemplate + +#: Writing state of a paper: 草稿 / 撰写中 / 已完成. +#: +#: Exactly the values a client may send; the API validates against this tuple, +#: so adding a state is a one-line change here plus one label in the UI. +PAPER_STATUSES: tuple[str, ...] = ("draft", "writing", "done") + +STATUS_DRAFT = "draft" +STATUS_WRITING = "writing" +STATUS_DONE = "done" + + +class Paper(TimestampMixin, Base): + """One paper: a working title, who is writing it, and where it is going.""" + + __tablename__ = "paper" + + id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True) + + #: Working title. Not unique: two papers may legitimately share a title + #: while one is being split off the other. + title: Mapped[str] = mapped_column(String(255), nullable=False) + + #: The template this paper is currently written against, or ``None`` for a + #: paper that has not chosen one yet. Nullable only for robustness — the + #: UI asks for a template at creation time — and a paper without one simply + #: renders its sentences under unset headings. + #: + #: ``RESTRICT`` documents the intent (a template in use must not vanish); + #: TiDB does not enforce it, so the API refuses the template delete too. + template_id: Mapped[int | None] = mapped_column( + ForeignKey("paper_template.id", ondelete="RESTRICT"), + nullable=True, + ) + + #: 摘要 — the paper's own abstract, free text. + abstract: Mapped[str | None] = mapped_column(Text, nullable=True) + + #: 作者. Free text, since authorship is written as it will be printed. + author: Mapped[str | None] = mapped_column(String(255), nullable=True) + + #: One of :data:`PAPER_STATUSES`. + status: Mapped[str] = mapped_column( + String(16), + nullable=False, + default=STATUS_DRAFT, + server_default=text("'draft'"), + ) + + #: 关键词, stored as one comma-separated string. A list would need a table + #: of its own for something that is displayed as a row of tags and searched + #: as text; the split/join happens at the edges. + keywords: Mapped[str | None] = mapped_column(String(255), nullable=True) + + #: 投稿目标期刊. + target_journal: Mapped[str | None] = mapped_column(String(255), nullable=True) + + #: Eager-loaded with the paper: every read of a paper shows its template + #: name, and a lazy load there would be one query per row in the table. + template: Mapped["PaperTemplate | None"] = relationship(lazy="joined") + + #: The paper's sentences, in document order. + #: + #: ``delete-orphan`` is doing real work: TiDB parses but does not enforce + #: ``ON DELETE CASCADE``, so removing a deleted paper's sentences is the + #: ORM's job. The ordering mirrors the render order exactly — paragraph + #: position first, then position inside the paragraph, then insertion. + sentences: Mapped[list["PaperSentence"]] = relationship( + back_populates="paper", + cascade="all, delete-orphan", + order_by=( + "PaperSentence.paper_template_filed_sort, " + "PaperSentence.sort, " + "PaperSentence.id" + ), + lazy="selectin", + ) + + def __repr__(self) -> str: # pragma: no cover - debugging aid + return f"" diff --git a/backend/app/models/paper_sentence.py b/backend/app/models/paper_sentence.py new file mode 100644 index 0000000..d1dc1ef --- /dev/null +++ b/backend/app/models/paper_sentence.py @@ -0,0 +1,105 @@ +"""Paper sentences (论文句子表) — the actual writing. + +One row is one sentence of one paper. The two columns that place it are the +whole design: + +``paper_template_filed_sort`` + The ``sort`` of the template placement this sentence belongs to — *not* a + foreign key to ``template_field``. It is deliberately the position rather + than the row id, because the position is the only thing two templates can + meaningfully have in common. Swap the paper's template and this sentence + lands on whatever the new template puts at that position, with no per + sentence editing. (The name keeps the spelling the feature was specified + with; it reads ``paper_template_field_sort``.) + +``sort`` + Where this sentence sits *inside* that paragraph. A paragraph is reassembled + by reading its rows in ascending ``sort``, which is why the column is + required and why the API never leaves it to chance. Values are sparse- + friendly — 10, 20, 30 leaves room to insert — and ties are legal, broken by + ``id`` so the order is always total and stable. + +``template_id`` records the template the sentence was *written against*. It is +provenance, not a lookup key: rendering never filters on it, which is exactly +why sentences survive a template switch. It is nullable because deleting an old +template clears it rather than leaving a dangling reference behind. + +Position in the paper is therefore not stored anywhere as a whole; the document +is assembled on read from the paper's template plus these rows. See +:func:`app.crud.paper.build_document`. +""" + +from typing import TYPE_CHECKING + +from sqlalchemy import ForeignKey, Index, Integer, Text +from sqlalchemy.orm import Mapped, mapped_column, relationship + +from app.db.base import Base +from app.models.mixins import TimestampMixin + +if TYPE_CHECKING: # pragma: no cover - typing only + from app.models.paper import Paper + from app.models.paper_sentence_reference import PaperSentenceReference + from app.models.paper_template import PaperTemplate + + +class PaperSentence(TimestampMixin, Base): + """One sentence, at one position, inside one paragraph of one paper.""" + + __tablename__ = "paper_sentence" + + __table_args__ = ( + # The document query is "every sentence of this paper, in paragraph then + # sentence order", so one composite index covers filter and sort both. + Index( + "ix_paper_sentence_paper_paragraph_sort", + "paper_id", + "paper_template_filed_sort", + "sort", + ), + Index("ix_paper_sentence_template_id", "template_id"), + ) + + id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True) + + paper_id: Mapped[int] = mapped_column( + ForeignKey("paper.id", ondelete="CASCADE"), + nullable=False, + ) + + #: The template this sentence was written against — provenance only. Reads + #: never filter on it; see the module docstring. + template_id: Mapped[int | None] = mapped_column( + ForeignKey("paper_template.id", ondelete="SET NULL"), + nullable=True, + ) + + #: Which paragraph, expressed as the template placement's ``sort``. + #: Required: a sentence with no paragraph has nowhere to be rendered. + paper_template_filed_sort: Mapped[int] = mapped_column(Integer, nullable=False) + + #: This sentence's position inside that paragraph. Ascending, ties broken + #: by ``id``. Required for the same reason. + sort: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + + #: The sentence itself. May be empty: an empty sentence is a legitimate + #: placeholder and renders as a blank line rather than disappearing. + content: Mapped[str] = mapped_column(Text, nullable=False, default="") + + paper: Mapped["Paper"] = relationship(back_populates="sentences") + + #: References quoted by this sentence, in citation order. Eager-loaded + #: because the document render always needs them, and one query per + #: sentence would be the difference between two queries and two hundred. + citations: Mapped[list["PaperSentenceReference"]] = relationship( + back_populates="sentence", + cascade="all, delete-orphan", + order_by="PaperSentenceReference.sort, PaperSentenceReference.id", + lazy="selectin", + ) + + def __repr__(self) -> str: # pragma: no cover - debugging aid + return ( + f"" + ) diff --git a/backend/app/models/paper_sentence_reference.py b/backend/app/models/paper_sentence_reference.py new file mode 100644 index 0000000..3abc44b --- /dev/null +++ b/backend/app/models/paper_sentence_reference.py @@ -0,0 +1,72 @@ +"""Citations attached to a sentence (句子引用关联表). + +A sentence may quote **several** references, which makes this a real +many-to-many relation rather than a column on ``paper_sentence`` — hence a +table of its own. + +Why there is no foreign key to a reference table +------------------------------------------------ +References are going to be maintained in their own table later. This one +therefore stores a plain ``reference_id`` integer and declares **no** foreign +key at all: a key to a table that does not exist yet cannot be validated, +declared or migrated. The column is nullable so a citation can be written +before it has been linked to a reference row. + +What makes a citation valid +--------------------------- +``quote`` — the quoted content — is required and must not be blank. A citation +that points at something without saying what it points at is not usable in a +document, so the API rejects it (``app.schemas.paper.CitationInput``) rather +than storing a dangling half-record. The reverse is fine: a citation may carry +its quoted content with ``reference_id`` still empty, and be linked later. +""" + +from typing import TYPE_CHECKING + +from sqlalchemy import ForeignKey, Index, Integer, Text +from sqlalchemy.orm import Mapped, mapped_column, relationship + +from app.db.base import Base + +if TYPE_CHECKING: # pragma: no cover - typing only + from app.models.paper_sentence import PaperSentence + + +class PaperSentenceReference(Base): + """One citation of one sentence: what is quoted, and optionally by which id.""" + + __tablename__ = "paper_sentence_reference" + + __table_args__ = ( + # Reads are "the citations of this sentence, in order". + Index("ix_paper_sentence_reference_sentence_sort", "sentence_id", "sort"), + # The reference side is indexed for the lookup that arrives with the + # reference table: "which sentences quote reference N". + Index("ix_paper_sentence_reference_reference_id", "reference_id"), + ) + + id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True) + + sentence_id: Mapped[int] = mapped_column( + ForeignKey("paper_sentence.id", ondelete="CASCADE"), + nullable=False, + ) + + #: The id this citation will resolve to once the reference library exists. + #: Deliberately not a foreign key yet — see the module docstring. + reference_id: Mapped[int | None] = mapped_column(Integer, nullable=True) + + #: 引用内容 — required, non-blank. The citation's whole payload today. + quote: Mapped[str] = mapped_column(Text, nullable=False) + + #: Order among the citations of one sentence. Assigned from the order the + #: client sent them, so ``[1]`` in the text is the first stored row. + sort: Mapped[int] = mapped_column(Integer, nullable=False, default=0) + + sentence: Mapped["PaperSentence"] = relationship(back_populates="citations") + + def __repr__(self) -> str: # pragma: no cover - debugging aid + return ( + f"" + ) diff --git a/backend/app/schemas/__init__.py b/backend/app/schemas/__init__.py index d5a1ef6..1694815 100644 --- a/backend/app/schemas/__init__.py +++ b/backend/app/schemas/__init__.py @@ -6,6 +6,23 @@ from app.schemas.common import ( PageResult, normalize_hex_color, ) +from app.schemas.paper import ( + CitationInput, + CitationRead, + PaperCitationRead, + PaperCreate, + PaperDocumentRead, + PaperListItem, + PaperRead, + PaperStatus, + PaperUpdate, + ParagraphListRead, + ParagraphRead, + ParagraphUpdate, + SentenceCreate, + SentenceRead, + SentenceUpdate, +) from app.schemas.paper_template import ( PaperTemplateCreate, PaperTemplateListItem, @@ -23,6 +40,21 @@ from app.schemas.section_field import ( __all__ = [ "BatchDeleteRequest", + "CitationInput", + "CitationRead", + "PaperCitationRead", + "PaperCreate", + "PaperDocumentRead", + "PaperListItem", + "PaperRead", + "PaperStatus", + "PaperUpdate", + "ParagraphListRead", + "ParagraphRead", + "ParagraphUpdate", + "SentenceCreate", + "SentenceRead", + "SentenceUpdate", "BatchDeleteResult", "PageResult", "PaperTemplateCreate", diff --git a/backend/app/schemas/paper.py b/backend/app/schemas/paper.py new file mode 100644 index 0000000..7903cdb --- /dev/null +++ b/backend/app/schemas/paper.py @@ -0,0 +1,377 @@ +"""Request/response schemas for papers (论文). + +Three shapes of payload live here: + +* the **paper** itself — title, author, status, keywords, target journal and + the template it is written against; +* **paragraph** payloads — the writer's unit of work. A paragraph is edited by + reading every sentence of it and writing the whole set back, so the update + schema is a full replacement; +* the **document** — what the reader sees: the template's paragraphs in order, + each carrying the sentences stored at its position, plus the flat citation + list for the 参考文献 block. + +Text normalisation is deliberately strict, because the two sides of the wire +have to agree on what counts as empty: + +* a *sentence* is one line — leading/trailing whitespace goes, and any run of + whitespace inside (including a pasted newline or an ideographic space) folds + to a single space. An all-whitespace sentence becomes the empty string, which + is legal and renders as a blank line; +* a *title* must not be blank once trimmed; +* a *citation* must quote something — non-blank ``quote`` — or it is rejected. +""" + +from datetime import datetime +from typing import Literal + +from pydantic import BaseModel, ConfigDict, Field, field_validator + +#: Writing states, mirroring :data:`app.models.paper.PAPER_STATUSES`. Kept as a +#: literal so the generated OpenAPI document describes the enum for clients. +PaperStatus = Literal["draft", "writing", "done"] + +#: Characters that separate keywords in the stored string. Half-width and +#: full-width commas, the ideographic comma, and semicolons — a user types +#: whichever one their input method offers. +KEYWORD_SEPARATORS = ",,、;;" + + +def fold_whitespace(value: str) -> str: + """Collapse every whitespace run in ``value`` to one space and trim it. + + ``str.split`` with no argument splits on Unicode whitespace, which covers + the ideographic space (U+3000) a Chinese input method produces, so a pasted + line cannot smuggle an invisible full-width blank into a "trimmed" field. + """ + return " ".join(value.split()) + + +def normalize_keywords(value: str | None) -> str | None: + """Canonicalise the keyword string: trimmed, de-duplicated, ``,``-joined. + + Returns ``None`` for a value that carries no keywords, so "cleared" and + "never set" are stored the same way. + """ + if value is None: + return None + + parts: list[str] = [] + for chunk in value.replace(";", ";").replace(",", ",").replace("、", ",").split(";"): + for item in chunk.split(","): + word = fold_whitespace(item) + if word and word not in parts: + parts.append(word) + return ", ".join(parts) or None + + +class CitationInput(BaseModel): + """One citation of one sentence. + + ``quote`` is required: the API refuses a citation that does not say what it + quotes. ``reference_id`` is optional, because the reference library is + still to come — a citation may be written now and linked later. + """ + + reference_id: int | None = Field(default=None, ge=1) + quote: str = Field(default="", max_length=4000) + + @field_validator("quote") + @classmethod + def _require_quote(cls, value: str) -> str: + text = value.strip() + if not text: + raise ValueError("引用的内容不能为空") + return text + + +class SentenceInput(BaseModel): + """One sentence of a paragraph, as the editor sends it. + + ``sort`` is optional: omit it and the server numbers the sentences from 1 + in the order they arrive, which is what the editor does by default. Send it + to keep a deliberate numbering (10, 20, 30 …) or a tie. + """ + + sort: int | None = Field(default=None, ge=-1_000_000, le=1_000_000) + content: str = "" + citations: list[CitationInput] = Field(default_factory=list) + + @field_validator("content") + @classmethod + def _normalise_content(cls, value: str) -> str: + return fold_whitespace(value) + + +class ParagraphUpdate(BaseModel): + """Payload for ``PUT /papers/{id}/paragraphs/{sort}``. + + A full replacement of the paragraph's sentences. ``target_sort`` moves the + whole paragraph: the sentences are written into another paragraph's + position, appended after whatever that paragraph already holds. That is the + manual form of what a template switch does by itself. + """ + + sentences: list[SentenceInput] = Field(default_factory=list) + target_sort: int | None = Field(default=None, ge=-1_000_000, le=1_000_000) + + +class SentenceCreate(SentenceInput): + """Payload for ``POST /papers/{id}/sentences`` — one new sentence.""" + + paper_template_filed_sort: int = Field(ge=-1_000_000, le=1_000_000) + + +class SentenceUpdate(BaseModel): + """Payload for ``PATCH /papers/{id}/sentences/{sentence_id}``. + + Every part is optional. ``citations`` is a full replacement when present. + """ + + sort: int | None = Field(default=None, ge=-1_000_000, le=1_000_000) + paper_template_filed_sort: int | None = Field( + default=None, ge=-1_000_000, le=1_000_000 + ) + content: str | None = None + citations: list[CitationInput] | None = None + + @field_validator("content") + @classmethod + def _normalise_content(cls, value: str | None) -> str | None: + return None if value is None else fold_whitespace(value) + + +# --- reads ------------------------------------------------------------------ + + +class CitationRead(BaseModel): + """A stored citation.""" + + model_config = ConfigDict(from_attributes=True) + + id: int + reference_id: int | None + quote: str + sort: int + + +class SentenceRead(BaseModel): + """A stored sentence with its citations.""" + + model_config = ConfigDict(from_attributes=True) + + id: int + paper_id: int + template_id: int | None + paper_template_filed_sort: int + sort: int + content: str + citations: list[CitationRead] + created_at: datetime + updated_at: datetime + + +class ParagraphRead(BaseModel): + """One paragraph of a paper: its heading, and the sentences under it. + + ``name`` is ``None`` when the paper's template has no placement at this + position — the case that makes a template switch non-destructive. The + heading is then rendered as *未设定* while the sentences keep their place in + the document; ``matched`` says which of the two happened. + """ + + paper_template_filed_sort: int + #: The placement this paragraph resolved to, when there was one. + template_field_id: int | None = None + field_id: int | None = None + #: Heading text as written in the template — the numbering is part of it. + name: str | None = None + level: int = 1 + font_size: float | None = None + font_color: str | None = None + #: ``True`` when the template has a placement at this position. + matched: bool = False + sentences: list[SentenceRead] + + +class PaperCitationRead(BaseModel): + """One citation as it appears in the paper's 参考文献 list. + + ``index`` is the number shown in the text as ``[index]``; it counts + citations in reading order across the whole document, so the list and the + markers cannot disagree. + """ + + index: int + id: int + reference_id: int | None + quote: str + sentence_id: int + sentence_content: str + paper_template_filed_sort: int + paragraph_name: str | None = None + + +class PaperBase(BaseModel): + """Shared body of the paper create payload.""" + + title: str = Field(min_length=1, max_length=255) + template_id: int | None = None + abstract: str | None = None + author: str | None = Field(default=None, max_length=255) + status: PaperStatus = "draft" + keywords: str | None = Field(default=None, max_length=255) + target_journal: str | None = Field(default=None, max_length=255) + + @field_validator("title") + @classmethod + def _require_title(cls, value: str) -> str: + text = value.strip() + if not text: + raise ValueError("论文标题不能为空") + return text + + @field_validator("author", "target_journal") + @classmethod + def _blank_to_none(cls, value: str | None) -> str | None: + """Treat a whitespace-only author or journal as "not set".""" + if value is None: + return None + return value.strip() or None + + @field_validator("keywords") + @classmethod + def _normalise_keywords(cls, value: str | None) -> str | None: + return normalize_keywords(value) + + +class PaperCreate(PaperBase): + """Payload for ``POST /papers``. + + Creation is metadata only: the structure comes from the chosen template, + read live at render time, so there is nothing to copy and no empty rows to + materialise. The paper appears with its whole outline and no content. + """ + + +class PaperUpdate(BaseModel): + """Payload for ``PATCH /papers/{id}`` — every part optional. + + Changing ``template_id`` is the template switch: it re-shapes the document + in one write. Sentence rows are untouched, which is why switching back + restores the previous layout exactly. + """ + + title: str | None = Field(default=None, min_length=1, max_length=255) + template_id: int | None = None + abstract: str | None = None + author: str | None = Field(default=None, max_length=255) + status: PaperStatus | None = None + keywords: str | None = Field(default=None, max_length=255) + target_journal: str | None = Field(default=None, max_length=255) + + @field_validator("title") + @classmethod + def _require_title(cls, value: str | None) -> str | None: + if value is None: + return None + text = value.strip() + if not text: + raise ValueError("论文标题不能为空") + return text + + @field_validator("author", "target_journal") + @classmethod + def _blank_to_none(cls, value: str | None) -> str | None: + if value is None: + return None + return value.strip() or None + + @field_validator("keywords") + @classmethod + def _normalise_keywords(cls, value: str | None) -> str | None: + return normalize_keywords(value) + + +class PaperListItem(BaseModel): + """A paper as it appears in the table, without its content.""" + + id: int + title: str + template_id: int | None + #: The template's name, or ``None`` when no template is chosen. + template_name: str | None + author: str | None + status: PaperStatus + keywords: str | None + target_journal: str | None + #: How many sentences have been written. + sentence_count: int + #: How many distinct paragraphs those sentences occupy. + paragraph_count: int + #: How many paragraphs the current template defines — the denominator of + #: the progress read-out. 0 when no template is chosen. + template_paragraph_count: int + created_at: datetime + updated_at: datetime + + +class PaperRead(PaperListItem): + """A paper with everything the table does not need.""" + + abstract: str | None + + +class PaperDocumentRead(BaseModel): + """The whole paper as it reads: structure, content, citations, warnings. + + Assembled by :func:`app.crud.paper.build_document` in a single pass, so a + client renders from one response and never has to stitch arrays together. + """ + + paper: PaperRead + #: Paragraphs in ascending position order. Every position that either the + #: template or a sentence mentions appears exactly once, so an empty + #: paragraph is still a paragraph — the structure survives with no content. + paragraphs: list[ParagraphRead] + #: Citations in reading order, numbered to match the in-text markers. + citations: list[PaperCitationRead] + #: Anything the reader should be told about the document as a whole, e.g. + #: a template that places two fields at the same position. + warnings: list[str] = Field(default_factory=list) + + +class ParagraphListRead(BaseModel): + """Response of a paragraph read: the paragraph plus the paper it belongs to. + + The editor needs to know which paper it is editing and where else it could + move the paragraph to, and both are already loaded by the document query. + """ + + paper_id: int + paper_title: str + template_id: int | None + paragraph: ParagraphRead + + +__all__ = [ + "CitationInput", + "CitationRead", + "KEYWORD_SEPARATORS", + "PaperBase", + "PaperCitationRead", + "PaperCreate", + "PaperDocumentRead", + "PaperListItem", + "PaperRead", + "PaperStatus", + "PaperUpdate", + "ParagraphListRead", + "ParagraphRead", + "ParagraphUpdate", + "SentenceCreate", + "SentenceRead", + "SentenceUpdate", + "fold_whitespace", + "normalize_keywords", +] diff --git a/backend/scripts/smoke_papers.py b/backend/scripts/smoke_papers.py new file mode 100644 index 0000000..1ccead7 --- /dev/null +++ b/backend/scripts/smoke_papers.py @@ -0,0 +1,328 @@ +"""End-to-end smoke test for the paper feature, against a running API. + +Run it while the backend is up:: + + .venv/bin/python scripts/smoke_papers.py + +It walks the whole writing loop — create a paper on a template, verify the +empty structure, fill a paragraph with sentences and citations, read the +document back, switch the template, and check that unmatched positions survive +under an unset heading — then deletes everything it created. Exits non-zero on +the first failed expectation, so it is usable as a gate. +""" + +from __future__ import annotations + +import json +import sys +import urllib.error +import urllib.parse +import urllib.request +from typing import Any + +BASE = "http://127.0.0.1:8000/api" + +_checks = 0 + + +def call(method: str, path: str, body: Any = None, expect: int = 200) -> Any: + """Perform one API call and assert its status code. + + The path is quoted so a Chinese search keyword travels as UTF-8 percent + escapes instead of blowing up the ASCII request line. + """ + data = json.dumps(body).encode() if body is not None else None + url = BASE + urllib.parse.quote(path, safe="/?&=%") + request = urllib.request.Request( + url, + data=data, + method=method, + headers={"Content-Type": "application/json"}, + ) + try: + with urllib.request.urlopen(request, timeout=20) as response: + status = response.status + payload = response.read().decode() + except urllib.error.HTTPError as error: + status = error.code + payload = error.read().decode() + + if status != expect: + raise AssertionError(f"{method} {path} -> {status}, expected {expect}: {payload}") + return json.loads(payload) if payload else None + + +def check(label: str, condition: bool, detail: str = "") -> None: + """Assert one expectation and report it.""" + global _checks + _checks += 1 + if not condition: + raise AssertionError(f"{label}: {detail}") + print(f" ok {label}") + + +def main() -> int: + created_papers: list[int] = [] + + try: + print("1. pick two templates with different paragraph positions") + templates = call("GET", "/templates?page_size=50")["items"] + check("templates exist", len(templates) >= 2, f"got {len(templates)}") + wide = max(templates, key=lambda item: item["field_count"]) + narrow = min(templates, key=lambda item: item["field_count"]) + wide_full = call("GET", f"/templates/{wide['id']}") + narrow_full = call("GET", f"/templates/{narrow['id']}") + print(f" wide={wide['name']} ({wide['field_count']}), narrow={narrow['name']}") + + print("2. create a paper") + paper = call( + "POST", + "/papers", + { + "title": "冒烟测试论文", + "template_id": wide["id"], + "author": "测试作者", + "status": "writing", + "keywords": "关键词A,关键词B;关键词A", + "target_journal": "测试期刊", + "abstract": "用于验证论文功能的临时数据。", + }, + expect=201, + ) + created_papers.append(paper["id"]) + check("keywords normalised", paper["keywords"] == "关键词A, 关键词B", paper["keywords"]) + check( + "paragraph denominator counts the template", + paper["template_paragraph_count"] == len({f["sort"] for f in wide_full["fields"]}), + str(paper["template_paragraph_count"]), + ) + check("no sentences yet", paper["sentence_count"] == 0) + + print("3. the empty paper already has its structure") + document = call("GET", f"/papers/{paper['id']}/document") + check( + "every template position is a paragraph", + len(document["paragraphs"]) == len({f["sort"] for f in wide_full["fields"]}), + str(len(document["paragraphs"])), + ) + check("all paragraphs empty", all(not p["sentences"] for p in document["paragraphs"])) + check("all paragraphs matched", all(p["matched"] for p in document["paragraphs"])) + check("no citations", document["citations"] == []) + + first_sort = document["paragraphs"][0]["paper_template_filed_sort"] + + print("4. write a paragraph: two sentences, citations on both") + document = call( + "PUT", + f"/papers/{paper['id']}/paragraphs/{first_sort}", + { + "sentences": [ + { + "content": " 第一句话,用于验证空白折叠。 ", + "citations": [{"reference_id": 7, "quote": "被引用的第一段内容"}], + }, + { + "content": "第二句话。", + "citations": [ + {"reference_id": 9, "quote": "引文甲"}, + {"quote": "只有引用内容、暂时没有引用 id"}, + ], + }, + ] + }, + ) + paragraph = document["paragraphs"][0] + check("two sentences", len(paragraph["sentences"]) == 2, str(len(paragraph["sentences"]))) + check("content trimmed", paragraph["sentences"][0]["content"] == "第一句话,用于验证空白折叠。") + check( + "sorts numbered from 1", + [s["sort"] for s in paragraph["sentences"]] == [1, 2], + str([s["sort"] for s in paragraph["sentences"]]), + ) + check("citations numbered", [c["sort"] for c in paragraph["sentences"][1]["citations"]] == [1, 2]) + check("three citations in the paper", len(document["citations"]) == 3) + check( + "citation index is reading order", + [c["index"] for c in document["citations"]] == [1, 2, 3], + str([c["index"] for c in document["citations"]]), + ) + check("sentence template stamped", paragraph["sentences"][0]["template_id"] == wide["id"]) + + print("5. explicit sorts decide the order, not the order they were sent") + second_sort = document["paragraphs"][1]["paper_template_filed_sort"] + reordered = call( + "PUT", + f"/papers/{paper['id']}/paragraphs/{second_sort}", + { + "sentences": [ + {"sort": 20, "content": "排序在后的句子。"}, + {"sort": 10, "content": "排序在前的句子。"}, + ] + }, + ) + second = next( + p + for p in reordered["paragraphs"] + if p["paper_template_filed_sort"] == second_sort + ) + check( + "ascending sort wins over arrival order", + [s["content"] for s in second["sentences"]] + == ["排序在前的句子。", "排序在后的句子。"], + str([s["content"] for s in second["sentences"]]), + ) + check("explicit sorts are stored as sent", [s["sort"] for s in second["sentences"]] == [10, 20]) + # Clear it again so the counts asserted below stay meaningful. + call( + "PUT", + f"/papers/{paper['id']}/paragraphs/{second_sort}", + {"sentences": []}, + ) + + print("6. reject a citation with no quote") + bad = call( + "PUT", + f"/papers/{paper['id']}/paragraphs/{first_sort}", + {"sentences": [{"content": "x", "citations": [{"reference_id": 3, "quote": " "}]}]}, + expect=422, + ) + check("citation without content refused", "detail" in bad) + + print("7. a sentence with no paragraph in the template survives a switch") + # A position neither template defines, so the check means the same + # thing before and after the switch below. + orphan_sort = ( + max(f["sort"] for f in wide_full["fields"] + narrow_full["fields"]) + 5 + ) + call( + "POST", + f"/papers/{paper['id']}/sentences", + {"paper_template_filed_sort": orphan_sort, "content": "新模板里没有这个位置的句子。"}, + expect=201, + ) + document = call("GET", f"/papers/{paper['id']}/document") + check( + "orphan position is rendered", + any( + p["paper_template_filed_sort"] == orphan_sort and not p["matched"] + for p in document["paragraphs"] + ), + ) + check("orphan heading has no name", next( + p for p in document["paragraphs"] if p["paper_template_filed_sort"] == orphan_sort + )["name"] is None) + + print("8. switch the template") + switched = call("PATCH", f"/papers/{paper['id']}", {"template_id": narrow["id"]}) + check("template changed", switched["template_id"] == narrow["id"]) + document = call("GET", f"/papers/{paper['id']}/document") + matched = [p for p in document["paragraphs"] if p["matched"]] + unmatched = [p for p in document["paragraphs"] if not p["matched"]] + check( + "structure follows the new template", + len(matched) == len({f["sort"] for f in narrow_full["fields"]}), + str(len(matched)), + ) + check("unmatched positions kept", len(unmatched) >= 1, str(len(unmatched))) + check( + "all sentences still present", + sum(len(p["sentences"]) for p in document["paragraphs"]) == 3, + ) + check( + "positions ascend", + [p["paper_template_filed_sort"] for p in document["paragraphs"]] + == sorted(p["paper_template_filed_sort"] for p in document["paragraphs"]), + ) + check( + "sentences re-stamped with the new template", + all( + s["template_id"] == narrow["id"] + for p in document["paragraphs"] + for s in p["sentences"] + ), + ) + + print("9. move one paragraph onto another position") + document = call( + "PUT", + f"/papers/{paper['id']}/paragraphs/{orphan_sort}", + {"sentences": [{"content": "搬到别处的一句。"}], "target_sort": first_sort}, + ) + target = next( + p for p in document["paragraphs"] if p["paper_template_filed_sort"] == first_sort + ) + check("moved sentence appended after the existing ones", len(target["sentences"]) == 3) + check( + "source position gone", + all(p["paper_template_filed_sort"] != orphan_sort for p in document["paragraphs"]), + ) + + print("10. paper list reports progress") + listing = call("GET", f"/papers?keyword=冒烟&status=writing") + check("found by keyword and status", listing["total"] == 1, str(listing["total"])) + row = listing["items"][0] + check("paragraph count", row["paragraph_count"] == 1, str(row["paragraph_count"])) + check("sentence count", row["sentence_count"] == 3, str(row["sentence_count"])) + + print("11. a template in use cannot be deleted") + conflict = call("DELETE", f"/templates/{narrow['id']}", expect=409) + check("template delete refused", "正被论文使用" in conflict["detail"], conflict["detail"]) + + print("12. single sentence edit and delete") + sentence_id = target["sentences"][0]["id"] + updated = call( + "PATCH", + f"/papers/{paper['id']}/sentences/{sentence_id}", + {"content": "改写后的一句话。", "sort": 5}, + ) + check("sentence updated", updated["content"] == "改写后的一句话。") + check("sentence sort updated", updated["sort"] == 5) + call("DELETE", f"/papers/{paper['id']}/sentences/{sentence_id}", expect=204) + document = call("GET", f"/papers/{paper['id']}/document") + check( + "sentence deleted", + sum(len(p["sentences"]) for p in document["paragraphs"]) == 2, + ) + + print("13. a paper with no template still works") + bare = call( + "POST", + "/papers", + {"title": "冒烟测试-无模板论文", "template_id": None, "status": "draft"}, + expect=201, + ) + created_papers.append(bare["id"]) + empty_doc = call("GET", f"/papers/{bare['id']}/document") + check("no template means no structure", empty_doc["paragraphs"] == []) + call( + "POST", + f"/papers/{bare['id']}/sentences", + {"paper_template_filed_sort": 1, "content": "没有模板时写下的第一句。"}, + expect=201, + ) + bare_doc = call("GET", f"/papers/{bare['id']}/document") + check("content creates its own paragraph", len(bare_doc["paragraphs"]) == 1) + check("that paragraph has no heading", bare_doc["paragraphs"][0]["matched"] is False) + check("and the reader is warned", len(bare_doc["warnings"]) >= 1) + call("DELETE", f"/papers/{bare['id']}", expect=204) + created_papers.remove(bare["id"]) + + print("14. delete the paper") + call("DELETE", f"/papers/{paper['id']}", expect=204) + created_papers.clear() + check("gone from the list", call("GET", "/papers")["total"] == 0) + + print(f"\nall {_checks} checks passed") + return 0 + finally: + # Never leave test rows behind, even on a failed expectation. + for paper_id in created_papers: + try: + call("DELETE", f"/papers/{paper_id}", expect=204) + print(f"cleaned up paper {paper_id}") + except Exception as error: # noqa: BLE001 - cleanup must not mask the failure + print(f"cleanup of paper {paper_id} failed: {error}", file=sys.stderr) + + +if __name__ == "__main__": + sys.exit(main())