Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6aea6a4e2c | ||
|
|
b10bf3e8c7 | ||
|
|
8c3273a0fc | ||
|
|
aec226af1a |
@@ -121,6 +121,9 @@ func main() {
|
|||||||
|
|
||||||
// Create library service
|
// Create library service
|
||||||
libraryService := services.NewLibraryService(queries)
|
libraryService := services.NewLibraryService(queries)
|
||||||
|
// The progress service verifies submitted anchors against the book
|
||||||
|
// itself — it needs to resolve library-relative file paths.
|
||||||
|
progressService.SetMediaPathResolver(libraryService)
|
||||||
|
|
||||||
// Sync Go AllowedExtensions into DB so API clients see correct extensions
|
// Sync Go AllowedExtensions into DB so API clients see correct extensions
|
||||||
libraryService.SyncAllowedExtensions(context.Background())
|
libraryService.SyncAllowedExtensions(context.Background())
|
||||||
|
|||||||
+40
-108
@@ -394,6 +394,11 @@ Authorization: Bearer <token>
|
|||||||
|
|
||||||
## Reading Progress
|
## Reading Progress
|
||||||
|
|
||||||
|
Full field reference: [api/progress/](api/progress/) — and read the
|
||||||
|
[Position Contract](api/progress/position-contract.md) (verification,
|
||||||
|
healing, the OPF spine numbering hazard, offset currencies) before
|
||||||
|
writing a client.
|
||||||
|
|
||||||
### Get Reading Progress
|
### Get Reading Progress
|
||||||
|
|
||||||
```http
|
```http
|
||||||
@@ -405,57 +410,56 @@ Authorization: Bearer <token>
|
|||||||
|
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
|
"id": "uuid",
|
||||||
"media_item_id": "uuid",
|
"media_item_id": "uuid",
|
||||||
"user_id": "uuid",
|
"user_id": "uuid",
|
||||||
"current_page": 45,
|
"percentage": 0.045,
|
||||||
"total_pages": 200,
|
"epubcfi": "epubcfi(/6/4!/4/8[_idContainer003]/58/1:598)",
|
||||||
"percentage": 0.225,
|
"context_text": "from day to day, but there was none in which ...",
|
||||||
"character_offset": 15432,
|
"character_offset": 16375,
|
||||||
"epubcfi": "epubcfi(/6/4/2:15)",
|
"current_page": null,
|
||||||
"chapter": 3,
|
"total_pages": null,
|
||||||
"chapter_progress": 0.5,
|
"chapter": null,
|
||||||
"last_read_at": "2026-01-31T10:00:00Z",
|
"chapter_progress": null,
|
||||||
"format_group": "reflowable",
|
"format_group": "reflowable",
|
||||||
"viewport_y": 0.12,
|
"total_characters": 592216,
|
||||||
"zoom_level": 1.0
|
"chapter_count": 1,
|
||||||
|
"last_read_at": "2026-09-26T14:52:34Z",
|
||||||
|
"last_sync_source": "koreader",
|
||||||
|
"css_selector": "body>div:nth-child(4)>p:nth-child(29)",
|
||||||
|
"anchor_href": "1984.xhtml",
|
||||||
|
"char_offset": 598
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
`css_selector`, `anchor_href` and `char_offset` are server-derived
|
||||||
|
restore handles, served for convertible reflowable books with a
|
||||||
|
resolvable anchor. `char_offset` is UTF-16 code units within the anchor
|
||||||
|
block's text; `character_offset` is a book-wide rune count. Resolve the
|
||||||
|
document by `anchor_href`, never by the CFI's spine step (OPF numbering
|
||||||
|
includes `linear="no"` items).
|
||||||
|
|
||||||
### Update Reading Progress
|
### Update Reading Progress
|
||||||
|
|
||||||
```http
|
```http
|
||||||
PUT /api/media-items/{media_id}/progress
|
PUT /api/media-items/{media_id}/progress
|
||||||
Authorization: Bearer <token>
|
Authorization: Bearer <token>
|
||||||
Content-Type: application/json
|
Content-Type: application/json
|
||||||
|
|
||||||
{
|
|
||||||
"source": "web",
|
|
||||||
"location": {
|
|
||||||
"percentage": 0.45678,
|
|
||||||
"epubcfi": "epubcfi(/6/4/2:15)",
|
|
||||||
"character": 15432,
|
|
||||||
"chapter": 3,
|
|
||||||
"page": 89,
|
|
||||||
"total_pages": 200
|
|
||||||
},
|
|
||||||
"device_metadata": {
|
|
||||||
"device_type": "web",
|
|
||||||
"user_agent": "Mozilla/5.0..."
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
```
|
||||||
|
|
||||||
**Response** (200):
|
|
||||||
|
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"sync_status": "success",
|
"percentage": 0.0415,
|
||||||
"progress_updated": true,
|
"context_text": "was at war with one of these Powers it was generally a",
|
||||||
"devices_notified": ["device-1", "device-2"],
|
"epubcfi": "epubcfi(/6/4!/4/8[_idContainer003]/62/1:456)"
|
||||||
"broadcast": true
|
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
All fields are optional. The server verifies the submission against the
|
||||||
|
book and heals it on mismatch; the response is the stored row after
|
||||||
|
verification. A `percentage` below 0.005 is ignored while stored
|
||||||
|
progress exceeds 0.01 (first-page anti-clobber).
|
||||||
|
|
||||||
### Delete Reading Progress
|
### Delete Reading Progress
|
||||||
|
|
||||||
```http
|
```http
|
||||||
@@ -1297,82 +1301,10 @@ Authorization: Bearer <device_token>
|
|||||||
|
|
||||||
## Universal Progress
|
## Universal Progress
|
||||||
|
|
||||||
### Get Universal Progress
|
Universal progress is the media-item progress — one row per (user,
|
||||||
|
media item) shared by every device. See [Reading Progress](#reading-progress)
|
||||||
```http
|
and the [Position Contract](api/progress/position-contract.md). There
|
||||||
GET /api/progress/{book_uuid}
|
are no separate `/api/progress/:id` GET/POST endpoints.
|
||||||
Authorization: Bearer <token>
|
|
||||||
```
|
|
||||||
|
|
||||||
**Response** (200):
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"book_id": "book-uuid",
|
|
||||||
"format_group": "reflowable",
|
|
||||||
"universal_progress": 0.45678,
|
|
||||||
"location_references": {
|
|
||||||
"percentage": 0.45678,
|
|
||||||
"epubcfi": "epubcfi(/6/4/2:15)",
|
|
||||||
"character": 15432,
|
|
||||||
"chapter": 3,
|
|
||||||
"chapter_progress": 0.234,
|
|
||||||
"viewport_y": 0.12
|
|
||||||
},
|
|
||||||
"device_progress": {
|
|
||||||
"koreader": {
|
|
||||||
"percentage": 0.45678,
|
|
||||||
"last_sync": "2026-01-30T20:00:00Z"
|
|
||||||
},
|
|
||||||
"kobo": {
|
|
||||||
"percentage": 45.6,
|
|
||||||
"last_sync": "2026-01-30T19:55:00Z"
|
|
||||||
},
|
|
||||||
"web": {
|
|
||||||
"display_page": 89,
|
|
||||||
"total_pages": 200,
|
|
||||||
"last_sync": "2026-01-30T20:05:00Z"
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"annotations": {
|
|
||||||
"highlights": [...],
|
|
||||||
"notes": [...],
|
|
||||||
"bookmarks": [...]
|
|
||||||
},
|
|
||||||
"conflicts": [
|
|
||||||
{
|
|
||||||
"id": "conflict-uuid",
|
|
||||||
"type": "progress",
|
|
||||||
"resolved": false,
|
|
||||||
"sources": ["koreader", "kobo"]
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
### Update Universal Progress
|
|
||||||
|
|
||||||
```http
|
|
||||||
POST /api/progress/{book_uuid}
|
|
||||||
Authorization: Bearer <token>
|
|
||||||
Content-Type: application/json
|
|
||||||
|
|
||||||
{
|
|
||||||
"source": "web|koreader|kobo|mobile",
|
|
||||||
"location": {
|
|
||||||
"percentage": 0.45678,
|
|
||||||
"epubcfi": "epubcfi(/6/4/2:15)",
|
|
||||||
"character": 15432,
|
|
||||||
"chapter": 3,
|
|
||||||
"page": 89,
|
|
||||||
"total_pages": 200
|
|
||||||
},
|
|
||||||
"device_metadata": {
|
|
||||||
"device_type": "web",
|
|
||||||
"user_agent": "..."
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Conflicts
|
## Conflicts
|
||||||
|
|
||||||
|
|||||||
@@ -117,10 +117,14 @@ See [Media Item Operations](media-items/)
|
|||||||
|
|
||||||
## Reading Progress
|
## Reading Progress
|
||||||
|
|
||||||
See [Progress Tracking](progress/)
|
See [Progress Tracking](progress/) — in particular the
|
||||||
|
[Position Contract](progress/position-contract.md) (verification,
|
||||||
|
healing, the OPF spine numbering hazard, and offset currencies) before
|
||||||
|
writing a client.
|
||||||
|
|
||||||
- GET /api/progress/:id - Get universal progress
|
- GET /api/media-items/:id/progress - Get reading progress + restore handles
|
||||||
- POST /api/progress/:id - Update universal progress
|
- PUT /api/media-items/:id/progress - Submit reading progress (verified/healed server-side)
|
||||||
|
- DELETE /api/media-items/:id/progress - Delete reading progress
|
||||||
- GET /api/progress/:id/history - Get progress history
|
- GET /api/progress/:id/history - Get progress history
|
||||||
|
|
||||||
## Notes & Highlights
|
## Notes & Highlights
|
||||||
|
|||||||
@@ -1,9 +1,12 @@
|
|||||||
# Get Metadata
|
# Get Metadata
|
||||||
|
|
||||||
Get metadata for a book from KOReader device.
|
Get a book's stored progress and annotations for a KOReader device —
|
||||||
|
the pull half of the device sync. The reference client calls this after
|
||||||
|
linking a book via [Resolve Book](resolve_book.md) and navigates to the
|
||||||
|
returned position.
|
||||||
|
|
||||||
**Endpoint**: `GET /api/sync/koreader/metadata/:uuid`
|
**Endpoint**: `GET /api/sync/koreader/metadata/:uuid`
|
||||||
**Auth**: Required (Device authentication)
|
**Auth**: Required (Device authentication — `Authorization: Bearer {device_token}`)
|
||||||
|
|
||||||
## Path Parameters
|
## Path Parameters
|
||||||
|
|
||||||
@@ -11,41 +14,55 @@ Get metadata for a book from KOReader device.
|
|||||||
| --------- | ------------- | -------- | ----------- |
|
| --------- | ------------- | -------- | ----------- |
|
||||||
| uuid | string (UUID) | Yes | Book UUID |
|
| uuid | string (UUID) | Yes | Book UUID |
|
||||||
|
|
||||||
## Device Authentication
|
|
||||||
|
|
||||||
This endpoint requires device authentication (not user JWT). Devices authenticate using their device credentials.
|
|
||||||
|
|
||||||
## Request Headers
|
|
||||||
|
|
||||||
| Header | Type | Required | Description |
|
|
||||||
| ------------ | ------ | -------- | ------------------------- |
|
|
||||||
| X-Device-ID | string | Yes | Device UUID |
|
|
||||||
| X-Device-Key | string | Yes | Device authentication key |
|
|
||||||
|
|
||||||
### Example Request
|
### Example Request
|
||||||
|
|
||||||
```http
|
```http
|
||||||
GET /api/sync/koreader/metadata/550e8400-e29b-41d4-a716-446655440000
|
GET /api/sync/koreader/metadata/774641f9-317b-4087-8e04-53bb4392ae56
|
||||||
X-Device-ID: 550e8400-e29b-41d4-a716-446655440000
|
Authorization: Bearer {device_token}
|
||||||
X-Device-Key: device-auth-key
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Response (200 OK)
|
## Response (200 OK)
|
||||||
|
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"id": "uuid",
|
"uuid": "774641f9-317b-4087-8e04-53bb4392ae56",
|
||||||
"title": "Book Title",
|
"title": "1984",
|
||||||
"authors": ["Author Name"],
|
"author": "George Orwell",
|
||||||
"path": "/path/to/book.epub",
|
"progress": {
|
||||||
"file_size": 1234567,
|
"percentage": 0.045,
|
||||||
"modified_at": "2026-02-08T10:00:00Z"
|
"koreader_xpointer": "/body/DocFragment[1]/body/p[29]/text().598",
|
||||||
|
"chapter": null,
|
||||||
|
"chapter_progress": null,
|
||||||
|
"page": null,
|
||||||
|
"total_pages": null
|
||||||
|
},
|
||||||
|
"annotations": {
|
||||||
|
"highlights": []
|
||||||
|
}
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Progress Object
|
||||||
|
|
||||||
|
| Field | Type | Description |
|
||||||
|
| ----- | ---- | ----------- |
|
||||||
|
| `percentage` | float | Stored position as a book fraction. |
|
||||||
|
| `koreader_xpointer` | string | The stored canonical position converted back to a CRE xpointer (UTF-16 `text().N` offset). **The device should navigate to this.** Reflowable books only. |
|
||||||
|
| `epubcfi` | string | The stored canonical CFI, when the conversion to a CRE xpointer is unavailable. Fallback after `koreader_xpointer`. |
|
||||||
|
| `character` | int | Book-wide rune offset (internal currency). |
|
||||||
|
| `chapter`, `chapter_progress` | int, float | Chapter position when known. |
|
||||||
|
| `page`, `total_pages` | int | Fixed-layout page position — the canonical locator for image-based books (CFI/xpointer are omitted for them). |
|
||||||
|
|
||||||
|
`progress` is `null` when the book has no stored progress.
|
||||||
|
|
||||||
|
The `annotations` object carries device-format highlights/bookmarks/notes
|
||||||
|
synced from other clients; its presence depends on annotation sync being
|
||||||
|
enabled.
|
||||||
|
|
||||||
## Error Responses
|
## Error Responses
|
||||||
|
|
||||||
| Code | Description |
|
| Code | Description |
|
||||||
| ---- | ---------------------------- |
|
| ---- | ----------- |
|
||||||
|
| 400 | Invalid book UUID |
|
||||||
| 401 | Device authentication failed |
|
| 401 | Device authentication failed |
|
||||||
| 404 | Book or device not found |
|
| 404 | Book not found |
|
||||||
|
|||||||
@@ -1,63 +1,87 @@
|
|||||||
# Sync Progress
|
# Sync Progress
|
||||||
|
|
||||||
Sync reading progress from KOReader device.
|
Push reading progress from a KOReader device.
|
||||||
|
|
||||||
**Endpoint**: `POST /api/sync/koreader/progress`
|
**Endpoint**: `POST /api/sync/koreader/progress`
|
||||||
**Auth**: Required (Device authentication)
|
**Auth**: Required (Device authentication — `Authorization: Bearer {device_token}`)
|
||||||
|
|
||||||
## Device Authentication
|
This is the device-native tier of the [Position
|
||||||
|
Contract](../progress/position-contract.md): the KOReader payload carries
|
||||||
This endpoint requires device authentication (not user JWT). Devices authenticate using their device credentials.
|
a CRE xpointer and the server converts it to the canonical standard CFI,
|
||||||
|
verifies it against the submitted `context_text`, and heals it on
|
||||||
|
mismatch — exactly like every other client.
|
||||||
|
|
||||||
## Request Body
|
## Request Body
|
||||||
|
|
||||||
| Field | Type | Required | Description |
|
| Field | Type | Required | Description |
|
||||||
| --------- | ------------- | -------- | ------------------------- |
|
| --------- | ------ | -------- | ----------- |
|
||||||
| device_id | string (UUID) | Yes | Device UUID |
|
| `books` | array | Yes | One book object (the reference client sends a single-element array). |
|
||||||
| progress | array | Yes | Array of progress objects |
|
| `sync_mode`| string | No | `immediate` (default) or `manual`. |
|
||||||
|
|
||||||
### Progress Object
|
### Book Object
|
||||||
|
|
||||||
| Field | Type | Required | Description |
|
| Field | Type | Required | Description |
|
||||||
| ----------- | ------- | -------- | ---------------------------------- |
|
| ----- | ---- | -------- | ----------- |
|
||||||
| book | string | Yes | Book identifier (filename or UUID) |
|
| `uuid` | string (UUID) | No | Bookhoard UUID, once the device has linked the book via [Resolve Book](resolve_book.md). |
|
||||||
| percent | float | Yes | Progress percentage (0-100) |
|
| `sha256` | string | Yes | File content hash (64 hex chars) — the primary book identity. |
|
||||||
| page | integer | No | Current page number |
|
| `title` | string | No | Document title. |
|
||||||
| total_pages | integer | No | Total pages in document |
|
| `authors` | array | No | Author names. |
|
||||||
| date_read | string | No | ISO 8601 timestamp of last read |
|
| `percentage` | float | Yes | Position as a fraction of the book (0..1). |
|
||||||
| updated_at | string | Yes | ISO 8601 timestamp |
|
| `context_text` | string | No | Up to 100 whitespace-normalized chars from the current position — enables the server's verification/healing. Strongly recommended. |
|
||||||
|
| `page` | int | No | Current page (fixed-layout books). |
|
||||||
|
| `total_pages` | int | No | Page count (fixed-layout books). |
|
||||||
|
| `epubcfi` | string | Reflowable only | **A CRE xpointer** (`/body/DocFragment[N]/body/...`), not a CFI — the field name is historical. Fixed-layout books must omit it and carry their position in `page`/`total_pages`. |
|
||||||
|
| `file_path` | string | No | Device-local file path (informational). |
|
||||||
|
| `device_info` | object | No | `{ koreader_version, device_model }`. |
|
||||||
|
|
||||||
### Example Request
|
### Example Request
|
||||||
|
|
||||||
|
```http
|
||||||
|
POST /api/sync/koreader/progress
|
||||||
|
Authorization: Bearer {device_token}
|
||||||
|
Content-Type: application/json
|
||||||
|
```
|
||||||
|
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"device_id": "550e8400-e29b-41d4-a716-446655440000",
|
"books": [
|
||||||
"progress": [
|
|
||||||
{
|
{
|
||||||
"book": "book.epub",
|
"uuid": "774641f9-317b-4087-8e04-53bb4392ae56",
|
||||||
"percent": 75.5,
|
"sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855",
|
||||||
"page": 150,
|
"title": "1984",
|
||||||
"total_pages": 200,
|
"authors": ["George Orwell"],
|
||||||
"date_read": "2026-02-08T10:00:00Z",
|
"percentage": 0.045,
|
||||||
"updated_at": "2026-02-08T10:00:00Z"
|
"context_text": "from day to day, but there was none in which Goldstein was not the principal figure. He was the prim",
|
||||||
|
"page": 14,
|
||||||
|
"total_pages": 311,
|
||||||
|
"epubcfi": "/body/DocFragment[1]/body/p[29]/text().598",
|
||||||
|
"device_info": {
|
||||||
|
"koreader_version": "v2026.07.1",
|
||||||
|
"device_model": "emulator"
|
||||||
}
|
}
|
||||||
]
|
}
|
||||||
|
],
|
||||||
|
"sync_mode": "immediate"
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
## Response (200 OK)
|
## Response (202 Accepted)
|
||||||
|
|
||||||
```json
|
```json
|
||||||
{
|
{
|
||||||
"message": "Progress synced successfully",
|
"sync_status": "ok",
|
||||||
"synced_count": 1
|
"books_synced": 1,
|
||||||
|
"timestamp": "2026-09-26T21:25:09Z"
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Per-book results and any detected sync conflicts are carried in
|
||||||
|
`book_results` and `conflicts` when present.
|
||||||
|
|
||||||
## Error Responses
|
## Error Responses
|
||||||
|
|
||||||
| Code | Description |
|
| Code | Description |
|
||||||
| ---- | ---------------------------- |
|
| ---- | ----------- |
|
||||||
|
| 400 | Invalid request format |
|
||||||
| 401 | Device authentication failed |
|
| 401 | Device authentication failed |
|
||||||
| 400 | Invalid request data |
|
| 500 | Database error |
|
||||||
| 404 | Device not found |
|
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
# Delete Media Item Progress
|
||||||
|
|
||||||
|
Delete the stored reading progress for a media item.
|
||||||
|
|
||||||
|
**Endpoint**: `DELETE /api/media-items/:id/progress`
|
||||||
|
**Auth**: Required (Bearer token)
|
||||||
|
|
||||||
|
## Path Parameters
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Description |
|
||||||
|
| --------- | ------------- | -------- | ---------------- |
|
||||||
|
| id | string (UUID) | Yes | Media item UUID |
|
||||||
|
|
||||||
|
### Example Request
|
||||||
|
|
||||||
|
```http
|
||||||
|
DELETE /api/media-items/774641f9-317b-4087-8e04-53bb4392ae56/progress
|
||||||
|
Authorization: Bearer eyJhbGciOiJIUzI1NiIs...
|
||||||
|
```
|
||||||
|
|
||||||
|
## Response (200 OK)
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"message": "reading progress deleted"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Error Responses
|
||||||
|
|
||||||
|
| Code | Description |
|
||||||
|
| ---- | ----------- |
|
||||||
|
| 400 | Invalid media item id |
|
||||||
|
| 401 | Invalid or expired token |
|
||||||
|
| 500 | Database error |
|
||||||
@@ -1,36 +0,0 @@
|
|||||||
# Delete Reading Progress
|
|
||||||
|
|
||||||
Delete reading progress for a media item.
|
|
||||||
|
|
||||||
**Endpoint**: `DELETE /api/media-items/{media_id}/progress`
|
|
||||||
**Auth**: Required
|
|
||||||
|
|
||||||
## Path Parameters
|
|
||||||
|
|
||||||
| Parameter | Type | Required | Description |
|
|
||||||
| --------- | ------ | -------- | --------------- |
|
|
||||||
| media_id | string | Yes | Media item UUID |
|
|
||||||
|
|
||||||
## Request Headers
|
|
||||||
|
|
||||||
| Header | Type | Required | Description |
|
|
||||||
| ------------- | ------ | -------- | ------------ |
|
|
||||||
| Authorization | string | Yes | Bearer token |
|
|
||||||
|
|
||||||
### Example Request
|
|
||||||
|
|
||||||
```http
|
|
||||||
DELETE /api/media-items/uuid/progress
|
|
||||||
Authorization: Bearer eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9...
|
|
||||||
```
|
|
||||||
|
|
||||||
## Response (204 No Content)
|
|
||||||
|
|
||||||
Progress deleted successfully.
|
|
||||||
|
|
||||||
## Error Responses
|
|
||||||
|
|
||||||
| Code | Description |
|
|
||||||
| ---- | ------------------------ |
|
|
||||||
| 401 | Invalid or expired token |
|
|
||||||
| 404 | Media item not found |
|
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
# Get Media Item Progress
|
||||||
|
|
||||||
|
Get the stored reading progress for a media item, plus the
|
||||||
|
server-derived restore handles for reflowable books.
|
||||||
|
|
||||||
|
**Endpoint**: `GET /api/media-items/:id/progress`
|
||||||
|
**Auth**: Required (Bearer token)
|
||||||
|
|
||||||
|
See [Position Contract](position-contract.md) for the semantics of every
|
||||||
|
field — what is verified, what the currencies are, and how clients
|
||||||
|
should restore.
|
||||||
|
|
||||||
|
## Path Parameters
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Description |
|
||||||
|
| --------- | ------------- | -------- | --------------- |
|
||||||
|
| id | string (UUID) | Yes | Media item UUID |
|
||||||
|
|
||||||
|
### Example Request
|
||||||
|
|
||||||
|
```http
|
||||||
|
GET /api/media-items/774641f9-317b-4087-8e04-53bb4392ae56/progress
|
||||||
|
Authorization: Bearer eyJhbGciOiJIUzI1NiIs...
|
||||||
|
```
|
||||||
|
|
||||||
|
## Response (200 OK) — no progress stored
|
||||||
|
|
||||||
|
When the item has no progress row, an empty fixed-layout-style stub is
|
||||||
|
returned (not 404):
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"current_page": 0,
|
||||||
|
"total_pages": null
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Response (200 OK) — progress stored
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"id": "e8239659-0ed6-42ae-a946-8440d7655b42",
|
||||||
|
"media_item_id": "774641f9-317b-4087-8e04-53bb4392ae56",
|
||||||
|
"user_id": "1b64992e-3408-4d84-9e03-e2dc4950e1dd",
|
||||||
|
"current_page": null,
|
||||||
|
"total_pages": null,
|
||||||
|
"last_read_at": "2026-09-26T14:52:34.806446Z",
|
||||||
|
"percentage": 0.045,
|
||||||
|
"character_offset": 16375,
|
||||||
|
"epubcfi": "epubcfi(/6/4!/4/8[_idContainer003]/58/1:598)",
|
||||||
|
"chapter": null,
|
||||||
|
"chapter_progress": null,
|
||||||
|
"format_group": "reflowable",
|
||||||
|
"total_characters": 592216,
|
||||||
|
"chapter_count": 1,
|
||||||
|
"last_sync_device": "web",
|
||||||
|
"last_sync_source": "koreader",
|
||||||
|
"last_sync_timestamp": "2026-09-26T14:52:34.806446Z",
|
||||||
|
"css_selector": "body>div:nth-child(4)>p:nth-child(29)",
|
||||||
|
"anchor_href": "1984.xhtml",
|
||||||
|
"char_offset": 598,
|
||||||
|
"context_text": "from day to day, but there was none in which Goldstein was not the principal figure. He was the prim"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Field Reference
|
||||||
|
|
||||||
|
Base fields (always present when a row exists):
|
||||||
|
|
||||||
|
| Field | Type | Description |
|
||||||
|
| ----- | ---- | ----------- |
|
||||||
|
| `percentage` | float | Position as a fraction of the whole book (0..1). |
|
||||||
|
| `epubcfi` | string | The stored canonical standard CFI. **Spine steps index the OPF spine as written, including `linear="no"` items — do not resolve them against a readium reading order.** See the [Position Contract](position-contract.md#spine-numbering-hazard--read-this-before-parsing-a-stored-cfi). |
|
||||||
|
| `context_text` | string | The stored verification context (≤100 whitespace-normalized chars from the anchor). |
|
||||||
|
| `character_offset` | int | Book-wide rune offset. Internal currency — consistent with `total_characters`. |
|
||||||
|
| `current_page`, `total_pages` | int | Fixed-layout page position (null for reflowable). |
|
||||||
|
| `chapter`, `chapter_progress` | int, float | Chapter index and within-chapter fraction, when known. |
|
||||||
|
| `format_group` | string | `reflowable`, `fixed_layout`, `comic_archive`, … Gates the restore handles. |
|
||||||
|
| `total_characters`, `chapter_count` | int | Book metrics, for client-side fraction math. |
|
||||||
|
| `last_sync_device`, `last_sync_source`, `last_sync_timestamp` | — | Which client last wrote the row. |
|
||||||
|
|
||||||
|
Restore handles (conditional — served only for convertible reflowable
|
||||||
|
books whose stored anchor re-resolves at GET time):
|
||||||
|
|
||||||
|
| Field | Type | Description |
|
||||||
|
| ----- | ---- | ----------- |
|
||||||
|
| `anchor_href` | string | The spine document containing the anchor (e.g. `1984.xhtml`). **Resolve your resource by this**, never by the CFI's spine step. |
|
||||||
|
| `css_selector` | string | Body-relative chain to the anchor's block element. |
|
||||||
|
| `char_offset` | int | Anchor offset within the block's concatenated text, in UTF-16 code units. |
|
||||||
|
| `epubcfi` | string | Re-served (possibly healed) canonical CFI — present whenever re-verification produced one. |
|
||||||
|
|
||||||
|
## Error Responses
|
||||||
|
|
||||||
|
| Code | Description |
|
||||||
|
| ---- | ----------- |
|
||||||
|
| 400 | Invalid media item id |
|
||||||
|
| 401 | Invalid or expired token |
|
||||||
|
| 500 | Database error |
|
||||||
@@ -1,52 +0,0 @@
|
|||||||
# Get Reading Progress
|
|
||||||
|
|
||||||
Retrieve reading progress for a specific media item.
|
|
||||||
|
|
||||||
**Endpoint**: `GET /api/media-items/{media_id}/progress`
|
|
||||||
**Auth**: Required
|
|
||||||
|
|
||||||
## Path Parameters
|
|
||||||
|
|
||||||
| Parameter | Type | Required | Description |
|
|
||||||
| --------- | ------ | -------- | --------------- |
|
|
||||||
| media_id | string | Yes | Media item UUID |
|
|
||||||
|
|
||||||
## Request Headers
|
|
||||||
|
|
||||||
| Header | Type | Required | Description |
|
|
||||||
| ------------- | ------ | -------- | ------------ |
|
|
||||||
| Authorization | string | Yes | Bearer token |
|
|
||||||
|
|
||||||
### Example Request
|
|
||||||
|
|
||||||
```http
|
|
||||||
GET /api/media-items/uuid/progress
|
|
||||||
Authorization: Bearer eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9...
|
|
||||||
```
|
|
||||||
|
|
||||||
## Response (200 OK)
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"media_item_id": "uuid",
|
|
||||||
"user_id": "uuid",
|
|
||||||
"current_page": 45,
|
|
||||||
"total_pages": 200,
|
|
||||||
"percentage": 0.225,
|
|
||||||
"character_offset": 15432,
|
|
||||||
"epubcfi": "epubcfi(/6/4/2:15)",
|
|
||||||
"chapter": 3,
|
|
||||||
"chapter_progress": 0.5,
|
|
||||||
"last_read_at": "2026-01-31T10:00:00Z",
|
|
||||||
"format_group": "reflowable",
|
|
||||||
"viewport_y": 0.12,
|
|
||||||
"zoom_level": 1.0
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Error Responses
|
|
||||||
|
|
||||||
| Code | Description |
|
|
||||||
| ---- | ------------------------ |
|
|
||||||
| 401 | Invalid or expired token |
|
|
||||||
| 404 | Media item not found |
|
|
||||||
@@ -1,50 +0,0 @@
|
|||||||
# Get Universal Progress
|
|
||||||
|
|
||||||
Get universal (device-agnostic) reading progress for a media item.
|
|
||||||
|
|
||||||
**Endpoint**: `GET /api/progress/:id`
|
|
||||||
**Auth**: Required
|
|
||||||
|
|
||||||
## Path Parameters
|
|
||||||
|
|
||||||
| Parameter | Type | Required | Description |
|
|
||||||
| --------- | ------------- | -------- | --------------- |
|
|
||||||
| id | string (UUID) | Yes | Media item UUID |
|
|
||||||
|
|
||||||
## Request Headers
|
|
||||||
|
|
||||||
| Header | Type | Required | Description |
|
|
||||||
| ------------- | ------ | -------- | ------------ |
|
|
||||||
| Authorization | string | Yes | Bearer token |
|
|
||||||
|
|
||||||
### Example Request
|
|
||||||
|
|
||||||
```http
|
|
||||||
GET /api/progress/550e8400-e29b-41d4-a716-446655440000
|
|
||||||
Authorization: Bearer eyJhbGciOiJIUzI1NiIsInR5cCI6IkpXVCJ9...
|
|
||||||
```
|
|
||||||
|
|
||||||
## Response (200 OK)
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"media_item_id": "uuid",
|
|
||||||
"percentage": 75.5,
|
|
||||||
"position": 1234,
|
|
||||||
"page": 150,
|
|
||||||
"total_pages": 200,
|
|
||||||
"finished": false,
|
|
||||||
"updated_at": "2026-02-08T10:00:00Z",
|
|
||||||
"device": {
|
|
||||||
"id": "device-uuid",
|
|
||||||
"name": "My Kobo"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Error Responses
|
|
||||||
|
|
||||||
| Code | Description |
|
|
||||||
| ---- | ------------------------ |
|
|
||||||
| 401 | Invalid or expired token |
|
|
||||||
| 404 | Media item not found |
|
|
||||||
@@ -0,0 +1,135 @@
|
|||||||
|
# Position Contract
|
||||||
|
|
||||||
|
How reading positions are represented, submitted, verified, and restored
|
||||||
|
across clients. This is the contract every client speaks — readium-based
|
||||||
|
apps, web (foliate), KOReader, and Kobo devices alike.
|
||||||
|
|
||||||
|
**The server is the position authority.** Every submission is
|
||||||
|
independently verified against the book itself before it is stored; a
|
||||||
|
client whose locator math is wrong cannot poison the stored position.
|
||||||
|
In exchange, the server hands back handles clients can apply directly,
|
||||||
|
so they never have to parse or trust CFIs themselves.
|
||||||
|
|
||||||
|
## The three-tier submission
|
||||||
|
|
||||||
|
Clients submit what their renderer can reliably observe. All three tiers
|
||||||
|
are accepted on `PUT /api/media-items/:id/progress` (and on the KOReader
|
||||||
|
device endpoints):
|
||||||
|
|
||||||
|
| Tier | Field | Currency | Notes |
|
||||||
|
| ------ | -------------- | ----------------------------------------------------- | ----- |
|
||||||
|
| 1 | `percentage` | Float 0..1 of the whole book | Universal. The only field every client can supply. |
|
||||||
|
| 2 | `context_text` | Up to 100 chars, whitespace-normalized, starting at the anchor | The verification anchor. The server extracts the text at the submitted structural anchor and compares. |
|
||||||
|
| 3 | `epubcfi` | Standard wrapped CFI, terminals in UTF-16 code units | Optional structural anchor. KOReader submits a CRE xpointer and the server converts it. |
|
||||||
|
|
||||||
|
Tier 2 is what makes the system self-correcting: percentages alone
|
||||||
|
cannot distinguish "the reader's locator is right" from "the reader
|
||||||
|
silently reported wherever it is currently scrolled".
|
||||||
|
|
||||||
|
## Verification and healing (ingest side)
|
||||||
|
|
||||||
|
On every progress save for a convertible reflowable book that carries a
|
||||||
|
`context_text`, the server:
|
||||||
|
|
||||||
|
1. Parses the submitted `epubcfi` (if any) and resolves it against the
|
||||||
|
book's own XHTML.
|
||||||
|
2. Extracts the text at the resolved anchor and cross-checks it with the
|
||||||
|
submitted `context_text`.
|
||||||
|
3. On a mismatch — or an unresolvable anchor, or no anchor at all —
|
||||||
|
heals the position by text search, using the submitted `percentage`
|
||||||
|
to disambiguate repeated phrases. The healed CFI and a recomputed
|
||||||
|
percentage are stored in place of the submitted ones.
|
||||||
|
4. Refreshes the book-wide `character_offset` column from the verified
|
||||||
|
anchor on every verified save, so it never goes stale behind the
|
||||||
|
anchor.
|
||||||
|
|
||||||
|
If verification fails outright (book file unreadable, context not found
|
||||||
|
and percentage cannot disambiguate), the error is logged and the
|
||||||
|
submission is stored as-is — verification never rejects a save, it only
|
||||||
|
corrects.
|
||||||
|
|
||||||
|
## The restore handles
|
||||||
|
|
||||||
|
`GET /api/media-items/:id/progress` serves, alongside the raw stored
|
||||||
|
fields, the server-derived handles for reflowable books with a
|
||||||
|
resolvable anchor:
|
||||||
|
|
||||||
|
| Field | Currency | Meaning |
|
||||||
|
| ------------- | ----------------------- | ------- |
|
||||||
|
| `anchor_href` | — | The spine document the anchor lives in (e.g. `1984.xhtml`). **This is how a client finds the right resource.** |
|
||||||
|
| `css_selector`| — | Body-relative `tag:nth-child(k)` chain of the anchor's block element (e.g. `body>div:nth-child(4)>p:nth-child(29)`). |
|
||||||
|
| `char_offset` | UTF-16 code units | Offset of the anchor within the concatenated text of that block. |
|
||||||
|
| `epubcfi` | UTF-16 terminals | The stored canonical CFI — re-served healed if the GET-time re-verification improved it. |
|
||||||
|
| `context_text`| — | The stored verification context (≤100 normalized chars from the anchor). |
|
||||||
|
|
||||||
|
**Recommended restore sequence for a client:**
|
||||||
|
|
||||||
|
1. Open the book and jump to the stored `percentage` (coarse floor).
|
||||||
|
2. Resolve `anchor_href` against your own spine (an ends-with match on
|
||||||
|
document hrefs) and open that document if you are not already there.
|
||||||
|
3. Query `css_selector` in that document, walk its text nodes counting
|
||||||
|
UTF-16 units to `char_offset`, and scroll that position into view.
|
||||||
|
4. If the anchor cannot be measured (renderer-specific laziness), the
|
||||||
|
percentage floor stands.
|
||||||
|
|
||||||
|
## SPINE NUMBERING HAZARD — read this before parsing a stored CFI
|
||||||
|
|
||||||
|
**`epubcfi` spine steps index the OPF spine AS WRITTEN, including
|
||||||
|
`linear="no"` items.** Several rendering engines (notably readium)
|
||||||
|
number their reading order EXCLUDING `linear="no"` items. When a book's
|
||||||
|
cover (or any other item) is `linear="no"`, the two numberings differ by
|
||||||
|
a constant offset from that item onward — a client resolving a stored
|
||||||
|
CFI's spine step against its own numbering lands in the WRONG DOCUMENT.
|
||||||
|
|
||||||
|
This is not hypothetical: "1984" epubs commonly have a `linear="no"`
|
||||||
|
cover, which makes readium spine 0 = OPF spine 1. The server heals such
|
||||||
|
numbering mismatches at ingest (that is what the context check is for),
|
||||||
|
but the durable rule for client authors is:
|
||||||
|
|
||||||
|
> **Never resolve a stored CFI's spine step yourself.** Resolve the
|
||||||
|
> document by `anchor_href`, then land with `css_selector` +
|
||||||
|
> `char_offset`. Treat the CFI as opaque server currency.
|
||||||
|
|
||||||
|
## Offset currencies
|
||||||
|
|
||||||
|
Two counting systems are in play, and they are deliberately kept apart:
|
||||||
|
|
||||||
|
| Quantity | Currency | Why |
|
||||||
|
| -------- | -------- | --- |
|
||||||
|
| CFI terminal offsets (`…/1:456`) | UTF-16 code units | The EPUB CFI spec, and what every client observes (JavaScript `.length`). |
|
||||||
|
| `char_offset` (block-relative handle) | UTF-16 code units | Same reason — clients walk DOM text with JS semantics. |
|
||||||
|
| KOReader CRE `text().N` offsets | UTF-16 code units | crengine is UCS-16 internally. |
|
||||||
|
| `character_offset` (book-wide column) | Unicode runes | Internal, consistent with `total_characters` and the percentage derivations. |
|
||||||
|
|
||||||
|
For all-BMP text the two currencies are identical. They diverge on
|
||||||
|
astral-plane characters (emoji, rare CJK ideographs): one rune, two
|
||||||
|
UTF-16 units. The server converts at every wire boundary; internal
|
||||||
|
arithmetic never crosses.
|
||||||
|
|
||||||
|
## `context_text` rules
|
||||||
|
|
||||||
|
- Starts at the anchor position (it may begin mid-word).
|
||||||
|
- Whitespace-normalized (all runs of whitespace collapse to single
|
||||||
|
spaces).
|
||||||
|
- At most 100 characters.
|
||||||
|
- Comparison is containment-based (client and server suffixes of the
|
||||||
|
same block verify in either direction); contexts shorter than 12
|
||||||
|
chars never match.
|
||||||
|
|
||||||
|
## Engine-specific notes
|
||||||
|
|
||||||
|
- **readium-based clients** (the Android app): submit
|
||||||
|
percentage + `context_text` + a CFI generated from the laid-out
|
||||||
|
WebView. Their locally-generated CFI spine steps use readium
|
||||||
|
numbering — the server heals the difference; nothing to do.
|
||||||
|
- **Web (foliate)**: submit all three tiers; foliate CFIs are the same
|
||||||
|
currency the server stores.
|
||||||
|
- **KOReader**: submits a CRE xpointer in the `epubcfi` field of the
|
||||||
|
device payload (historical field name; it is a CRE xpointer, not a
|
||||||
|
CFI). The server converts CRE → canonical at ingest and canonical →
|
||||||
|
CRE on pull (`koreader_xpointer` in the metadata response).
|
||||||
|
- **Kobo**: submits kepub CFI locators, converted server-side the same
|
||||||
|
way.
|
||||||
|
- **Fixed-layout content** (PDF/CBZ): the page index is the canonical
|
||||||
|
locator; CFI/xpointer are meaningless and neither submitted nor
|
||||||
|
served.
|
||||||
@@ -0,0 +1,91 @@
|
|||||||
|
# Update Media Item Progress
|
||||||
|
|
||||||
|
Submit reading progress for a media item. This is the position-authority
|
||||||
|
ingest point: the server independently verifies the submission against
|
||||||
|
the book and heals it when the client's projection is wrong.
|
||||||
|
|
||||||
|
**Endpoint**: `PUT /api/media-items/:id/progress`
|
||||||
|
**Auth**: Required (Bearer token)
|
||||||
|
**Content-Type**: `application/json`
|
||||||
|
|
||||||
|
See the [Position Contract](position-contract.md) for the three-tier
|
||||||
|
submission model, the verification/healing semantics, and the offset
|
||||||
|
currencies.
|
||||||
|
|
||||||
|
## Path Parameters
|
||||||
|
|
||||||
|
| Parameter | Type | Required | Description |
|
||||||
|
| --------- | ------------- | -------- | ---------------- |
|
||||||
|
| id | string (UUID) | Yes | Media item UUID |
|
||||||
|
|
||||||
|
## Request Body
|
||||||
|
|
||||||
|
All fields are optional; submit what your renderer can observe.
|
||||||
|
|
||||||
|
| Field | Type | Description |
|
||||||
|
| ----- | ---- | ----------- |
|
||||||
|
| `percentage` | float | Position as a fraction of the whole book (0..1). Tier 1 — the universal field. |
|
||||||
|
| `context_text` | string | Up to 100 whitespace-normalized chars starting at the anchor. Tier 2 — enables verification and healing. |
|
||||||
|
| `epubcfi` | string | Standard wrapped CFI (UTF-16 terminals). Tier 3 — the structural anchor. Note: the server re-derives the stored canonical CFI; a client's own spine numbering is healed if it disagrees with the OPF spine. |
|
||||||
|
| `character_offset` | int | Book-wide rune offset. Accepted but recomputed server-side from the verified anchor on every verified save. |
|
||||||
|
| `current_page`, `total_pages` | int | Fixed-layout position. For fixed-layout formats the page index is the canonical locator. |
|
||||||
|
| `chapter`, `chapter_progress` | int, float | Chapter index and within-chapter fraction. |
|
||||||
|
| `reading_mode` | string | `paged` or `scrolled` (fixed-layout reader state). |
|
||||||
|
| `zoom_level`, `scroll_position_x`, `scroll_position_y` | float | Fixed-layout viewport state. |
|
||||||
|
|
||||||
|
### Example Request (reflowable, all three tiers)
|
||||||
|
|
||||||
|
```http
|
||||||
|
PUT /api/media-items/774641f9-317b-4087-8e04-53bb4392ae56/progress
|
||||||
|
Authorization: Bearer eyJhbGciOiJIUzI1NiIs...
|
||||||
|
Content-Type: application/json
|
||||||
|
```
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"percentage": 0.0415,
|
||||||
|
"context_text": "was at war with one of these Powers it was generally at peace with the other. But what was strange w",
|
||||||
|
"epubcfi": "epubcfi(/6/4!/4/8[_idContainer003]/62/1:456)"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Example Request (fixed-layout)
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"percentage": 0.15,
|
||||||
|
"current_page": 30,
|
||||||
|
"total_pages": 194,
|
||||||
|
"reading_mode": "paged"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Behavior
|
||||||
|
|
||||||
|
- The submission is verified against the book's XHTML when
|
||||||
|
`context_text` is present and the book is convertible reflowable: the
|
||||||
|
CFI is resolved, the text at the anchor is compared with
|
||||||
|
`context_text`, and mismatches heal by text search (percentage
|
||||||
|
disambiguates repeats). See [Position Contract](position-contract.md).
|
||||||
|
- **Anti-clobber guard**: a submission with `percentage < 0.005` is
|
||||||
|
ignored with `{"status": "ignored"}` when the stored row already holds
|
||||||
|
a percentage above `0.01` — re-opening a book at its first page does
|
||||||
|
not wipe real progress.
|
||||||
|
- The response is the stored row after verification, including the
|
||||||
|
healed `epubcfi` and the refreshed `character_offset` when
|
||||||
|
verification ran.
|
||||||
|
|
||||||
|
## Response (200 OK)
|
||||||
|
|
||||||
|
The saved progress row (same shape as
|
||||||
|
[GET](get_media_progress.md), minus the GET-time handles). The
|
||||||
|
`epubcfi` and `percentage` in the response are the server-verified
|
||||||
|
values, which may differ from the submitted ones when healing occurred.
|
||||||
|
|
||||||
|
## Error Responses
|
||||||
|
|
||||||
|
| Code | Description |
|
||||||
|
| ---- | ----------- |
|
||||||
|
| 400 | Invalid request data |
|
||||||
|
| 401 | Invalid or expired token |
|
||||||
|
| 500 | Database error |
|
||||||
@@ -1,68 +0,0 @@
|
|||||||
# Update Reading Progress
|
|
||||||
|
|
||||||
Update reading progress for a media item. This will sync across all devices via WebSocket.
|
|
||||||
|
|
||||||
**Endpoint**: `PUT /api/media-items/{media_id}/progress`
|
|
||||||
**Auth**: Required
|
|
||||||
**Content-Type**: `application/json`
|
|
||||||
|
|
||||||
## Path Parameters
|
|
||||||
|
|
||||||
| Parameter | Type | Required | Description |
|
|
||||||
| --------- | ------ | -------- | --------------- |
|
|
||||||
| media_id | string | Yes | Media item UUID |
|
|
||||||
|
|
||||||
## Request Body
|
|
||||||
|
|
||||||
| Field | Type | Required | Description |
|
|
||||||
| --------------------------- | ------- | -------- | ------------------------------------------------- |
|
|
||||||
| source | string | Yes | Progress source (e.g., "web", "koreader", "kobo") |
|
|
||||||
| location | object | Yes | Location information |
|
|
||||||
| location.percentage | float | No | Progress percentage (0-1) |
|
|
||||||
| location.epubcfi | string | No | EPUB CFI location |
|
|
||||||
| location.character | integer | No | Character offset |
|
|
||||||
| location.chapter | integer | No | Chapter number |
|
|
||||||
| location.page | integer | No | Current page |
|
|
||||||
| location.total_pages | integer | No | Total pages |
|
|
||||||
| device_metadata | object | No | Device metadata |
|
|
||||||
| device_metadata.device_type | string | No | Device type |
|
|
||||||
| device_metadata.user_agent | string | No | User agent string |
|
|
||||||
|
|
||||||
### Example Request
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"source": "web",
|
|
||||||
"location": {
|
|
||||||
"percentage": 0.45678,
|
|
||||||
"epubcfi": "epubcfi(/6/4/2:15)",
|
|
||||||
"character": 15432,
|
|
||||||
"chapter": 3,
|
|
||||||
"page": 89,
|
|
||||||
"total_pages": 200
|
|
||||||
},
|
|
||||||
"device_metadata": {
|
|
||||||
"device_type": "web",
|
|
||||||
"user_agent": "Mozilla/5.0..."
|
|
||||||
}
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Response (200 OK)
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"sync_status": "success",
|
|
||||||
"progress_updated": true,
|
|
||||||
"devices_notified": ["device-1", "device-2"],
|
|
||||||
"broadcast": true
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Error Responses
|
|
||||||
|
|
||||||
| Code | Description |
|
|
||||||
| ---- | ------------------------ |
|
|
||||||
| 400 | Invalid location data |
|
|
||||||
| 401 | Invalid or expired token |
|
|
||||||
| 404 | Media item not found |
|
|
||||||
@@ -1,56 +0,0 @@
|
|||||||
# Update Universal Progress
|
|
||||||
|
|
||||||
Update universal reading progress for a media item.
|
|
||||||
|
|
||||||
**Endpoint**: `POST /api/progress/:id`
|
|
||||||
**Auth**: Required
|
|
||||||
**Content-Type**: `application/json`
|
|
||||||
|
|
||||||
## Path Parameters
|
|
||||||
|
|
||||||
| Parameter | Type | Required | Description |
|
|
||||||
| --------- | ------------- | -------- | --------------- |
|
|
||||||
| id | string (UUID) | Yes | Media item UUID |
|
|
||||||
|
|
||||||
## Request Body
|
|
||||||
|
|
||||||
| Field | Type | Required | Description |
|
|
||||||
| ---------- | ------------- | -------- | ------------------------------------------- |
|
|
||||||
| percentage | float | No | Progress percentage (0-100) |
|
|
||||||
| position | integer | No | Current position in bytes |
|
|
||||||
| page | integer | No | Current page number |
|
|
||||||
| finished | boolean | No | Whether the book is finished |
|
|
||||||
| device_id | string (UUID) | No | Device UUID (optional, for tracking source) |
|
|
||||||
|
|
||||||
### Example Request
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"percentage": 75.5,
|
|
||||||
"position": 1234,
|
|
||||||
"page": 150,
|
|
||||||
"finished": false,
|
|
||||||
"device_id": "device-uuid"
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Response (200 OK)
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"media_item_id": "uuid",
|
|
||||||
"percentage": 75.5,
|
|
||||||
"position": 1234,
|
|
||||||
"page": 150,
|
|
||||||
"finished": false,
|
|
||||||
"updated_at": "2026-02-08T10:00:00Z"
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
## Error Responses
|
|
||||||
|
|
||||||
| Code | Description |
|
|
||||||
| ---- | ------------------------ |
|
|
||||||
| 400 | Invalid request data |
|
|
||||||
| 401 | Invalid or expired token |
|
|
||||||
| 404 | Media item not found |
|
|
||||||
@@ -953,6 +953,37 @@ func (mh *MediaHandler) GetMediaReadingProgress(c *echo.Context) error {
|
|||||||
"last_sync_timestamp": progress.LastSyncTimestamp,
|
"last_sync_timestamp": progress.LastSyncTimestamp,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Position authority: re-verify the stored anchor and derive the
|
||||||
|
// client-facing handles (cssSelector + anchor document href) server-
|
||||||
|
// side, so clients scroll to an address instead of parsing CFIs.
|
||||||
|
if progress.Epubcfi.Valid && wsync.IsConvertibleFormat(progress.FormatGroup) && mh.libraryService != nil {
|
||||||
|
contextText := ""
|
||||||
|
if progress.ContextText.Valid {
|
||||||
|
contextText = progress.ContextText.String
|
||||||
|
}
|
||||||
|
pct := 0.0
|
||||||
|
if progress.Percentage.Valid {
|
||||||
|
pct = progress.Percentage.Float64
|
||||||
|
}
|
||||||
|
if mediaItem, err := mh.db.GetMediaItem(c.Request().Context(), pgtype.UUID{Bytes: mediaUUID, Valid: true}); err == nil {
|
||||||
|
if path, perr := mh.libraryService.ResolveMediaPath(c.Request().Context(), mediaItem.LibraryID, mediaItem.FilePath); perr == nil && path != "" {
|
||||||
|
if finalCFI, sel, href, charOff, _, _, _, verr := wsync.VerifyProgressAnchor(path, progress.Epubcfi.String, contextText, pct); verr == nil && sel != "" {
|
||||||
|
resp["css_selector"] = sel
|
||||||
|
resp["anchor_href"] = href
|
||||||
|
if charOff != nil {
|
||||||
|
resp["char_offset"] = *charOff
|
||||||
|
}
|
||||||
|
if finalCFI != "" {
|
||||||
|
resp["epubcfi"] = finalCFI
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if progress.ContextText.Valid {
|
||||||
|
resp["context_text"] = progress.ContextText.String
|
||||||
|
}
|
||||||
|
|
||||||
return c.JSON(http.StatusOK, resp)
|
return c.JSON(http.StatusOK, resp)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+115
-24
@@ -431,7 +431,9 @@ func firstTextDescendant(n *html.Node) *html.Node {
|
|||||||
// textNodeAtRuneOffset walks text nodes under elem in document order and
|
// textNodeAtRuneOffset walks text nodes under elem in document order and
|
||||||
// returns the node containing the rune offset plus the local offset within
|
// returns the node containing the rune offset plus the local offset within
|
||||||
// that node. Offsets beyond the end clamp to the last node.
|
// that node. Offsets beyond the end clamp to the last node.
|
||||||
func textNodeAtRuneOffset(elem *html.Node, offset int) (*html.Node, int) {
|
// textNodeAtUTF16Offset resolves a crengine text().N offset — UTF-16 code
|
||||||
|
// units — to (text node, rune offset) within elem's text.
|
||||||
|
func textNodeAtUTF16Offset(elem *html.Node, offset int) (*html.Node, int) {
|
||||||
if offset < 0 {
|
if offset < 0 {
|
||||||
offset = 0
|
offset = 0
|
||||||
}
|
}
|
||||||
@@ -441,15 +443,15 @@ func textNodeAtRuneOffset(elem *html.Node, offset int) (*html.Node, int) {
|
|||||||
var walk func(*html.Node) bool
|
var walk func(*html.Node) bool
|
||||||
walk = func(node *html.Node) bool {
|
walk = func(node *html.Node) bool {
|
||||||
if node.Type == html.TextNode {
|
if node.Type == html.TextNode {
|
||||||
length := utf8.RuneCountInString(node.Data)
|
length := utf16Len(node.Data)
|
||||||
if remaining < length {
|
if remaining < length {
|
||||||
target = node
|
target = node
|
||||||
local = remaining
|
local = utf16ToRuneIndex(node.Data, remaining)
|
||||||
return true
|
return true
|
||||||
}
|
}
|
||||||
remaining -= length
|
remaining -= length
|
||||||
target = node
|
target = node
|
||||||
local = length
|
local = utf8.RuneCountInString(node.Data)
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
for child := node.FirstChild; child != nil; child = child.NextSibling {
|
for child := node.FirstChild; child != nil; child = child.NextSibling {
|
||||||
@@ -507,7 +509,7 @@ func (c *CFIConverter) convertByStructuralPath(body *html.Node, xp *CREXPointer,
|
|||||||
var textNode *html.Node
|
var textNode *html.Node
|
||||||
var localOffset int
|
var localOffset int
|
||||||
if xp.CharOffset > 0 {
|
if xp.CharOffset > 0 {
|
||||||
textNode, localOffset = textNodeAtRuneOffset(elem, xp.CharOffset)
|
textNode, localOffset = textNodeAtUTF16Offset(elem, xp.CharOffset)
|
||||||
} else {
|
} else {
|
||||||
textNode = firstTextDescendant(elem)
|
textNode = firstTextDescendant(elem)
|
||||||
localOffset = 0
|
localOffset = 0
|
||||||
@@ -1058,24 +1060,96 @@ func indexChildNodes(parent *html.Node) []indexedNode {
|
|||||||
return nodes
|
return nodes
|
||||||
}
|
}
|
||||||
|
|
||||||
func findTextChunkIndex(parent *html.Node, textNode *html.Node) (int, int) {
|
// Character-offset currency policy: the EPUB CFI spec and crengine both
|
||||||
|
// count UTF-16 code units (JavaScript `.length` semantics — what foliate,
|
||||||
|
// readium, KOReader and Kobo clients all observe), so every offset that
|
||||||
|
// CROSSES the wire — CFI terminals, CRE text() offsets, the served
|
||||||
|
// char_offset handle — is UTF-16. Internal arithmetic (book-level
|
||||||
|
// character_offset, percentage fractions) stays rune-based, consistent
|
||||||
|
// with TotalCharacters. These helpers convert at the boundaries; for
|
||||||
|
// all-BMP text the two currencies are identical, so ASCII books are
|
||||||
|
// unaffected.
|
||||||
|
|
||||||
|
// utf16Len returns the UTF-16 code-unit length of s.
|
||||||
|
func utf16Len(s string) int {
|
||||||
|
n := 0
|
||||||
|
for _, r := range s {
|
||||||
|
if r >= 0x10000 {
|
||||||
|
n += 2
|
||||||
|
} else {
|
||||||
|
n++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return n
|
||||||
|
}
|
||||||
|
|
||||||
|
// utf16ToRuneIndex converts a UTF-16 code-unit offset within s to a rune
|
||||||
|
// index (clamped to len(runes)).
|
||||||
|
func utf16ToRuneIndex(s string, u16 int) int {
|
||||||
|
if u16 <= 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
units := 0
|
||||||
|
i := 0
|
||||||
|
for _, r := range s {
|
||||||
|
if units >= u16 {
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
if r >= 0x10000 {
|
||||||
|
units += 2
|
||||||
|
} else {
|
||||||
|
units++
|
||||||
|
}
|
||||||
|
i++
|
||||||
|
}
|
||||||
|
return i
|
||||||
|
}
|
||||||
|
|
||||||
|
// runeToUTF16Index converts a rune index within s to a UTF-16 code-unit
|
||||||
|
// offset (clamped to the string's unit length).
|
||||||
|
func runeToUTF16Index(s string, runeIdx int) int {
|
||||||
|
if runeIdx <= 0 {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
units := 0
|
||||||
|
i := 0
|
||||||
|
for _, r := range s {
|
||||||
|
if i >= runeIdx {
|
||||||
|
return units
|
||||||
|
}
|
||||||
|
if r >= 0x10000 {
|
||||||
|
units += 2
|
||||||
|
} else {
|
||||||
|
units++
|
||||||
|
}
|
||||||
|
i++
|
||||||
|
}
|
||||||
|
return units
|
||||||
|
}
|
||||||
|
|
||||||
|
// findTextChunk locates the indexed text chunk containing textNode and
|
||||||
|
// returns its chunk index, the rune offset of the node within the chunk,
|
||||||
|
// and the chunk text up to and including the node (for UTF-16 conversion
|
||||||
|
// of chunk-relative offsets).
|
||||||
|
func findTextChunk(parent *html.Node, textNode *html.Node) (int, int, string) {
|
||||||
indexed := indexChildNodes(parent)
|
indexed := indexChildNodes(parent)
|
||||||
for i, node := range indexed {
|
for i, node := range indexed {
|
||||||
if node.isTextChunk() {
|
if node.isTextChunk() {
|
||||||
for j, tn := range node.textChunk {
|
var sb strings.Builder
|
||||||
if tn == textNode {
|
|
||||||
chunkOffset := 0
|
chunkOffset := 0
|
||||||
for k := 0; k < j; k++ {
|
for _, tn := range node.textChunk {
|
||||||
chunkOffset += utf8.RuneCountInString(node.textChunk[k].Data)
|
if tn == textNode {
|
||||||
|
return i, chunkOffset, sb.String() + textNode.Data
|
||||||
}
|
}
|
||||||
return i, chunkOffset
|
chunkOffset += utf8.RuneCountInString(tn.Data)
|
||||||
|
sb.WriteString(tn.Data)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
return -1, 0, ""
|
||||||
return -1, 0
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
func findElementCFIIndex(parent *html.Node, element *html.Node) int {
|
func findElementCFIIndex(parent *html.Node, element *html.Node) int {
|
||||||
indexed := indexChildNodes(parent)
|
indexed := indexChildNodes(parent)
|
||||||
for i, node := range indexed {
|
for i, node := range indexed {
|
||||||
@@ -1110,15 +1184,17 @@ func buildCFI(spineIndex int, textNode *html.Node, charOffset int) (string, erro
|
|||||||
return "", fmt.Errorf("text node has no parent")
|
return "", fmt.Errorf("text node has no parent")
|
||||||
}
|
}
|
||||||
|
|
||||||
chunkIdx, chunkOffset := findTextChunkIndex(parent, textNode)
|
chunkIdx, chunkOffset, chunkText := findTextChunk(parent, textNode)
|
||||||
if chunkIdx == -1 {
|
if chunkIdx == -1 {
|
||||||
return "", fmt.Errorf("text node not found in parent's indexed children")
|
return "", fmt.Errorf("text node not found in parent's indexed children")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The CFI terminal offset is UTF-16 code units (spec currency); the
|
||||||
|
// internal charOffset is runes. Convert over the chunk text.
|
||||||
totalOffset := chunkOffset + charOffset
|
totalOffset := chunkOffset + charOffset
|
||||||
|
|
||||||
var parts []string
|
var parts []string
|
||||||
parts = append(parts, fmt.Sprintf("/%d:%d", chunkIdx, totalOffset))
|
parts = append(parts, fmt.Sprintf("/%d:%d", chunkIdx, runeToUTF16Index(chunkText, totalOffset)))
|
||||||
|
|
||||||
current := parent
|
current := parent
|
||||||
for current != nil {
|
for current != nil {
|
||||||
@@ -1245,9 +1321,16 @@ func countTextChars(n *html.Node) int {
|
|||||||
return count
|
return count
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// countTextCharsBefore counts the text characters preceding target within
|
||||||
|
// its document: preceding siblings at each ancestor level, strictly
|
||||||
|
// EXCLUDING target's own subtree (callers add the in-node offset
|
||||||
|
// separately). Starting the walk at target itself — or recursing after
|
||||||
|
// counting the current node's subtree — would re-count the accumulated
|
||||||
|
// document once per ancestor level and multiply the result by the node's
|
||||||
|
// depth.
|
||||||
func countTextCharsBefore(target *html.Node) int {
|
func countTextCharsBefore(target *html.Node) int {
|
||||||
count := 0
|
count := 0
|
||||||
for c := target; c != nil; c = c.PrevSibling {
|
for c := target.PrevSibling; c != nil; c = c.PrevSibling {
|
||||||
count += countTextChars(c)
|
count += countTextChars(c)
|
||||||
}
|
}
|
||||||
if target.Parent != nil {
|
if target.Parent != nil {
|
||||||
@@ -1519,24 +1602,31 @@ func resolveCFIToNode(doc *html.Node, steps []cfiStep) (*html.Node, int, error)
|
|||||||
entry := indexed[lastStep.Index]
|
entry := indexed[lastStep.Index]
|
||||||
|
|
||||||
if entry.isTextChunk() {
|
if entry.isTextChunk() {
|
||||||
textOffset := 0
|
// The CFI terminal offset arrives in UTF-16 code units (spec
|
||||||
|
// currency — foliate/readium/KOReader/Kobo all emit UTF-16).
|
||||||
|
// Walk the chunk in UTF-16 units, then convert the hit position
|
||||||
|
// to the internal rune offset.
|
||||||
|
u16Remaining := 0
|
||||||
if lastStep.HasOffset {
|
if lastStep.HasOffset {
|
||||||
textOffset = lastStep.Offset
|
u16Remaining = lastStep.Offset
|
||||||
}
|
}
|
||||||
|
|
||||||
var targetNode *html.Node
|
var targetNode *html.Node
|
||||||
remainingOffset := textOffset
|
runeIntoTarget := 0
|
||||||
for _, tn := range entry.textChunk {
|
for _, tn := range entry.textChunk {
|
||||||
textLen := utf8.RuneCountInString(tn.Data)
|
units := utf16Len(tn.Data)
|
||||||
if remainingOffset < textLen || (remainingOffset == textLen && targetNode == nil) {
|
if u16Remaining < units || (u16Remaining == units && targetNode == nil) {
|
||||||
targetNode = tn
|
targetNode = tn
|
||||||
|
runeIntoTarget = utf16ToRuneIndex(tn.Data, u16Remaining)
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
remainingOffset -= textLen
|
u16Remaining -= units
|
||||||
targetNode = tn
|
targetNode = tn
|
||||||
|
runeIntoTarget = utf8.RuneCountInString(tn.Data)
|
||||||
}
|
}
|
||||||
if targetNode == nil && len(entry.textChunk) > 0 {
|
if targetNode == nil && len(entry.textChunk) > 0 {
|
||||||
targetNode = entry.textChunk[len(entry.textChunk)-1]
|
targetNode = entry.textChunk[len(entry.textChunk)-1]
|
||||||
|
runeIntoTarget = utf8.RuneCountInString(targetNode.Data)
|
||||||
}
|
}
|
||||||
|
|
||||||
parent := targetNode.Parent
|
parent := targetNode.Parent
|
||||||
@@ -1547,7 +1637,7 @@ func resolveCFIToNode(doc *html.Node, steps []cfiStep) (*html.Node, int, error)
|
|||||||
}
|
}
|
||||||
totalOffset += countTextChars(c)
|
totalOffset += countTextChars(c)
|
||||||
}
|
}
|
||||||
totalOffset += remainingOffset
|
totalOffset += runeIntoTarget
|
||||||
|
|
||||||
return targetNode, totalOffset, nil
|
return targetNode, totalOffset, nil
|
||||||
}
|
}
|
||||||
@@ -1606,7 +1696,8 @@ func buildCREXPointer(spineIndex int, node *html.Node, charOffset int) (string,
|
|||||||
xpointer := fmt.Sprintf("/body/DocFragment[%d]/body%s", fragIndex, strings.Join(parts, ""))
|
xpointer := fmt.Sprintf("/body/DocFragment[%d]/body%s", fragIndex, strings.Join(parts, ""))
|
||||||
|
|
||||||
if charOffset > 0 || (node.Type == html.TextNode) {
|
if charOffset > 0 || (node.Type == html.TextNode) {
|
||||||
xpointer += fmt.Sprintf("/text().%d", charOffset)
|
// crengine counts UTF-16 code units; charOffset is internal runes.
|
||||||
|
xpointer += fmt.Sprintf("/text().%d", runeToUTF16Index(node.Data, charOffset))
|
||||||
}
|
}
|
||||||
|
|
||||||
return xpointer, nil
|
return xpointer, nil
|
||||||
|
|||||||
@@ -133,7 +133,7 @@ func writeTestEPUB(t *testing.T) string {
|
|||||||
}
|
}
|
||||||
docs := []spineDoc{
|
docs := []spineDoc{
|
||||||
{"doc1.xhtml", "<body><div><p>Chapter one opening page.</p></div></body>"},
|
{"doc1.xhtml", "<body><div><p>Chapter one opening page.</p></div></body>"},
|
||||||
{"doc2.xhtml", "<body><div><p>The family of Dashwood had long been settled in Sussex.</p><p>Their estate was large, and their residence was at Norland Park.</p></div></body>"},
|
{"doc2.xhtml", "<body><div><p>The family of Dashwood had long been settled in Sussex.</p><p>Their estate was large, and their residence was at Norland Park.</p><p>The family crest shows a globe \U0001F30D and a rocket \U0001F680 flying onward.</p></div></body>"},
|
||||||
{"doc3.xhtml", "<body><div><p>Chapter three contents.</p></div></body>"},
|
{"doc3.xhtml", "<body><div><p>Chapter three contents.</p></div></body>"},
|
||||||
{"doc4.xhtml", "<body><div><p>Chapter four contents.</p></div></body>"},
|
{"doc4.xhtml", "<body><div><p>Chapter four contents.</p></div></body>"},
|
||||||
{"doc5.xhtml", "<body><div><p>Chapter five contents.</p></div></body>"},
|
{"doc5.xhtml", "<body><div><p>Chapter five contents.</p></div></body>"},
|
||||||
|
|||||||
@@ -0,0 +1,349 @@
|
|||||||
|
package sync
|
||||||
|
|
||||||
|
// Ingest-side position authority for reading progress.
|
||||||
|
//
|
||||||
|
// Clients submit (percentage, context_text, epubcfi). The epubcfi is a
|
||||||
|
// standard wrapped CFI — epubcfi(/6/N!/…) — resolvable against the same
|
||||||
|
// document this package parses, so every submission is INDEPENDENTLY
|
||||||
|
// verified: the text at the resolved anchor is extracted and compared
|
||||||
|
// with the submitted context. A mismatch (or an unresolvable anchor)
|
||||||
|
// heals the position by text search, with the submitted percentage
|
||||||
|
// disambiguating repeated phrases, instead of trusting a client-side
|
||||||
|
// projection. This keeps buggy clients from poisoning stored positions:
|
||||||
|
// a resolver that silently returns "wherever I'm currently scrolled"
|
||||||
|
// fails the context check and gets healed to the true location.
|
||||||
|
//
|
||||||
|
// The anchor's block element is also derived as a cssSelector plus a
|
||||||
|
// block-relative character offset and served back — the readium-native
|
||||||
|
// handle clients scroll to, so they never have to parse or trust CFIs
|
||||||
|
// themselves.
|
||||||
|
//
|
||||||
|
// Note: readium-based clients number their readingOrder excluding
|
||||||
|
// linear="no" spine items, while this package's spine index follows the
|
||||||
|
// OPF spine as written. The spine index is therefore internal-only;
|
||||||
|
// client-facing responses carry the anchor document's href instead.
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"strings"
|
||||||
|
"unicode/utf8"
|
||||||
|
|
||||||
|
"golang.org/x/net/html"
|
||||||
|
)
|
||||||
|
|
||||||
|
// IsConvertibleFormat reports whether a format group uses standard CFIs
|
||||||
|
// as its structural locator currency (i.e. whether CFI verification and
|
||||||
|
// cssSelector derivation apply to it).
|
||||||
|
func IsConvertibleFormat(formatGroup string) bool {
|
||||||
|
return isConvertible(string(formatGroup))
|
||||||
|
}
|
||||||
|
|
||||||
|
func truncateRunes(s string, n int) string {
|
||||||
|
r := []rune(s)
|
||||||
|
if len(r) <= n {
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
return string(r[:n])
|
||||||
|
}
|
||||||
|
|
||||||
|
// anchorBlock returns the nearest non-inline (block-level) ancestor of a
|
||||||
|
// resolved text/element node — the server-side equivalent of the
|
||||||
|
// readers' computed-style block walk.
|
||||||
|
func anchorBlock(node *html.Node) *html.Node {
|
||||||
|
if node == nil {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if node.Type == html.ElementNode && !isInlineFormatting(node) {
|
||||||
|
return node
|
||||||
|
}
|
||||||
|
return findBlockParent(node)
|
||||||
|
}
|
||||||
|
|
||||||
|
// blockContextText extracts the normalized text from (node, runeOff) to
|
||||||
|
// the end of the anchor block — the same excerpt rule the readers use for
|
||||||
|
// context_text, ≤100 chars.
|
||||||
|
func blockContextText(block, node *html.Node, runeOff int) string {
|
||||||
|
segments := collectInlineText(block)
|
||||||
|
var sb strings.Builder
|
||||||
|
started := false
|
||||||
|
for _, seg := range segments {
|
||||||
|
if !started && seg.node == node {
|
||||||
|
started = true
|
||||||
|
runes := []rune(string(seg.runes))
|
||||||
|
if runeOff < len(runes) {
|
||||||
|
sb.WriteString(string(runes[runeOff:]))
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if started {
|
||||||
|
sb.WriteString(string(seg.runes))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return truncateRunes(normalizeWhitespace(sb.String()), 100)
|
||||||
|
}
|
||||||
|
|
||||||
|
// contextMatches reports whether a server-extracted context and a client-
|
||||||
|
// submitted context describe the same anchor. Both are suffixes of the
|
||||||
|
// same block text when the anchors share a block, so containment in
|
||||||
|
// either direction verifies; empty or very short contexts never match.
|
||||||
|
func contextMatches(serverCtx, submitted string) bool {
|
||||||
|
s := truncateRunes(normalizeWhitespace(submitted), 100)
|
||||||
|
t := truncateRunes(normalizeWhitespace(serverCtx), 100)
|
||||||
|
if s == "" || t == "" {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
short, long := s, t
|
||||||
|
if len([]rune(short)) > len([]rune(long)) {
|
||||||
|
short, long = long, short
|
||||||
|
}
|
||||||
|
if len([]rune(short)) < 12 {
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
return strings.Contains(long, short)
|
||||||
|
}
|
||||||
|
|
||||||
|
// cssSelectorFor mirrors the readers' selOf: a body-relative
|
||||||
|
// tag:nth-child(k) chain (k = 1-based position among element siblings).
|
||||||
|
func cssSelectorFor(block *html.Node) string {
|
||||||
|
var segs []string
|
||||||
|
n := block
|
||||||
|
for n != nil && n.Type == html.ElementNode && n.Data != "body" {
|
||||||
|
k := 1
|
||||||
|
sib := n.Parent.FirstChild
|
||||||
|
for sib != nil && sib != n {
|
||||||
|
if sib.Type == html.ElementNode {
|
||||||
|
k++
|
||||||
|
}
|
||||||
|
sib = sib.NextSibling
|
||||||
|
}
|
||||||
|
segs = append([]string{n.Data + ":nth-child(" + fmt.Sprintf("%d", k) + ")"}, segs...)
|
||||||
|
n = n.Parent
|
||||||
|
}
|
||||||
|
return "body>" + strings.Join(segs, ">")
|
||||||
|
}
|
||||||
|
|
||||||
|
// blockCharOffset computes the offset of (node, runeOff) within the
|
||||||
|
// concatenated text of its block, in UTF-16 code units — the client-side
|
||||||
|
// scroll-target currency (JavaScript .length semantics).
|
||||||
|
func blockCharOffset(block, node *html.Node, runeOff int) int {
|
||||||
|
segments := collectInlineText(block)
|
||||||
|
var sb strings.Builder
|
||||||
|
for _, seg := range segments {
|
||||||
|
if seg.node == node {
|
||||||
|
runes := seg.runes
|
||||||
|
if runeOff < len(runes) {
|
||||||
|
runes = runes[:runeOff]
|
||||||
|
}
|
||||||
|
sb.WriteString(string(runes))
|
||||||
|
return runeToUTF16Index(sb.String(), utf8.RuneCountInString(sb.String()))
|
||||||
|
}
|
||||||
|
sb.WriteString(string(seg.runes))
|
||||||
|
}
|
||||||
|
return utf16Len(sb.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
// bookCharOffset computes the book-wide rune offset of (node, runeOff) —
|
||||||
|
// the reading_progress.character_offset column's currency, consistent with
|
||||||
|
// TotalCharacters and the percentage derivations.
|
||||||
|
func bookCharOffset(conv *CFIConverter, spineIndex int, node *html.Node, runeOff int) int {
|
||||||
|
spine, err := conv.loadSpine()
|
||||||
|
if err != nil {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
before := 0
|
||||||
|
for i := 0; i < spineIndex && i < len(spine.items); i++ {
|
||||||
|
doc, _, derr := conv.getContentDoc(i + 1)
|
||||||
|
if derr != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if b := findBody(doc); b != nil {
|
||||||
|
before += countTextChars(b)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return before + countTextCharsBefore(node) + runeOff
|
||||||
|
}
|
||||||
|
|
||||||
|
// ProgressAnchor is the full server-computed apply handle for a stored
|
||||||
|
// standard CFI: the anchor block's cssSelector, the character offset
|
||||||
|
// within that block's text, and the spine document's href.
|
||||||
|
type ProgressAnchor struct {
|
||||||
|
CSSSelector string
|
||||||
|
CharOffset int
|
||||||
|
Href string
|
||||||
|
HealedCFI string
|
||||||
|
Healed bool
|
||||||
|
HealedPct *float64
|
||||||
|
}
|
||||||
|
|
||||||
|
// VerifyProgressAnchor resolves a client-submitted standard CFI against
|
||||||
|
// the EPUB, cross-checks the submitted context text, and heals the anchor
|
||||||
|
// by text search on any mismatch. charOffset is the anchor's block-
|
||||||
|
// relative UTF-16 offset (the served char_offset handle); bookOffset is
|
||||||
|
// the anchor's book-wide rune offset (the character_offset column's
|
||||||
|
// currency) — callers refresh the column from it on every verified save
|
||||||
|
// so it never goes stale behind the anchor.
|
||||||
|
func VerifyProgressAnchor(epubPath, epubcfi, contextText string, percentage float64) (finalCFI string, cssSelector string, anchorHref string, charOffset *int, bookOffset *int, healedPct *float64, healed bool, err error) {
|
||||||
|
finalCFI = epubcfi
|
||||||
|
anchorHref = ""
|
||||||
|
|
||||||
|
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
||||||
|
if err != nil {
|
||||||
|
cfi, sel, href, off, book, pct, healedFlag, herr := healFromContext(epubPath, contextText, percentage)
|
||||||
|
return cfi, sel, href, off, book, pct, healedFlag, herr
|
||||||
|
}
|
||||||
|
conv := cachedConverter(epubPath)
|
||||||
|
doc, docHref, err := conv.getContentDoc(spineIndex + 1)
|
||||||
|
if err != nil {
|
||||||
|
return healFromContext(epubPath, contextText, percentage)
|
||||||
|
}
|
||||||
|
node, runeOff, rerr := resolveCFIToNode(doc, localSteps)
|
||||||
|
if rerr != nil {
|
||||||
|
return healFromContext(epubPath, contextText, percentage)
|
||||||
|
}
|
||||||
|
anchorHref = docHref
|
||||||
|
|
||||||
|
block := anchorBlock(node)
|
||||||
|
serverCtx := blockContextText(block, node, runeOff)
|
||||||
|
if contextMatches(serverCtx, contextText) {
|
||||||
|
off := blockCharOffset(block, node, runeOff)
|
||||||
|
book := bookCharOffset(conv, spineIndex, node, runeOff)
|
||||||
|
return finalCFI, cssSelectorFor(block), anchorHref, &off, &book, nil, false, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Mismatch: heal by text search.
|
||||||
|
hCFI, hPct, hSel, hHref, hBook, herr := healAnchorByText(epubPath, contextText, percentage, spineIndex)
|
||||||
|
if herr != nil {
|
||||||
|
return finalCFI, "", hHref, nil, nil, nil, false, fmt.Errorf("context mismatch (server %q vs client %q) and heal failed: %w",
|
||||||
|
truncateRunes(serverCtx, 40), truncateRunes(contextText, 40), herr)
|
||||||
|
}
|
||||||
|
hOff := blockCharOffsetFor(epubPath, hCFI)
|
||||||
|
return hCFI, hSel, hHref, &hOff, &hBook, &hPct, true, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func healFromContext(epubPath, contextText string, percentage float64) (string, string, string, *int, *int, *float64, bool, error) {
|
||||||
|
cfi, healedPct, sel, href, book, err := healAnchorByText(epubPath, contextText, percentage, -1)
|
||||||
|
if err != nil {
|
||||||
|
return "", "", "", nil, nil, nil, false, err
|
||||||
|
}
|
||||||
|
off := blockCharOffsetFor(epubPath, cfi)
|
||||||
|
return cfi, sel, href, &off, &book, &healedPct, true, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func healAnchorByText(epubPath, contextText string, percentage float64, spineIndex int) (string, float64, string, string, int, error) {
|
||||||
|
if epubPath == "" {
|
||||||
|
return "", 0, "", "", 0, fmt.Errorf("no epub available for text anchoring")
|
||||||
|
}
|
||||||
|
conv := cachedConverter(epubPath)
|
||||||
|
spine, err := conv.loadSpine()
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, "", "", 0, err
|
||||||
|
}
|
||||||
|
needle := truncateRunes(normalizeWhitespace(contextText), 40)
|
||||||
|
if len([]rune(needle)) < 12 {
|
||||||
|
return "", 0, "", "", 0, fmt.Errorf("context too short to anchor")
|
||||||
|
}
|
||||||
|
|
||||||
|
type match struct {
|
||||||
|
spine int
|
||||||
|
node *html.Node
|
||||||
|
off int
|
||||||
|
}
|
||||||
|
var matches []match
|
||||||
|
totalChars := 0
|
||||||
|
charsBefore := make([]int, len(spine.items))
|
||||||
|
for i := range spine.items {
|
||||||
|
doc, _, derr := conv.getContentDoc(i + 1)
|
||||||
|
if derr != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
body := findBody(doc)
|
||||||
|
if body == nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
charsBefore[i] = totalChars
|
||||||
|
totalChars += countTextChars(body)
|
||||||
|
if n, runeOff := findTextInNode(body, needle); n != nil {
|
||||||
|
matches = append(matches, match{spine: i, node: n, off: runeOff})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(matches) == 0 || totalChars <= 0 {
|
||||||
|
return "", 0, "", "", 0, fmt.Errorf("context not found in book")
|
||||||
|
}
|
||||||
|
|
||||||
|
best := matches[0]
|
||||||
|
bestDist := -1.0
|
||||||
|
for _, m := range matches {
|
||||||
|
frac := (float64(charsBefore[m.spine]) + float64(countTextCharsBefore(m.node)+m.off)) / float64(totalChars)
|
||||||
|
d := frac - percentage
|
||||||
|
if d < 0 {
|
||||||
|
d = -d
|
||||||
|
}
|
||||||
|
if bestDist < 0 || d < bestDist {
|
||||||
|
best = m
|
||||||
|
bestDist = d
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
bookOff := charsBefore[best.spine] + countTextCharsBefore(best.node) + best.off
|
||||||
|
healedPct := float64(bookOff) / float64(totalChars)
|
||||||
|
cfi, err := buildCFI(best.spine, best.node, best.off)
|
||||||
|
if err != nil {
|
||||||
|
return "", 0, "", "", 0, err
|
||||||
|
}
|
||||||
|
sel := ""
|
||||||
|
if block := anchorBlock(best.node); block != nil {
|
||||||
|
sel = cssSelectorFor(block)
|
||||||
|
}
|
||||||
|
return cfi, healedPct, sel, spine.items[best.spine].href, bookOff, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func blockCharOffsetFor(epubPath, cfi string) int {
|
||||||
|
n, off, _, herr := cfiTextAtAnchorInternal(epubPath, cfi)
|
||||||
|
if herr != nil {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
block := anchorBlock(n)
|
||||||
|
if block == nil {
|
||||||
|
return 0
|
||||||
|
}
|
||||||
|
return blockCharOffset(block, n, off)
|
||||||
|
}
|
||||||
|
|
||||||
|
// cfiTextAtAnchorInternal resolves a stored standard CFI to its anchor
|
||||||
|
// text node, rune offset, and containing document.
|
||||||
|
func cfiTextAtAnchorInternal(epubPath, cfi string) (*html.Node, int, *html.Node, error) {
|
||||||
|
spineIndex, localSteps, err := parseEPUBCFI(cfi)
|
||||||
|
if err != nil {
|
||||||
|
return nil, 0, nil, err
|
||||||
|
}
|
||||||
|
conv := cachedConverter(epubPath)
|
||||||
|
doc, _, err := conv.getContentDoc(spineIndex + 1)
|
||||||
|
if err != nil {
|
||||||
|
return nil, 0, nil, err
|
||||||
|
}
|
||||||
|
node, off, err := resolveCFIToNode(doc, localSteps)
|
||||||
|
return node, off, doc, err
|
||||||
|
}
|
||||||
|
|
||||||
|
// ProgressCSSSelector derives the anchor block's cssSelector from a stored
|
||||||
|
// standard CFI.
|
||||||
|
func ProgressCSSSelector(epubPath, epubcfi string) (string, error) {
|
||||||
|
spineIndex, localSteps, err := parseEPUBCFI(epubcfi)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
conv := cachedConverter(epubPath)
|
||||||
|
doc, _, err := conv.getContentDoc(spineIndex + 1)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
node, _, err := resolveCFIToNode(doc, localSteps)
|
||||||
|
if err != nil {
|
||||||
|
return "", err
|
||||||
|
}
|
||||||
|
block := anchorBlock(node)
|
||||||
|
if block == nil {
|
||||||
|
return "", fmt.Errorf("no block ancestor for anchor")
|
||||||
|
}
|
||||||
|
return cssSelectorFor(block), nil
|
||||||
|
}
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
package sync
|
||||||
|
|
||||||
|
import (
|
||||||
|
"os"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestManualRealBookAnchor(t *testing.T) {
|
||||||
|
path := "/home/nymusicman/Code/bookhoard/uploads/Ebooks/George Orwell/1984 (1269)/1984 - George Orwell.epub"
|
||||||
|
if _, err := os.Stat(path); err != nil {
|
||||||
|
t.Skip("real book not present on this machine")
|
||||||
|
}
|
||||||
|
cfi := "epubcfi(/6/2!/4[x1984]/8[_idContainer003]/62/1:456)"
|
||||||
|
ctx := "was at war with one of these Powers it was generally at peace with the other"
|
||||||
|
finalCFI, sel, href, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, cfi, ctx, 0.0415)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("VerifyProgressAnchor error: %v", err)
|
||||||
|
}
|
||||||
|
t.Logf("final=%q\n healed=%v healedPct=%v\n selector=%q href=%q charOff=%v", finalCFI, healed, healedPct, sel, href, charOff)
|
||||||
|
sel2, err2 := ProgressCSSSelector(path, cfi)
|
||||||
|
t.Logf("ProgressCSSSelector: %q err=%v", sel2, err2)
|
||||||
|
}
|
||||||
@@ -0,0 +1,248 @@
|
|||||||
|
package sync
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"testing"
|
||||||
|
)
|
||||||
|
|
||||||
|
// The fixture (writeTestEPUB in cfi_converter_test.go): 6 spine docs.
|
||||||
|
// doc2 = spine index 1 with two paragraphs:
|
||||||
|
// p1: "The family of Dashwood had long been settled in Sussex."
|
||||||
|
// p2: "Their estate was large, and their residence was at Norland Park."
|
||||||
|
// Hand-derived local paths (html→body /4, body→div /4, div→p /4|/6,
|
||||||
|
// p→text chunk /1):
|
||||||
|
const (
|
||||||
|
dashwoodCFI = "epubcfi(/6/4!/4/2/2/1:0)"
|
||||||
|
dashwoodText = "The family of Dashwood had long been settled in Sussex."
|
||||||
|
estateCFI = "epubcfi(/6/4!/4/2/4/1:0)"
|
||||||
|
estateText = "Their estate was large, and their residence was at Norland Park."
|
||||||
|
dashwoodSelect = "body>div:nth-child(1)>p:nth-child(1)"
|
||||||
|
estateSelect = "body>div:nth-child(1)>p:nth-child(2)"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestVerifyProgressAnchorAcceptsExact(t *testing.T) {
|
||||||
|
path := writeTestEPUB(t)
|
||||||
|
finalCFI, sel, _, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, dashwoodCFI, dashwoodText, 0.3)
|
||||||
|
_ = charOff
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("verify error: %v", err)
|
||||||
|
}
|
||||||
|
if healed {
|
||||||
|
t.Errorf("exact anchor should not heal")
|
||||||
|
}
|
||||||
|
if healedPct != nil {
|
||||||
|
t.Errorf("exact anchor should not carry a healed percentage")
|
||||||
|
}
|
||||||
|
if finalCFI != dashwoodCFI {
|
||||||
|
t.Errorf("finalCFI = %q, want unchanged %q", finalCFI, dashwoodCFI)
|
||||||
|
}
|
||||||
|
if sel != dashwoodSelect {
|
||||||
|
t.Errorf("cssSelector = %q, want %q", sel, dashwoodSelect)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestVerifyProgressAnchorHealsMismatch(t *testing.T) {
|
||||||
|
path := writeTestEPUB(t)
|
||||||
|
// Anchored at p1 but the context is p2's text: the classic
|
||||||
|
// client-projection bug — the server must heal to the true location.
|
||||||
|
finalCFI, _, _, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, dashwoodCFI, estateText, 0.3)
|
||||||
|
_ = charOff
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("verify error: %v", err)
|
||||||
|
}
|
||||||
|
if !healed {
|
||||||
|
t.Fatalf("expected healing, got none (cfi=%q)", finalCFI)
|
||||||
|
}
|
||||||
|
if finalCFI != estateCFI {
|
||||||
|
t.Errorf("healed CFI = %q, want %q", finalCFI, estateCFI)
|
||||||
|
}
|
||||||
|
if healedPct == nil || *healedPct <= 0 {
|
||||||
|
t.Errorf("healed percentage not recomputed: %v", healedPct)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Healing must converge: re-verifying the healed anchor is a no-op.
|
||||||
|
finalCFI2, _, _, _, _, healedPct2, healed2, err := VerifyProgressAnchor(path, finalCFI, estateText, 0.3)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("re-verify error: %v", err)
|
||||||
|
}
|
||||||
|
if healed2 {
|
||||||
|
t.Errorf("healed anchor should be stable, got healed again to %q (pct %v)", finalCFI2, healedPct2)
|
||||||
|
}
|
||||||
|
if finalCFI2 != finalCFI {
|
||||||
|
t.Errorf("re-verify CFI = %q, want %q", finalCFI2, finalCFI)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestVerifyProgressAnchorHealsUnresolvable(t *testing.T) {
|
||||||
|
path := writeTestEPUB(t)
|
||||||
|
// Element index 99 is out of range in doc2 — the anchor cannot resolve.
|
||||||
|
bogus := "epubcfi(/6/4!/4/2/99/1:0)"
|
||||||
|
finalCFI, _, _, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, bogus, dashwoodText, 0.05)
|
||||||
|
_ = charOff
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("verify error: %v", err)
|
||||||
|
}
|
||||||
|
if !healed {
|
||||||
|
t.Fatal("unresolvable anchor should heal from context")
|
||||||
|
}
|
||||||
|
if finalCFI != dashwoodCFI {
|
||||||
|
t.Errorf("healed CFI = %q, want the Dashwood anchor %q", finalCFI, dashwoodCFI)
|
||||||
|
}
|
||||||
|
if healedPct == nil {
|
||||||
|
t.Errorf("healed percentage not recomputed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestVerifyProgressAnchorDumbClient(t *testing.T) {
|
||||||
|
path := writeTestEPUB(t)
|
||||||
|
// No CFI at all — a client that only knows percentage + context is
|
||||||
|
// fully supported: the server anchors structurally from the text.
|
||||||
|
finalCFI, sel, _, charOff, _, healedPct, healed, err := VerifyProgressAnchor(path, "", estateText, 0.3)
|
||||||
|
_ = charOff
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("verify error: %v", err)
|
||||||
|
}
|
||||||
|
if !healed {
|
||||||
|
t.Fatal("context-only submission should count as anchored-by-heal")
|
||||||
|
}
|
||||||
|
if finalCFI != estateCFI {
|
||||||
|
t.Errorf("anchored CFI = %q, want %q", finalCFI, estateCFI)
|
||||||
|
}
|
||||||
|
if sel != estateSelect {
|
||||||
|
t.Errorf("cssSelector = %q, want %q", sel, estateSelect)
|
||||||
|
}
|
||||||
|
if healedPct == nil {
|
||||||
|
t.Errorf("percentage not recomputed for context-only anchor")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestContextMatches(t *testing.T) {
|
||||||
|
block := "The family of Dashwood had long been settled in Sussex."
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
serverCtx string
|
||||||
|
submitted string
|
||||||
|
want bool
|
||||||
|
}{
|
||||||
|
{"exact", block, block, true},
|
||||||
|
{"suffix of block", block, "settled in Sussex.", true},
|
||||||
|
{"block is suffix", "settled in Sussex.", block, true},
|
||||||
|
{"mid-block substring", block, "Dashwood had long been", true},
|
||||||
|
{"different text", block, "completely unrelated words here", false},
|
||||||
|
{"too short", block, "the", false},
|
||||||
|
{"empty submitted", block, "", false},
|
||||||
|
{"empty server", "", "some long enough context text", false},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
if got := contextMatches(tc.serverCtx, tc.submitted); got != tc.want {
|
||||||
|
t.Errorf("contextMatches(%q, %q) = %v, want %v", tc.serverCtx, tc.submitted, got, tc.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestParseStandardCFIRange(t *testing.T) {
|
||||||
|
// Web progress CFIs are range CFIs; parse must resolve to the start arm.
|
||||||
|
spineIndex, localSteps, err := parseEPUBCFI("epubcfi(/6/4!/4/4:0,/4/4:53)")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("parse error: %v", err)
|
||||||
|
}
|
||||||
|
if spineIndex != 1 {
|
||||||
|
t.Errorf("spineIndex = %d, want 1", spineIndex)
|
||||||
|
}
|
||||||
|
if len(localSteps) != 4 {
|
||||||
|
t.Fatalf("steps = %d, want 4 (parent + start arm)", len(localSteps))
|
||||||
|
}
|
||||||
|
last := localSteps[len(localSteps)-1]
|
||||||
|
if last.Index != 4 || !last.HasOffset || last.Offset != 53 {
|
||||||
|
t.Errorf("start-arm step = %+v, want /4:0", last)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The emoji in fixture doc2's third paragraph are astral plane runes:
|
||||||
|
// one rune, two UTF-16 code units. Every wire offset (CFI terminals,
|
||||||
|
// the served char_offset) must therefore be UTF-16, while the book-wide
|
||||||
|
// character_offset column stays rune-based.
|
||||||
|
//
|
||||||
|
// Case 1: the context starts at rune 31 of the paragraph text ("The
|
||||||
|
// family crest shows a globe " = 31 BMP runes), so its UTF-16 offset is
|
||||||
|
// also 31 — the currencies agree.
|
||||||
|
//
|
||||||
|
// Case 2: the context starts at rune 33 ("and a rocket ..."), with the
|
||||||
|
// astral 🌍 (rune 31, units 31-32) BEFORE the offset — the UTF-16 offset
|
||||||
|
// is 34, one more than the rune offset. That +1 is the whole point of
|
||||||
|
// the boundary conversion.
|
||||||
|
//
|
||||||
|
// Local path: p3 of doc2's div = /4/2/6, text chunk /1.
|
||||||
|
func TestAstralOffsetCurrency(t *testing.T) {
|
||||||
|
path := writeTestEPUB(t)
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
context string
|
||||||
|
wantTerm int
|
||||||
|
}{
|
||||||
|
{"before any emoji", "🌍 and a rocket 🚀 flying onward.", 31},
|
||||||
|
{"after one emoji", "and a rocket 🚀 flying onward.", 34},
|
||||||
|
}
|
||||||
|
for _, tc := range cases {
|
||||||
|
t.Run(tc.name, func(t *testing.T) {
|
||||||
|
wantCFI := fmt.Sprintf("epubcfi(/6/4!/4/2/6/1:%d)", tc.wantTerm)
|
||||||
|
|
||||||
|
finalCFI, sel, _, charOff, bookOff, _, healed, err := VerifyProgressAnchor(path, "", tc.context, 0.3)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("heal error: %v", err)
|
||||||
|
}
|
||||||
|
if !healed {
|
||||||
|
t.Fatalf("expected heal from context-only submission")
|
||||||
|
}
|
||||||
|
if finalCFI != wantCFI {
|
||||||
|
t.Errorf("healed CFI = %q, want %q (UTF-16 terminal)", finalCFI, wantCFI)
|
||||||
|
}
|
||||||
|
if sel != "body>div:nth-child(1)>p:nth-child(3)" {
|
||||||
|
t.Errorf("cssSelector = %q", sel)
|
||||||
|
}
|
||||||
|
if charOff == nil || *charOff != tc.wantTerm {
|
||||||
|
t.Errorf("block char_offset = %v, want %d UTF-16 units", charOff, tc.wantTerm)
|
||||||
|
}
|
||||||
|
if bookOff == nil || *bookOff <= 0 {
|
||||||
|
t.Errorf("book offset = %v, want a positive book-wide rune offset", bookOff)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Round trip: the healed CFI must verify exactly, with the
|
||||||
|
// same UTF-16 block handle and a stable book offset.
|
||||||
|
finalCFI2, _, _, charOff2, bookOff2, healedPct2, healed2, err := VerifyProgressAnchor(path, finalCFI, tc.context, 0.3)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("re-verify error: %v", err)
|
||||||
|
}
|
||||||
|
if healed2 || healedPct2 != nil {
|
||||||
|
t.Errorf("exact round trip should not heal (healed=%v)", healed2)
|
||||||
|
}
|
||||||
|
if finalCFI2 != wantCFI {
|
||||||
|
t.Errorf("re-verified CFI = %q, want %q", finalCFI2, wantCFI)
|
||||||
|
}
|
||||||
|
if charOff2 == nil || *charOff2 != tc.wantTerm {
|
||||||
|
t.Errorf("re-verified block char_offset = %v, want %d", charOff2, tc.wantTerm)
|
||||||
|
}
|
||||||
|
if bookOff2 == nil || bookOff == nil || *bookOff2 != *bookOff {
|
||||||
|
t.Errorf("book offset unstable: %v vs %v", bookOff2, bookOff)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// The book-wide offset must order anchors the way the book orders them:
|
||||||
|
// a later paragraph in the same document has a strictly larger
|
||||||
|
// character_offset.
|
||||||
|
func TestBookOffsetOrdering(t *testing.T) {
|
||||||
|
path := writeTestEPUB(t)
|
||||||
|
_, _, _, _, book1, _, _, err1 := VerifyProgressAnchor(path, dashwoodCFI, dashwoodText, 0.3)
|
||||||
|
_, _, _, _, book2, _, _, err2 := VerifyProgressAnchor(path, estateCFI, estateText, 0.3)
|
||||||
|
if err1 != nil || err2 != nil {
|
||||||
|
t.Fatalf("verify errors: %v %v", err1, err2)
|
||||||
|
}
|
||||||
|
if book1 == nil || book2 == nil {
|
||||||
|
t.Fatalf("book offsets missing: %v %v", book1, book2)
|
||||||
|
}
|
||||||
|
if *book2 <= *book1 {
|
||||||
|
t.Errorf("estate offset %d should exceed dashwood offset %d", *book2, *book1)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -274,8 +274,18 @@ func EstimatedPages(totalCharacters int64) int {
|
|||||||
type ProgressService struct {
|
type ProgressService struct {
|
||||||
db *database.Queries
|
db *database.Queries
|
||||||
connManager *ConnectionManager
|
connManager *ConnectionManager
|
||||||
|
mediaPaths MediaPathResolver
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// MediaPathResolver resolves a media item's library-relative file path to
|
||||||
|
// an absolute path, so the progress service can verify submitted anchors
|
||||||
|
// against the book itself.
|
||||||
|
type MediaPathResolver interface {
|
||||||
|
ResolveMediaPath(ctx context.Context, libraryID pgtype.UUID, relativePath string) (string, error)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s *ProgressService) SetMediaPathResolver(r MediaPathResolver) { s.mediaPaths = r }
|
||||||
|
|
||||||
func NewProgressService(db *database.Queries, connManager *ConnectionManager) *ProgressService {
|
func NewProgressService(db *database.Queries, connManager *ConnectionManager) *ProgressService {
|
||||||
return &ProgressService{db: db, connManager: connManager}
|
return &ProgressService{db: db, connManager: connManager}
|
||||||
}
|
}
|
||||||
@@ -435,6 +445,43 @@ func (s *ProgressService) SaveProgress(ctx context.Context, req SaveProgressRequ
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Canonical position authority: resolve the submitted CFI against the
|
||||||
|
// book itself, cross-check the submitted context text, and heal the
|
||||||
|
// anchor by text search on any mismatch. Clients submit what their
|
||||||
|
// renderer can reliably observe (percentage + visible text); the
|
||||||
|
// server owns the structural math and keeps buggy clients from
|
||||||
|
// poisoning stored positions.
|
||||||
|
if !isFixed && s.mediaPaths != nil && params.ContextText.Valid &&
|
||||||
|
isConvertible(string(formatGroup)) && formatGroup != "" {
|
||||||
|
if path, perr := s.mediaPaths.ResolveMediaPath(ctx, mediaItem.LibraryID, mediaItem.FilePath); perr == nil && path != "" {
|
||||||
|
submittedCFI := ""
|
||||||
|
if params.Epubcfi.Valid {
|
||||||
|
submittedCFI = params.Epubcfi.String
|
||||||
|
}
|
||||||
|
submittedPct := 0.0
|
||||||
|
if params.Percentage.Valid {
|
||||||
|
submittedPct = params.Percentage.Float64
|
||||||
|
}
|
||||||
|
if finalCFI, _, _, _, bookOff, healedPct, healed, verr := VerifyProgressAnchor(path, submittedCFI, params.ContextText.String, submittedPct); verr != nil {
|
||||||
|
log.Printf("Bookhoard: progress anchor verify failed for %s: %v", req.MediaItemID.String(), verr)
|
||||||
|
} else {
|
||||||
|
if healed || (!params.Epubcfi.Valid && finalCFI != "") {
|
||||||
|
params.Epubcfi = pgtype.Text{String: finalCFI, Valid: true}
|
||||||
|
if healedPct != nil {
|
||||||
|
params.Percentage = pgtype.Float8{Float64: *healedPct, Valid: true}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// The verified book-wide offset refreshes the column on
|
||||||
|
// EVERY verified save — not only heals — so it never goes
|
||||||
|
// stale behind the anchor (and a block-relative offset is
|
||||||
|
// never written into the book-level column).
|
||||||
|
if bookOff != nil {
|
||||||
|
params.CharacterOffset = pgtype.Int8{Int64: int64(*bookOff), Valid: true}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
conflictDetected := false
|
conflictDetected := false
|
||||||
if hasExisting && existing.LastSyncSource.Valid && existing.LastSyncSource.String != req.Source {
|
if hasExisting && existing.LastSyncSource.Valid && existing.LastSyncSource.String != req.Source {
|
||||||
if existing.LastSyncTimestamp.Valid {
|
if existing.LastSyncTimestamp.Valid {
|
||||||
|
|||||||
Reference in New Issue
Block a user