Author SHA1 Message Date
juliaandClaude Sonnet 4.6 78c23557da feat: fruit search by name/synonym and type filter (story #06)
Backend: List repo query gains name (ILIKE with wildcard escaping on
name + synonym LEFT JOIN, SELECT DISTINCT) and types (ANY cast) params;
handler parses ?name= and ?type=, resolves combined-type aliases from
domain.FruitTypeAliases, validates plain types against validFruitTypes
to prevent Postgres enum cast errors. ORDER BY f.name, f.id for stable
pagination.

Frontend: listFruits gains optional {name, type} params; fruitStore adds
searchName/searchType state and setSearch action (resets offset, refetches);
FruitList.vue wires text input (debounced 300 ms + Enter) and flat type
dropdown (17 enum values + 4 aliases per spec §5 order); debounce timer
cleared on unmount.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-18 12:46:38 +02:00
julia 50bfa30b06 Merge pull request 'feat: import existing publications from XML (story #05)' (#5) from feature/05-import-publications into main
Reviewed-on: #5
2026-06-18 10:20:01 +00:00
juliaandClaude Sonnet 4.6 e8d48d4fcb feat: import existing publications from XML (story #05)
scripts/import_publications.py imports all osw=1 publications from
03-data/osws.xml — upserts pub rows, loads cover images, scans
03-data/osdb/{pubId}/ for fruit images and PDFs, links matched fruits
by osdb_number. Idempotent re-runs. 17 unit tests. Makefile updated
to include import_publications_test in make test. Added
scripts/README_import.md with run order. Added spec-first rule to
.claude/CLAUDE.md.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-18 11:47:45 +02:00
16 changed files with 768 additions and 30 deletions
+11
View File
@@ -0,0 +1,11 @@
## Taiga
- TAIGA_URL: https://taiga.db-extern.de
- TAIGA_PROJECT_SLUG: obstsortendatenbank
## Stories
Before implementing a story: read its spec file in docs/planning/specs/ first.
## Branching
Stories developed sequentially; prior story branch may not be merged to main yet.
Before creating feature branch: run `git branch -r | grep feature/` — if predecessor
story branch exists on remote and is unmerged, base new branch on it, not main.
+1 -1
View File
@@ -34,7 +34,7 @@ run-frontend:
test:
cd backend && go test ./...
cd frontend && npm run test -- --run
cd scripts && python3 -m unittest import_fruits_test -v
cd scripts && python3 -m unittest import_fruits_test import_publications_test -v
## fmt: format all code
fmt:
+2 -1
View File
@@ -6,7 +6,8 @@ A full-stack fruit-variety database (Go + Vue 3).
- Users can view a "Hello, OSDB!" landing page that confirms the backend is reachable via the Vite proxy.
- Users can create, view, edit, and delete fruit varieties, including managing synonyms and uploading images.
- Administrators can bulk-import the legacy fruit database from XML using `scripts/import_fruits.py`.
- Administrators can bulk-import the legacy fruit database from XML using `scripts/import_fruits.py`, and then import all publications (cover images, linked fruits, PDFs, fruit images) using `scripts/import_publications.py`.
- Users can search fruits by name or synonym (case-insensitive, debounced) and filter by type or combined-type alias (e.g. "Birnen- und Quittensorten") from the fruit list.
- Users can manage publications (books, catalogues) with cover images, linked fruits, per-fruit PDF descriptions, and fruit images; fruit detail pages show publication descriptions and images.
## Quick Start
+8
View File
@@ -2,6 +2,14 @@ package domain
import "time"
// FruitTypeAliases maps combined-type alias labels to the enum values they expand to.
var FruitTypeAliases = map[string][]string{
"Birnen- und Quittensorten": {"Birnensorten", "Quittensorten"},
"Aprikosen und Pfirsiche": {"Aprikosen", "Pfirsiche"},
"Mirabellen und Reineclauden": {"Mirabellen", "Renekloden"},
"Pflaumen und Zwetschen": {"Pflaumen", "Zwetschen"},
}
type Fruit struct {
ID int `json:"id"`
Name string `json:"name"`
+13 -2
View File
@@ -21,7 +21,7 @@ var (
// FruitRepository is the consumer-defined interface the handler depends on.
// The pg implementation in the repository package satisfies this structurally.
type FruitRepository interface {
List(ctx context.Context, limit, offset int) ([]domain.Fruit, int, error)
List(ctx context.Context, limit, offset int, name string, types []string) ([]domain.Fruit, int, error)
Get(ctx context.Context, id int) (domain.Fruit, error)
Create(ctx context.Context, dto domain.FruitWriteDTO) (domain.Fruit, error)
Update(ctx context.Context, id int, dto domain.FruitWriteDTO) (domain.Fruit, error)
@@ -107,7 +107,18 @@ func (h *FruitHandler) List(c echo.Context) error {
}
}
fruits, total, err := h.repo.List(c.Request().Context(), limit, offset)
name := c.QueryParam("name")
var types []string
if typeParam := c.QueryParam("type"); typeParam != "" {
if expanded, ok := domain.FruitTypeAliases[typeParam]; ok {
types = expanded
} else if _, ok := validFruitTypes[typeParam]; ok {
types = []string{typeParam}
}
// unknown typeParam → types stays nil → no filter applied
}
fruits, total, err := h.repo.List(c.Request().Context(), limit, offset, name, types)
if err != nil {
return c.JSON(http.StatusInternalServerError, map[string]string{"error": "internal server error"})
}
+97 -1
View File
@@ -52,9 +52,32 @@ func newFakeRepo() *fakeRepo {
}
}
func (r *fakeRepo) List(_ context.Context, limit, offset int) ([]domain.Fruit, int, error) {
func (r *fakeRepo) List(_ context.Context, limit, offset int, name string, types []string) ([]domain.Fruit, int, error) {
nameLower := strings.ToLower(name)
typeSet := make(map[string]struct{}, len(types))
for _, t := range types {
typeSet[t] = struct{}{}
}
all := make([]domain.Fruit, 0, len(r.fruits))
for _, f := range r.fruits {
if name != "" {
nameMatch := strings.Contains(strings.ToLower(f.Name), nameLower)
synMatch := false
for _, s := range f.Synonyms {
if strings.Contains(strings.ToLower(s), nameLower) {
synMatch = true
break
}
}
if !nameMatch && !synMatch {
continue
}
}
if len(types) > 0 {
if _, ok := typeSet[f.FruitType]; !ok {
continue
}
}
all = append(all, f)
}
total := len(all)
@@ -241,6 +264,79 @@ func TestFruitList_WithItems(t *testing.T) {
}
}
func TestFruitList_FilterByName(t *testing.T) {
repo := newFakeRepo()
repo.fruits[1] = domain.Fruit{ID: 1, Name: "Boskop", OSDBNumber: "A001", FruitType: "Apfelsorten", Synonyms: []string{}, Images: []domain.FruitImage{}}
repo.fruits[2] = domain.Fruit{ID: 2, Name: "Cox Orange", OSDBNumber: "A002", FruitType: "Apfelsorten", Synonyms: []string{}, Images: []domain.FruitImage{}}
h := handler.NewFruitHandler(repo)
e := newEcho()
req := httptest.NewRequest(http.MethodGet, "/api/v1/fruits?name=Boskop", nil)
rec := httptest.NewRecorder()
c := e.NewContext(req, rec)
if err := h.List(c); err != nil {
t.Fatal(err)
}
if rec.Code != http.StatusOK {
t.Fatalf("want 200 got %d", rec.Code)
}
var resp domain.FruitListResponse
json.Unmarshal(rec.Body.Bytes(), &resp)
if resp.Total != 1 {
t.Fatalf("want 1 result got %d", resp.Total)
}
if resp.Items[0].Name != "Boskop" {
t.Fatalf("want Boskop got %s", resp.Items[0].Name)
}
}
func TestFruitList_FilterByType(t *testing.T) {
repo := newFakeRepo()
repo.fruits[1] = domain.Fruit{ID: 1, Name: "Boskop", OSDBNumber: "A001", FruitType: "Apfelsorten", Synonyms: []string{}, Images: []domain.FruitImage{}}
repo.fruits[2] = domain.Fruit{ID: 2, Name: "Williams", OSDBNumber: "B001", FruitType: "Birnensorten", Synonyms: []string{}, Images: []domain.FruitImage{}}
h := handler.NewFruitHandler(repo)
e := newEcho()
req := httptest.NewRequest(http.MethodGet, "/api/v1/fruits?type=Apfelsorten", nil)
rec := httptest.NewRecorder()
c := e.NewContext(req, rec)
if err := h.List(c); err != nil {
t.Fatal(err)
}
if rec.Code != http.StatusOK {
t.Fatalf("want 200 got %d", rec.Code)
}
var resp domain.FruitListResponse
json.Unmarshal(rec.Body.Bytes(), &resp)
if resp.Total != 1 {
t.Fatalf("want 1 result got %d", resp.Total)
}
if resp.Items[0].FruitType != "Apfelsorten" {
t.Fatalf("want Apfelsorten got %s", resp.Items[0].FruitType)
}
}
func TestFruitList_AliasExpandsToMultipleTypes(t *testing.T) {
repo := newFakeRepo()
repo.fruits[1] = domain.Fruit{ID: 1, Name: "Williams", OSDBNumber: "B001", FruitType: "Birnensorten", Synonyms: []string{}, Images: []domain.FruitImage{}}
repo.fruits[2] = domain.Fruit{ID: 2, Name: "Quitte", OSDBNumber: "Q001", FruitType: "Quittensorten", Synonyms: []string{}, Images: []domain.FruitImage{}}
repo.fruits[3] = domain.Fruit{ID: 3, Name: "Boskop", OSDBNumber: "A001", FruitType: "Apfelsorten", Synonyms: []string{}, Images: []domain.FruitImage{}}
h := handler.NewFruitHandler(repo)
e := newEcho()
req := httptest.NewRequest(http.MethodGet, "/api/v1/fruits?type=Birnen-+und+Quittensorten", nil)
rec := httptest.NewRecorder()
c := e.NewContext(req, rec)
if err := h.List(c); err != nil {
t.Fatal(err)
}
if rec.Code != http.StatusOK {
t.Fatalf("want 200 got %d", rec.Code)
}
var resp domain.FruitListResponse
json.Unmarshal(rec.Body.Bytes(), &resp)
if resp.Total != 2 {
t.Fatalf("want 2 results (Birnensorten + Quittensorten) got %d", resp.Total)
}
}
// -- Get --
func TestFruitGet_Found(t *testing.T) {
+25 -4
View File
@@ -4,6 +4,7 @@ import (
"context"
"errors"
"fmt"
"strings"
"github.com/jackc/pgx/v5"
"github.com/jackc/pgx/v5/pgconn"
@@ -33,15 +34,35 @@ func mapPgError(err error) error {
return err
}
func (r *FruitRepo) List(ctx context.Context, limit, offset int) ([]domain.Fruit, int, error) {
func escapeLike(s string) string {
s = strings.ReplaceAll(s, `\`, `\\`)
s = strings.ReplaceAll(s, `%`, `\%`)
s = strings.ReplaceAll(s, `_`, `\_`)
return s
}
func (r *FruitRepo) List(ctx context.Context, limit, offset int, name string, types []string) ([]domain.Fruit, int, error) {
escaped := escapeLike(name)
countRow := r.pool.QueryRow(ctx,
`SELECT COUNT(DISTINCT f.id)
FROM fruits f
LEFT JOIN fruit_synonyms fs ON fs.fruit_id = f.id
WHERE ($1 = '' OR f.name ILIKE '%' || $1 || '%' ESCAPE '\' OR fs.synonym ILIKE '%' || $1 || '%' ESCAPE '\')
AND ($2::fruit_type[] IS NULL OR f.fruit_type = ANY($2::fruit_type[]))`,
escaped, types)
var total int
if err := r.pool.QueryRow(ctx, "SELECT COUNT(*) FROM fruits").Scan(&total); err != nil {
if err := countRow.Scan(&total); err != nil {
return nil, 0, err
}
rows, err := r.pool.Query(ctx,
`SELECT id, name, osdb_number, comment, fruit_type, created_at, updated_at
FROM fruits ORDER BY id LIMIT $1 OFFSET $2`, limit, offset)
`SELECT DISTINCT f.id, f.name, f.osdb_number, f.comment, f.fruit_type, f.created_at, f.updated_at
FROM fruits f
LEFT JOIN fruit_synonyms fs ON fs.fruit_id = f.id
WHERE ($1 = '' OR f.name ILIKE '%' || $1 || '%' ESCAPE '\' OR fs.synonym ILIKE '%' || $1 || '%' ESCAPE '\')
AND ($2::fruit_type[] IS NULL OR f.fruit_type = ANY($2::fruit_type[]))
ORDER BY f.name, f.id LIMIT $3 OFFSET $4`,
escaped, types, limit, offset)
if err != nil {
return nil, 0, err
}
@@ -97,7 +97,7 @@ func TestFruitRepoIntegration(t *testing.T) {
}
// List
fruits, total, err := repo.List(ctx, 50, 0)
fruits, total, err := repo.List(ctx, 50, 0, "", nil)
if err != nil {
t.Fatalf("List: %v", err)
}
+19 -10
View File
@@ -19,18 +19,17 @@ type MockResponse = {
const fetchMock = vi.fn((url: string, init?: RequestInit): Promise<MockResponse> => {
const method = init?.method ?? 'GET'
if (url === '/api/v1/fruits?limit=50&offset=0' && method === 'GET') {
if (url.startsWith('/api/v1/fruits?') && method === 'GET') {
const params = new URLSearchParams(url.split('?')[1])
return Promise.resolve({
ok: true,
status: 200,
json: async () => ({ items: [], total: 0, limit: 50, offset: 0 }),
})
}
if (url === '/api/v1/fruits?limit=10&offset=20' && method === 'GET') {
return Promise.resolve({
ok: true,
status: 200,
json: async () => ({ items: [], total: 0, limit: 10, offset: 20 }),
json: async () => ({
items: [],
total: 0,
limit: Number(params.get('limit') ?? 50),
offset: Number(params.get('offset') ?? 0),
}),
})
}
if (url === '/api/v1/fruits/1' && method === 'GET') {
@@ -99,9 +98,19 @@ describe('listFruits', () => {
})
it('passes custom limit and offset', async () => {
const result = await listFruits(10, 20)
const result = await listFruits({ limit: 10, offset: 20 })
expect(result.offset).toBe(20)
})
it('appends name param when provided', async () => {
await listFruits({ name: 'Boskop' })
expect(fetchMock).toHaveBeenLastCalledWith(expect.stringContaining('name=Boskop'))
})
it('appends type param when provided', async () => {
await listFruits({ type: 'Apfelsorten' })
expect(fetchMock).toHaveBeenLastCalledWith(expect.stringContaining('type=Apfelsorten'))
})
})
describe('getFruit', () => {
+11 -2
View File
@@ -77,8 +77,17 @@ async function checkOk(res: Response): Promise<Response> {
return res
}
export async function listFruits(limit = 50, offset = 0): Promise<FruitListResponse> {
const res = await fetch(`/api/v1/fruits?limit=${limit}&offset=${offset}`)
export async function listFruits(params?: {
limit?: number
offset?: number
name?: string
type?: string
}): Promise<FruitListResponse> {
const { limit = 50, offset = 0, name, type } = params ?? {}
let url = `/api/v1/fruits?limit=${limit}&offset=${offset}`
if (name) url += `&name=${encodeURIComponent(name)}`
if (type) url += `&type=${encodeURIComponent(type)}`
const res = await fetch(url)
return (await checkOk(res)).json()
}
+11
View File
@@ -115,4 +115,15 @@ describe('fruitStore', () => {
expect(store.fruits.find((f) => f.id === 1)).toBeUndefined()
expect(store.total).toBe(1)
})
it('setSearch updates search state and resets offset', async () => {
const store = useFruitStore()
await store.fetchFruits()
store.offset = 50
await store.setSearch('Boskop', 'Apfelsorten')
expect(store.searchName).toBe('Boskop')
expect(store.searchType).toBe('Apfelsorten')
expect(store.offset).toBe(0)
expect(fetchMock).toHaveBeenLastCalledWith(expect.stringContaining('name=Boskop'))
})
})
+16 -2
View File
@@ -18,12 +18,19 @@ export const useFruitStore = defineStore('fruit', () => {
const current = ref<Fruit | null>(null)
const loading = ref(false)
const error = ref<string | null>(null)
const searchName = ref('')
const searchType = ref('')
async function fetchFruits(lim = limit.value, off = offset.value) {
loading.value = true
error.value = null
try {
const resp = await listFruits(lim, off)
const resp = await listFruits({
limit: lim,
offset: off,
name: searchName.value || undefined,
type: searchType.value || undefined,
})
fruits.value = resp.items
total.value = resp.total
limit.value = resp.limit
@@ -35,6 +42,13 @@ export const useFruitStore = defineStore('fruit', () => {
}
}
async function setSearch(name: string, type: string) {
searchName.value = name
searchType.value = type
offset.value = 0
await fetchFruits(limit.value, 0)
}
async function fetchFruit(id: number) {
loading.value = true
error.value = null
@@ -71,5 +85,5 @@ export const useFruitStore = defineStore('fruit', () => {
if (current.value?.id === id) current.value = null
}
return { fruits, total, limit, offset, current, loading, error, fetchFruits, fetchFruit, create, update, remove }
return { fruits, total, limit, offset, current, loading, error, searchName, searchType, fetchFruits, fetchFruit, create, update, remove, setSearch }
})
+65 -6
View File
@@ -1,14 +1,46 @@
<script setup lang="ts">
import { onMounted, computed } from 'vue'
import { onMounted, onUnmounted, computed, ref } from 'vue'
import { RouterLink } from 'vue-router'
import { useFruitStore } from '../stores/fruitStore'
const SEARCH_TYPES = [
{ label: '(Alle)', value: '' },
{ label: 'Apfelsorten', value: 'Apfelsorten' },
{ label: 'Birnensorten', value: 'Birnensorten' },
{ label: 'Quittensorten', value: 'Quittensorten' },
{ label: 'Birnen- und Quittensorten', value: 'Birnen- und Quittensorten' },
{ label: 'Aprikosen', value: 'Aprikosen' },
{ label: 'Pfirsiche', value: 'Pfirsiche' },
{ label: 'Aprikosen und Pfirsiche', value: 'Aprikosen und Pfirsiche' },
{ label: 'Mirabellen', value: 'Mirabellen' },
{ label: 'Renekloden', value: 'Renekloden' },
{ label: 'Mirabellen und Reineclauden', value: 'Mirabellen und Reineclauden' },
{ label: 'Pflaumen', value: 'Pflaumen' },
{ label: 'Zwetschen', value: 'Zwetschen' },
{ label: 'Pflaumen und Zwetschen', value: 'Pflaumen und Zwetschen' },
{ label: 'Sauerkirschen', value: 'Sauerkirschen' },
{ label: 'Süßkirschen', value: 'Süßkirschen' },
{ label: 'Brombeeren', value: 'Brombeeren' },
{ label: 'Erdbeeren', value: 'Erdbeeren' },
{ label: 'Himbeeren', value: 'Himbeeren' },
{ label: 'Johannisbeeren', value: 'Johannisbeeren' },
{ label: 'Stachelbeeren', value: 'Stachelbeeren' },
{ label: 'Wein', value: 'Wein' },
]
const store = useFruitStore()
const nameInput = ref(store.searchName)
const typeSelect = ref(store.searchType)
let debounceTimer: ReturnType<typeof setTimeout> | null = null
onMounted(async () => {
await store.fetchFruits()
})
onUnmounted(() => {
if (debounceTimer) clearTimeout(debounceTimer)
})
const hasPrev = computed(() => store.offset > 0)
const hasNext = computed(() => store.offset + store.limit < store.total)
@@ -19,6 +51,23 @@ async function prev() {
async function next() {
await store.fetchFruits(store.limit, store.offset + store.limit)
}
function onNameInput() {
if (debounceTimer) clearTimeout(debounceTimer)
debounceTimer = setTimeout(() => {
store.setSearch(nameInput.value, typeSelect.value)
}, 300)
}
function onNameEnter() {
if (debounceTimer) clearTimeout(debounceTimer)
store.setSearch(nameInput.value, typeSelect.value)
}
function onTypeChange() {
if (debounceTimer) clearTimeout(debounceTimer)
store.setSearch(nameInput.value, typeSelect.value)
}
</script>
<template>
@@ -33,14 +82,24 @@ async function next() {
</RouterLink>
</div>
<!-- Search placeholder (story #06) -->
<div class="mb-4">
<div class="mb-4 flex gap-3">
<input
v-model="nameInput"
type="text"
placeholder="Suche... (kommt in Story #06)"
disabled
class="w-full border border-gray-300 rounded px-3 py-2 bg-gray-100 text-gray-400 cursor-not-allowed"
placeholder="Name oder Synonym suchen..."
class="flex-1 border border-gray-300 rounded px-3 py-2 focus:outline-none focus:ring-2 focus:ring-green-500"
@input="onNameInput"
@keydown.enter="onNameEnter"
/>
<select
v-model="typeSelect"
class="border border-gray-300 rounded px-3 py-2 focus:outline-none focus:ring-2 focus:ring-green-500"
@change="onTypeChange"
>
<option v-for="opt in SEARCH_TYPES" :key="opt.value" :value="opt.value">
{{ opt.label }}
</option>
</select>
</div>
<div v-if="store.loading" class="text-gray-500">Wird geladen...</div>
+37
View File
@@ -0,0 +1,37 @@
# Import Scripts
## Prerequisites
- Postgres running and migrated (`make migrate-up`)
- `DATABASE_URL` set, e.g.:
```
export DATABASE_URL=postgres://user:pass@localhost:5432/osdb
```
- `03-data/` directory present at repo root
- `psycopg2` installed (`pip install psycopg2-binary`)
---
## Run Order
Scripts must be run in order — publications import depends on fruits being present.
### 1. Import fruits
```bash
DATABASE_URL=... python3 scripts/import_fruits.py
```
Imports fruits, synonyms, and fruit images from `03-data/obstsorten.xml`.
Idempotent: upserts fruits by `osdb_number`.
### 2. Import publications
```bash
DATABASE_URL=... python3 scripts/import_publications.py
```
Imports publications from `03-data/osws.xml` (only `<obj>` elements with `<osw>=1`).
For each publication: upserts the row, loads cover image, links fruits found by
filesystem scan of `03-data/osdb/{pubId}/`, and imports PDFs and fruit images.
Idempotent: clears and re-inserts linked data on re-run.
+215
View File
@@ -0,0 +1,215 @@
#!/usr/bin/env python3
"""
Import publications from 03-data/osws.xml into the OSDB database.
Usage:
DATABASE_URL=postgres://... python3 scripts/import_publications.py
Idempotent: upserts publication by osdb_pub_id; deletes and re-inserts
linked fruits, descriptions, and fruit images per publication on re-run.
"""
import os
import re
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
import psycopg2
DATA_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "03-data")
def parse_publications(xml_path: str) -> list:
"""
Parse osws.xml and return list of publication dicts for <obj> elements with <osw>=1.
Each dict: id, title, author (None if "."), img_path (relative to data root, or None).
"""
tree = ET.parse(xml_path)
root = tree.getroot()
pubs = []
for obj in root.findall("obj"):
if (obj.findtext("osw") or "").strip() != "1":
continue
author_raw = (obj.findtext("author") or "").strip()
img = obj.findtext("img")
pubs.append({
"id": (obj.findtext("id") or "").strip(),
"title": (obj.findtext("name") or "").strip(),
"author": author_raw if author_raw and author_raw != "." else None,
"img_path": img.strip() if img else None,
})
return pubs
def scan_pub_dir(pub_dir: str, pub_id: str) -> dict:
"""
Scan pub_dir for files matching {osdb_number}_{pub_id}_s0.jpg and {osdb_number}_{pub_id}.pdf.
Returns dict mapping osdb_number → {img_path?: str, pdf_path?: str}.
Ignores _tn.jpg thumbnails and any other files.
"""
d = Path(pub_dir)
if not d.is_dir():
return {}
result = {}
img_re = re.compile(rf"^(.+)_{re.escape(pub_id)}_s0\.jpg$")
pdf_re = re.compile(rf"^(.+)_{re.escape(pub_id)}\.pdf$")
for f in d.iterdir():
m = img_re.match(f.name)
if m:
result.setdefault(m.group(1), {})["img_path"] = str(f)
continue
m = pdf_re.match(f.name)
if m:
result.setdefault(m.group(1), {})["pdf_path"] = str(f)
return result
def import_publications(conn, data_dir: str = None) -> dict:
"""
Import all publications from osws.xml into the database.
Returns counts: publications, covers, fruits_linked, pdfs, images, skipped_fruits.
"""
if data_dir is None:
data_dir = DATA_DIR
data_path = Path(data_dir)
pubs = parse_publications(str(data_path / "osws.xml"))
counts = {
"publications": 0,
"covers": 0,
"fruits_linked": 0,
"pdfs": 0,
"images": 0,
"skipped_fruits": 0,
}
with conn.cursor() as cur:
for pub in pubs:
pub_id = pub["id"]
# Load cover image bytes (skip silently if file missing or unreadable)
cover_data = None
if pub["img_path"]:
cover_path = data_path / pub["img_path"]
try:
cover_data = cover_path.read_bytes()
counts["covers"] += 1
except OSError:
pass
# Upsert publication row
cur.execute(
"""
INSERT INTO publications (title, author, osdb_pub_id, image_data, updated_at)
VALUES (%s, %s, %s, %s, NOW())
ON CONFLICT (osdb_pub_id) DO UPDATE
SET title = EXCLUDED.title,
author = EXCLUDED.author,
image_data = EXCLUDED.image_data,
updated_at = NOW()
RETURNING id
""",
(
pub["title"],
pub["author"],
pub_id,
psycopg2.Binary(cover_data) if cover_data else None,
),
)
db_pub_id = cur.fetchone()[0]
counts["publications"] += 1
# Scan filesystem for linked fruits
pub_dir = str(data_path / "osdb" / pub_id)
fruit_files = scan_pub_dir(pub_dir, pub_id)
# Resolve osdb_numbers → DB fruit IDs; warn and skip unknowns
linked = {}
for osdb_number, files in fruit_files.items():
cur.execute("SELECT id FROM fruits WHERE osdb_number = %s", (osdb_number,))
row = cur.fetchone()
if row is None:
print(
f" WARNING: osdb_number not in fruits table, skipping: {osdb_number}",
file=sys.stderr,
)
counts["skipped_fruits"] += 1
continue
linked[osdb_number] = {"db_id": row[0], **files}
# Idempotent: clear previous linked data for this publication
cur.execute("DELETE FROM publication_fruit_images WHERE publication_id = %s", (db_pub_id,))
cur.execute("DELETE FROM publication_descriptions WHERE publication_id = %s", (db_pub_id,))
cur.execute("DELETE FROM publication_fruits WHERE publication_id = %s", (db_pub_id,))
# Re-insert linked fruits, PDFs, and images
for osdb_number, info in linked.items():
fruit_db_id = info["db_id"]
cur.execute(
"INSERT INTO publication_fruits (publication_id, fruit_id) VALUES (%s, %s)",
(db_pub_id, fruit_db_id),
)
counts["fruits_linked"] += 1
if "pdf_path" in info:
pdf_data = Path(info["pdf_path"]).read_bytes()
cur.execute(
"""
INSERT INTO publication_descriptions (publication_id, fruit_id, pdf_data)
VALUES (%s, %s, %s)
""",
(db_pub_id, fruit_db_id, psycopg2.Binary(pdf_data)),
)
counts["pdfs"] += 1
if "img_path" in info:
img_data = Path(info["img_path"]).read_bytes()
filename = Path(info["img_path"]).name
cur.execute(
"""
INSERT INTO publication_fruit_images
(publication_id, fruit_id, filename, data, image_type)
VALUES (%s, %s, %s, %s, 'fruit')
""",
(db_pub_id, fruit_db_id, filename, psycopg2.Binary(img_data)),
)
counts["images"] += 1
conn.commit()
return counts
def main():
database_url = os.environ.get("DATABASE_URL")
if not database_url:
print("Error: DATABASE_URL environment variable not set", file=sys.stderr)
sys.exit(1)
print("Connecting to database...")
conn = psycopg2.connect(database_url)
try:
print("Importing publications...")
counts = import_publications(conn)
print("\nDone.")
print(f" Publications upserted: {counts['publications']}")
print(f" Cover images loaded: {counts['covers']}")
print(f" Fruits linked: {counts['fruits_linked']}")
print(f" PDFs imported: {counts['pdfs']}")
print(f" Fruit images imported: {counts['images']}")
print(f" Fruits skipped: {counts['skipped_fruits']}")
except Exception as e:
print(f"Error: {e}", file=sys.stderr)
conn.rollback()
sys.exit(1)
finally:
conn.close()
if __name__ == "__main__":
main()
+236
View File
@@ -0,0 +1,236 @@
import os
import tempfile
import unittest
from pathlib import Path
from unittest.mock import MagicMock, call, patch
from import_publications import parse_publications, scan_pub_dir, import_publications
MINIMAL_XML_STR = """<?xml version="1.0" encoding="iso-8859-1" ?>
<root>
<obj>
<id>skip_me</id>
<name>Not a publication</name>
<author>Nobody</author>
<osw>0</osw>
</obj>
<obj>
<id>ber</id>
<name>Bernisches Stammregister</name>
<author>.</author>
<img>osdb/ber/ber_s0.jpg</img>
<osw>1</osw>
</obj>
<obj>
<id>cal</id>
<name>Obst- und Beeren</name>
<author>Calwer</author>
<img>osdb/cal/cal_s0.jpg</img>
<osw>1</osw>
</obj>
<obj>
<id>noc</id>
<name>No Cover</name>
<author>Someone</author>
<osw>1</osw>
</obj>
</root>
"""
MINIMAL_XML = MINIMAL_XML_STR.encode("iso-8859-1")
def _write_xml(tmp_dir, content=MINIMAL_XML):
p = Path(tmp_dir) / "osws.xml"
p.write_bytes(content)
return str(p)
class TestParsePublications(unittest.TestCase):
def setUp(self):
self.tmp = tempfile.mkdtemp()
self.xml_path = _write_xml(self.tmp)
def test_excludes_osw_not_1(self):
pubs = parse_publications(self.xml_path)
ids = [p["id"] for p in pubs]
self.assertNotIn("skip_me", ids)
def test_includes_osw_1(self):
pubs = parse_publications(self.xml_path)
ids = [p["id"] for p in pubs]
self.assertIn("ber", ids)
self.assertIn("cal", ids)
def test_author_dot_becomes_none(self):
pubs = parse_publications(self.xml_path)
ber = next(p for p in pubs if p["id"] == "ber")
self.assertIsNone(ber["author"])
def test_author_string_preserved(self):
pubs = parse_publications(self.xml_path)
cal = next(p for p in pubs if p["id"] == "cal")
self.assertEqual(cal["author"], "Calwer")
def test_img_path_present(self):
pubs = parse_publications(self.xml_path)
cal = next(p for p in pubs if p["id"] == "cal")
self.assertEqual(cal["img_path"], "osdb/cal/cal_s0.jpg")
def test_img_path_absent_is_none(self):
pubs = parse_publications(self.xml_path)
noc = next(p for p in pubs if p["id"] == "noc")
self.assertIsNone(noc["img_path"])
def test_title_mapped(self):
pubs = parse_publications(self.xml_path)
cal = next(p for p in pubs if p["id"] == "cal")
self.assertEqual(cal["title"], "Obst- und Beeren")
class TestScanPubDir(unittest.TestCase):
def setUp(self):
self.tmp = tempfile.mkdtemp()
def test_returns_empty_for_missing_directory(self):
result = scan_pub_dir(os.path.join(self.tmp, "nonexistent"), "xyz")
self.assertEqual(result, {})
def test_extracts_osdb_number_from_image(self):
d = Path(self.tmp)
(d / "apfel_ber_s0.jpg").write_bytes(b"img")
result = scan_pub_dir(str(d), "ber")
self.assertIn("apfel", result)
self.assertEqual(result["apfel"]["img_path"], str(d / "apfel_ber_s0.jpg"))
def test_extracts_osdb_number_from_pdf(self):
d = Path(self.tmp)
(d / "birne_cal.pdf").write_bytes(b"pdf")
result = scan_pub_dir(str(d), "cal")
self.assertIn("birne", result)
self.assertEqual(result["birne"]["pdf_path"], str(d / "birne_cal.pdf"))
def test_combines_img_and_pdf_for_same_fruit(self):
d = Path(self.tmp)
(d / "apfel_ber_s0.jpg").write_bytes(b"img")
(d / "apfel_ber.pdf").write_bytes(b"pdf")
result = scan_pub_dir(str(d), "ber")
self.assertIn("img_path", result["apfel"])
self.assertIn("pdf_path", result["apfel"])
def test_ignores_thumbnail_files(self):
d = Path(self.tmp)
(d / "apfel_ber_tn.jpg").write_bytes(b"tn")
result = scan_pub_dir(str(d), "ber")
self.assertEqual(result, {})
def test_multi_segment_osdb_number(self):
d = Path(self.tmp)
(d / "berner_grauechapfel_ber_s0.jpg").write_bytes(b"img")
result = scan_pub_dir(str(d), "ber")
self.assertIn("berner_grauechapfel", result)
class TestImportPublications(unittest.TestCase):
def _make_data_dir(self, pubs):
"""Build a temp data dir with osws.xml and osdb/<id>/ dirs."""
tmp = tempfile.mkdtemp()
lines = ['<?xml version="1.0" encoding="iso-8859-1" ?>', "<root>"]
for p in pubs:
lines.append(" <obj>")
lines.append(f" <id>{p['id']}</id>")
lines.append(f" <name>{p['name']}</name>")
author = p.get("author", "Test")
lines.append(f" <author>{author}</author>")
if p.get("img"):
lines.append(f" <img>{p['img']}</img>")
lines.append(" <osw>1</osw>")
lines.append(" </obj>")
lines.append("</root>")
Path(tmp, "osws.xml").write_bytes("\n".join(lines).encode("iso-8859-1"))
for p in pubs:
pub_dir = Path(tmp, "osdb", p["id"])
pub_dir.mkdir(parents=True, exist_ok=True)
for fruit in p.get("fruits", []):
if fruit.get("img"):
(pub_dir / f"{fruit['osdb_number']}_{p['id']}_s0.jpg").write_bytes(b"\x89PNG")
if fruit.get("pdf"):
(pub_dir / f"{fruit['osdb_number']}_{p['id']}.pdf").write_bytes(b"%PDF")
if p.get("cover_bytes"):
cover_rel = p["img"]
cover_path = Path(tmp, cover_rel)
cover_path.parent.mkdir(parents=True, exist_ok=True)
cover_path.write_bytes(p["cover_bytes"])
return tmp
def _make_conn(self, pub_db_id=42, fruit_db_ids=None):
"""Return a mock psycopg2 connection."""
conn = MagicMock()
cur = MagicMock()
conn.cursor.return_value.__enter__.return_value = cur
# fetchone side_effect: first call = pub id; subsequent = fruit lookups.
# None in fruit_db_ids → bare None (simulates no row found); int → (int,) tuple.
fruit_db_ids = fruit_db_ids or []
side_effects = [(pub_db_id,)] + [((fid,) if fid is not None else None) for fid in fruit_db_ids]
cur.fetchone.side_effect = side_effects
return conn, cur
def test_missing_cover_image_does_not_abort_import(self):
data_dir = self._make_data_dir([
{"id": "ber", "name": "Test", "img": "osdb/ber/ber_s0.jpg"},
])
# cover file NOT created → missing
conn, cur = self._make_conn(pub_db_id=1)
counts = import_publications(conn, data_dir)
self.assertEqual(counts["publications"], 1)
self.assertEqual(counts["covers"], 0)
def test_fruit_not_in_db_is_skipped_with_warning(self):
data_dir = self._make_data_dir([
{"id": "cal", "name": "Cal", "fruits": [
{"osdb_number": "unknown_fruit", "img": True},
]},
])
conn, cur = self._make_conn(pub_db_id=5, fruit_db_ids=[None])
counts = import_publications(conn, data_dir)
self.assertEqual(counts["skipped_fruits"], 1)
self.assertEqual(counts["fruits_linked"], 0)
def test_happy_path_upserts_pub_links_fruits_imports_files(self):
data_dir = self._make_data_dir([
{
"id": "pom",
"name": "Pomologie",
"author": "Diel",
"img": "osdb/pom/pom_s0.jpg",
"cover_bytes": b"\x89PNG",
"fruits": [
{"osdb_number": "apfelsorte_alpha", "img": True, "pdf": True},
],
}
])
conn, cur = self._make_conn(pub_db_id=7, fruit_db_ids=[99])
counts = import_publications(conn, data_dir)
self.assertEqual(counts["publications"], 1)
self.assertEqual(counts["covers"], 1)
self.assertEqual(counts["fruits_linked"], 1)
self.assertEqual(counts["pdfs"], 1)
self.assertEqual(counts["images"], 1)
self.assertEqual(counts["skipped_fruits"], 0)
def test_osw_not_1_never_imported(self):
tmp = tempfile.mkdtemp()
xml = b"""<?xml version="1.0" encoding="iso-8859-1"?>
<root>
<obj><id>skip</id><name>Skip</name><author>X</author><osw>0</osw></obj>
</root>"""
Path(tmp, "osws.xml").write_bytes(xml)
conn, cur = self._make_conn()
counts = import_publications(conn, tmp)
self.assertEqual(counts["publications"], 0)
cur.execute.assert_not_called()
if __name__ == "__main__":
unittest.main()