Coverage for gws-app/gws/plugin/alkis/data/indexer.py: 0%
683 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-05 13:35 +0200
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-05 13:35 +0200
1"""Build the ALKIS index from source data."""
3from typing import Optional, Iterable
5import re
6from typing import Generic, TypeVar
8import shapely
9import shapely.strtree
10import shapely.wkb
12import gws
13import gws.lib.osx
14import gws.lib.datetimex as dtx
15from gws.lib.cli import ProgressIndicator
16import gws.plugin.postgres.provider
18from . import types as dt
19from . import index
20from . import norbit6
22from .geo_info_dok import gid6 as gid
25def run(ix: index.Object, data_schema: str, with_force=False, with_cache=False):
26 """Build the ALKIS index.
28 Reads the source tables with the norBIT GeoInfoDok 6 reader and writes
29 all index tables that do not have data yet. Does nothing if the index is
30 already complete, unless ``with_force`` is set.
32 Args:
33 ix: Index to build.
34 data_schema: Schema with the ALKIS source tables.
35 with_force: Drop all index tables before building.
36 with_cache: Cache source data and collected objects in the cache
37 directory and reuse them in later runs.
39 Raises:
40 ``gws.Error``: If the index schema does not exist.
41 """
43 if not ix.has_schema():
44 raise gws.Error(f'ALKIS: schema {ix.schema!r} does not exist')
46 if with_force:
47 ix.drop()
48 elif ix.status().complete:
49 return
51 rdr = norbit6.Object(ix.db, schema=data_schema)
52 rr = _Runner(ix, rdr, with_cache)
53 rr.run()
56##
58T = TypeVar("T")
61class _ObjectDict(Generic[T]):
62 """Collection of entities of one type, by uid."""
64 def __init__(self, cls):
65 """Create an empty collection.
67 Args:
68 cls: Entity class.
69 """
71 self.d = {}
72 self.cls = cls
74 def add(self, uid, recs) -> T:
75 """Create an entity and add it to the collection.
77 The entity is historic if all of its records are.
79 Args:
80 uid: Entity uid.
81 recs: Entity records.
83 Returns:
84 The new entity.
85 """
87 o = self.cls(uid=uid, recs=recs)
88 o.isHistoric = all(r.isHistoric for r in recs)
89 self.d[o.uid] = o
90 return o
92 def get(self, uid, default=None) -> Optional[T]:
93 """Return an entity by uid.
95 Args:
96 uid: Entity uid.
97 default: Value to return if the uid is not found.
99 Returns:
100 The entity or the default value.
101 """
103 return self.d.get(uid, default)
105 def get_many(self, uids) -> list[T]:
106 """Return entities by uids.
108 Unknown uids are skipped, duplicates are returned once.
110 Args:
111 uids: Entity uids.
113 Returns:
114 A list of entities, in the order of ``uids``.
115 """
117 res = {}
119 for uid in uids:
120 if uid not in res:
121 o = self.d.get(uid)
122 if o:
123 res[uid] = o
125 return list(res.values())
127 def get_from_ptr(self, obj: dt.Entity, attr):
128 """Return the entities referenced by an attribute of the records of an entity.
130 The attribute is removed from the records.
132 Args:
133 obj: Entity whose records hold the references.
134 attr: Attribute name, holding a uid or a list of uids.
136 Returns:
137 A list of referenced entities.
138 """
140 uids = []
142 for r in obj.recs:
143 v = _pop(r, attr)
144 if isinstance(v, list):
145 uids.extend(v)
146 elif isinstance(v, str):
147 uids.append(v)
149 return self.get_many(uids)
151 def __iter__(self) -> Iterable[T]:
152 yield from self.d.values()
154 def __len__(self):
155 return len(self.d)
158class _ObjectMap:
159 """Collections of all entities collected by an indexer."""
161 def __init__(self):
162 self.Anschrift: _ObjectDict[dt.Anschrift] = _ObjectDict(dt.Anschrift)
163 self.Buchungsblatt: _ObjectDict[dt.Buchungsblatt] = _ObjectDict(dt.Buchungsblatt)
164 self.Buchungsstelle: _ObjectDict[dt.Buchungsstelle] = _ObjectDict(dt.Buchungsstelle)
165 self.Flurstueck: _ObjectDict[dt.Flurstueck] = _ObjectDict(dt.Flurstueck)
166 self.Gebaeude: _ObjectDict[dt.Gebaeude] = _ObjectDict(dt.Gebaeude)
167 self.Lage: _ObjectDict[dt.Lage] = _ObjectDict(dt.Lage)
168 self.Namensnummer: _ObjectDict[dt.Namensnummer] = _ObjectDict(dt.Namensnummer)
169 self.Part: _ObjectDict[dt.Part] = _ObjectDict(dt.Part)
170 self.Person: _ObjectDict[dt.Person] = _ObjectDict(dt.Person)
172 self.placeAll: dict = {}
173 self.placeIdx: dict = {}
174 self.catalog: dict = {}
177class _Indexer:
178 """Base class for indexers.
180 An indexer collects entities of some kinds from the source data into its
181 object map and writes them into index tables. The object map can be cached
182 between runs.
183 """
185 CACHE_KEY: str = ''
186 """Cache file name, caching is disabled if empty."""
188 def __init__(self, runner: '_Runner'):
189 """Create an indexer.
191 Args:
192 runner: Runner that owns this indexer.
193 """
195 self.rr = runner
196 self.ix: index.Object = runner.ix
197 self.om = _ObjectMap()
199 def load_or_collect(self):
200 """Load the object map from the cache, or collect it and store it in the cache."""
202 if not self.load_cache():
203 self.collect()
204 self.store_cache()
206 def load_cache(self):
207 """Load the object map from the cache.
209 Returns:
210 ``True`` if the object map was loaded.
211 """
213 if not self.rr.withCache or not self.CACHE_KEY:
214 return False
215 cpath = self.rr.cacheDir + '/' + self.CACHE_KEY
216 if not gws.u.is_file(cpath):
217 return False
218 om = gws.u.unserialize_from_path(cpath)
219 if not om:
220 return False
221 gws.log.info(f'ALKIS: use cache {self.CACHE_KEY!r}')
222 self.om = om
223 return True
225 def store_cache(self):
226 """Store the object map in the cache, if caching is enabled."""
228 if not self.rr.withCache or not self.CACHE_KEY:
229 return
230 cpath = self.rr.cacheDir + '/' + self.CACHE_KEY
231 gws.u.serialize_to_path(self.om, cpath)
232 gws.log.info(f'ALKIS: store cache {self.CACHE_KEY!r}')
234 def collect(self):
235 """Collect entities from the source data into the object map."""
237 pass
239 def write_table(self, table_id, values):
240 """Create and fill an index table, unless it already has data.
242 Args:
243 table_id: Table id.
244 values: Rows as dicts.
245 """
247 if self.ix.has_table(table_id):
248 return
249 with ProgressIndicator(f'ALKIS: write {table_id!r}', len(values)) as progress:
250 self.ix.create_table(table_id, values, progress)
252 def write(self):
253 """Write the collected entities into the index tables."""
255 pass
258class _PlaceIndexer(_Indexer):
259 """Indexer for places (Administrations- und Verwaltungseinheiten).
261 Collects current Laender, Regierungsbezirke, Kreise, Gemeinden, Gemarkungen,
262 Buchungsblattbezirke and Dienststellen. Place codes are built from the key
263 parts as in the official municipality key, see
264 https://de.wikipedia.org/wiki/Amtlicher_Gemeindeschl%C3%BCssel
265 """
267 CACHE_KEY = 'obj_place'
269 empty1 = dt.EnumPair(code='0', text='')
270 """Empty place value for one-digit codes."""
271 empty2 = dt.EnumPair(code='00', text='')
272 """Empty place value for two-digit codes."""
274 def add(self, kind, ax, key_obj, **kwargs):
275 """Add a place, unless it is historic.
277 Args:
278 kind: Place kind, a ``PlaceKind`` value.
279 ax: Source object.
280 key_obj: Key object of the place.
281 **kwargs: Higher level places.
283 Returns:
284 The place value, or ``None`` if the place is historic.
285 """
287 if ax.lebenszeitintervall.endet is not None:
288 return
290 code = self.code(kind, key_obj)
291 value = dt.EnumPair(code, ax.bezeichnung)
293 p = dt.Place(**kwargs)
295 p.uid = kind + code
296 p.kind = kind
297 setattr(p, kind, value)
299 self.om.placeAll[p.uid] = p
300 self.om.placeIdx[p.uid] = value
302 return value
304 def collect(self):
305 self.om.placeAll = {}
306 self.om.placeIdx = {}
308 for ax in self.rr.read_flat(gid.AX_Bundesland):
309 self.add('land', ax, ax.schluessel)
311 for ax in self.rr.read_flat(gid.AX_Regierungsbezirk):
312 o = ax.schluessel
313 self.add('regierungsbezirk', ax, o, land=self.get_land(o))
315 for ax in self.rr.read_flat(gid.AX_KreisRegion):
316 o = ax.schluessel
317 self.add('kreis', ax, o, land=self.get_land(o), regierungsbezirk=self.get_regierungsbezirk(o))
319 for ax in self.rr.read_flat(gid.AX_Gemeinde):
320 o = ax.gemeindekennzeichen
321 self.add('gemeinde', ax, o, land=self.get_land(o), regierungsbezirk=self.get_regierungsbezirk(o), kreis=self.get_kreis(o))
323 # @TODO map Gemarkung to Gemeinde (see https://de.wikipedia.org/wiki/Liste_der_Gemarkungen_in_Nordrhein-Westfalen etc)
325 for ax in self.rr.read_flat(gid.AX_Gemarkung):
326 if str(ax.schluessel.gemarkungsnummer) in self.ix.excludeGemarkung:
327 continue
328 o = ax.schluessel
329 self.add('gemarkung', ax, o, land=self.get_land(o))
331 for ax in self.rr.read_flat(gid.AX_Buchungsblattbezirk):
332 o = ax.schluessel
333 self.add('buchungsblattbezirk', ax, o, land=self.get_land(o))
335 for ax in self.rr.read_flat(gid.AX_Dienststelle):
336 o = ax.schluessel
337 self.add('dienststelle', ax, o, land=self.get_land(o))
339 def write(self):
340 values = []
342 for place in self.om.placeAll.values():
343 values.append(dict(
344 uid=place.uid,
345 data=index.serialize(place),
346 ))
348 self.write_table(index.TABLE_PLACE, values)
350 def get_land(self, o):
351 """Return the Land for a key object.
353 Args:
354 o: Key object.
356 Returns:
357 The place value, or an empty value if not found.
358 """
360 return self.get('land', o) or self.empty2
362 def get_regierungsbezirk(self, o):
363 """Return the Regierungsbezirk for a key object.
365 Args:
366 o: Key object.
368 Returns:
369 The place value, or an empty value if not found.
370 """
372 return self.get('regierungsbezirk', o) or self.empty1
374 def get_kreis(self, o):
375 """Return the Kreis for a key object.
377 Args:
378 o: Key object.
380 Returns:
381 The place value, or an empty value if not found.
382 """
384 return self.get('kreis', o) or self.empty2
386 def get_gemeinde(self, o):
387 """Return the Gemeinde for a key object.
389 Args:
390 o: Key object.
392 Returns:
393 The place value, or an empty value if not found.
394 """
396 return self.get('gemeinde', o) or self.empty1
398 def get_gemarkung(self, o):
399 """Return the Gemarkung for a key object.
401 Args:
402 o: Key object.
404 Returns:
405 The place value, or an empty value if not found.
406 """
408 return self.get('gemarkung', o) or self.empty1
410 def get_buchungsblattbezirk(self, o):
411 """Return the Buchungsblattbezirk for a key object.
413 Args:
414 o: Key object.
416 Returns:
417 The place value, or an empty value if not found.
418 """
420 return self.get('buchungsblattbezirk', o) or self.empty1
422 def get_dienststelle(self, o):
423 """Return the Dienststelle for a key object.
425 Args:
426 o: Key object.
428 Returns:
429 The place value, or an empty value if not found.
430 """
432 return self.get('dienststelle', o) or self.empty1
434 def get(self, kind, o):
435 """Return a place by kind and key object.
437 Args:
438 kind: Place kind.
439 o: Key object.
441 Returns:
442 The place value, or ``None`` if not found.
443 """
445 return self.om.placeIdx.get(kind + self.code(kind, o))
447 def is_empty(self, p: dt.EnumPair):
448 """Check whether a place value is empty.
450 Args:
451 p: Place value.
453 Returns:
454 ``True`` if the code is ``0`` or ``00``.
455 """
457 return p.code == '0' or p.code == '00'
459 CODES = {
460 'land': lambda o: o.land,
461 'regierungsbezirk': lambda o: o.land + (o.regierungsbezirk or '0'),
462 'kreis': lambda o: o.land + (o.regierungsbezirk or '0') + o.kreis,
463 'gemeinde': lambda o: o.land + (o.regierungsbezirk or '0') + o.kreis + o.gemeinde,
464 'gemarkung': lambda o: o.land + o.gemarkungsnummer,
465 'buchungsblattbezirk': lambda o: o.land + o.bezirk,
466 'dienststelle': lambda o: o.land + o.stelle,
467 }
468 """Functions that build a place code from a key object, by place kind."""
470 def code(self, kind, o):
471 """Build a place code.
473 Args:
474 kind: Place kind.
475 o: Key object.
477 Returns:
478 The place code.
479 """
481 return self.CODES[kind](o)
484class _LageIndexer(_Indexer):
485 """Indexer for location designations (Lage) and buildings (Gebaeude).
487 The coordinates of a Lage are taken from its house number label (``AP_PTO``
488 with ``art=HNR``). Buildings are linked to the Lage they point to.
489 Building geometries are not stored.
490 """
492 CACHE_KEY = 'obj_lage'
494 def collect(self):
495 for ax in self.rr.read_flat(gid.AX_LagebezeichnungKatalogeintrag):
496 self.om.catalog[self.lage_key(ax.schluessel)] = ax.bezeichnung
498 for cls in (gid.AX_LagebezeichnungMitHausnummer, gid.AX_LagebezeichnungOhneHausnummer):
499 for uid, axs in self.rr.read_grouped(cls):
500 self.om.Lage.add(uid, [
501 _from_ax(
502 dt.LageRecord,
503 ax,
504 strasse=self.strasse(ax),
505 hausnummer=index.normalize_hausnummer(ax.hausnummer),
506 )
507 for ax in axs
508 ])
510 # use the PTO (art=HNR) geometry for lage coordinates
511 # PTO.dientZurDarstellungVon -> lage.uid
513 for pto in self.rr.read_flat(gid.AP_PTO):
514 if pto.lebenszeitintervall.endet is not None:
515 continue
516 art = _pop(pto, 'art') or ''
517 if art.upper() != 'HNR':
518 continue
520 uids = _pop(pto, 'dientZurDarstellungVon')
521 if not uids or not isinstance(uids, list):
522 continue
524 geom = _geom_of(pto)
525 if not geom:
526 continue
528 x = geom.centroid.x
529 y = geom.centroid.y
530 for la in self.om.Lage.get_many(uids):
531 la.x = x
532 la.y = y
534 # read related Gebaeude records
536 atts = _meta_attributes(gid.METADATA['AX_Gebaeude'])
538 for uid, axs in self.rr.read_grouped(gid.AX_Gebaeude):
539 self.om.Gebaeude.add(uid, [
540 _from_ax(
541 dt.GebaeudeRecord,
542 ax,
543 name=', '.join(ax.name) if ax.name else None,
544 amtlicheFlaeche=ax.grundflaeche or 0,
545 props=self.rr.props_from(ax, atts),
546 _zeigtAuf=ax.zeigtAuf,
547 )
548 for ax in axs
549 ])
551 for ge in self.om.Gebaeude:
552 for r in ge.recs:
553 geom = _geom_of(r)
554 r.geomFlaeche = round(geom.area, 2) if geom else 0
556 # omit Gebaeude geometries for now
557 for ge in self.om.Gebaeude:
558 for r in ge.recs:
559 _pop(r, 'geom')
561 for la in self.om.Lage:
562 la.gebaeudeList = []
564 # AX_Gebaeude.zeigtAuf -> AX_LagebezeichnungMitHausnummer
565 for ge in self.om.Gebaeude:
566 for la in self.om.Lage.get_from_ptr(ge, '_zeigtAuf'):
567 la.gebaeudeList.append(ge)
569 def strasse(self, ax):
570 """Return the street name of a source Lage object.
572 Args:
573 ax: Source Lage object.
575 Returns:
576 The unencoded street name, or the name from the catalog for an encoded one.
577 """
579 if isinstance(ax.lagebezeichnung, str):
580 return ax.lagebezeichnung
581 return self.om.catalog.get(self.lage_key(ax.lagebezeichnung), '')
583 def lage_key(self, r):
584 """Return the catalog key for an encoded location.
586 Args:
587 r: Encoded location or catalog key object.
589 Returns:
590 A string of the key parts joined with commas.
591 """
593 return _comma([
594 getattr(r, 'land'),
595 getattr(r, 'regierungsbezirk'),
596 getattr(r, 'kreis'),
597 getattr(r, 'gemeinde'),
598 getattr(r, 'lage'),
599 ])
601 def write(self):
602 values = []
604 for la in self.om.Lage:
605 values.append(dict(
606 uid=la.uid,
607 rc=len(la.recs),
608 data=index.serialize(la),
609 ))
611 self.write_table(index.TABLE_LAGE, values)
614class _BuchungIndexer(_Indexer):
615 """Indexer for land register data.
617 Collects Buchungsblaetter with their Buchungsstellen and Namensnummern,
618 Persons and their Anschriften, and the parent-child relations of
619 Buchungsstellen.
620 """
622 CACHE_KEY = 'obj_buchungsblatt'
624 buchungsblattkennzeichenMap: dict[str, dt.Buchungsblatt] = {}
625 """Buchungsblaetter by their identifier."""
627 def collect(self):
628 for uid, axs in self.rr.read_grouped(gid.AX_Anschrift):
629 self.om.Anschrift.add(uid, [
630 _from_ax(
631 dt.AnschriftRecord,
632 ax,
633 ort=ax.ort_AmtlichesOrtsnamensverzeichnis or ax.ort_Post,
634 plz=ax.postleitzahlPostzustellung,
635 telefon=ax.telefon[0] if ax.telefon else None
636 )
637 for ax in axs
638 ])
640 for uid, axs in self.rr.read_grouped(gid.AX_Person):
641 self.om.Person.add(uid, [
642 _from_ax(
643 dt.PersonRecord,
644 ax,
645 anrede=ax.anrede.text if ax.anrede else None,
646 _hat=ax.hat,
647 )
648 for ax in axs
649 ])
651 # AX_Person.hat -> [AX_Anschrift]
652 for pe in self.om.Person:
653 pe.anschriftList = self.om.Anschrift.get_from_ptr(pe, '_hat')
655 for uid, axs in self.rr.read_grouped(gid.AX_Namensnummer):
656 self.om.Namensnummer.add(uid, [
657 _from_ax(
658 dt.NamensnummerRecord,
659 ax,
660 anteil=_anteil(ax),
661 _benennt=ax.benennt,
662 _istBestandteilVon=ax.istBestandteilVon
663 )
664 for ax in axs
665 ])
667 # AX_Namensnummer.benennt -> AX_Person
668 for nn in self.om.Namensnummer:
669 nn.laufendeNummer = nn.recs[-1].laufendeNummerNachDIN1421
670 nn.personList = self.om.Person.get_from_ptr(nn, '_benennt')
672 for uid, axs in self.rr.read_grouped(gid.AX_Buchungsstelle):
673 self.om.Buchungsstelle.add(uid, [
674 _from_ax(
675 dt.BuchungsstelleRecord,
676 ax,
677 anteil=_anteil(ax),
678 _an=ax.an,
679 _zu=ax.zu,
680 _istBestandteilVon=ax.istBestandteilVon,
681 )
682 for ax in axs
683 ])
685 for bs in self.om.Buchungsstelle:
686 bs.laufendeNummer = bs.recs[-1].laufendeNummer
687 bs.fsUids = []
688 bs.flurstueckskennzeichenList = []
690 for uid, axs in self.rr.read_grouped(gid.AX_Buchungsblatt):
691 self.om.Buchungsblatt.add(uid, [
692 _from_ax(
693 dt.BuchungsblattRecord,
694 ax,
695 buchungsblattbezirk=self.rr.place.get_buchungsblattbezirk(ax.buchungsblattbezirk),
696 )
697 for ax in axs
698 ])
700 for bb in self.om.Buchungsblatt:
701 bb.buchungsstelleList = []
702 bb.namensnummerList = []
703 bb.buchungsblattkennzeichen = bb.recs[-1].buchungsblattkennzeichen
704 self.buchungsblattkennzeichenMap[bb.buchungsblattkennzeichen] = bb
706 # AX_Buchungsstelle.istBestandteilVon -> AX_Buchungsblatt
707 for bs in self.om.Buchungsstelle:
708 bb_list = self.om.Buchungsblatt.get_from_ptr(bs, '_istBestandteilVon')
709 bs.buchungsblattUids = [bb.uid for bb in bb_list]
710 bs.buchungsblattkennzeichenList = [bb.buchungsblattkennzeichen for bb in bb_list]
711 for bb in bb_list:
712 bb.buchungsstelleList.append(bs)
714 # AX_Namensnummer.istBestandteilVon -> AX_Buchungsblatt
715 for nn in self.om.Namensnummer:
716 bb_list = self.om.Buchungsblatt.get_from_ptr(nn, '_istBestandteilVon')
717 nn.buchungsblattUids = [bb.uid for bb in bb_list]
718 nn.buchungsblattkennzeichenList = [bb.buchungsblattkennzeichen for bb in bb_list]
719 for bb in bb_list:
720 bb.namensnummerList.append(nn)
722 for bb in self.om.Buchungsblatt:
723 bb.buchungsstelleList.sort(key=_sortkey_buchungsstelle)
724 bb.namensnummerList.sort(key=_sortkey_namensnummer)
726 # AX_Buchungsstelle.an -> [AX_Buchungsstelle]
727 # AX_Buchungsstelle.zu -> [AX_Buchungsstelle]
728 # see Erläuterungen zu ALKIS Version 6, page 116-119
730 for bs in self.om.Buchungsstelle:
731 bs.childUids = []
732 bs.parentUids = []
733 bs.parentkennzeichenList = []
735 for bs in self.om.Buchungsstelle:
736 parent_uids = set()
737 parent_knz = set()
739 for r in bs.recs:
740 parent_uids.update(_pop(r, '_an'))
741 parent_uids.update(_pop(r, '_zu'))
743 for parent_bs in self.om.Buchungsstelle.get_many(parent_uids):
744 parent_bs.childUids.append(bs.uid)
745 bs.parentUids.append(parent_bs.uid)
746 for bb_knz in parent_bs.buchungsblattkennzeichenList:
747 parent_knz.add(bb_knz + '.' + parent_bs.laufendeNummer)
749 bs.parentkennzeichenList = sorted(parent_knz)
751 def write(self):
752 values = []
754 for bb in self.om.Buchungsblatt:
755 values.append(dict(
756 uid=bb.uid,
757 rc=len(bb.recs),
758 data=index.serialize(bb),
759 ))
761 self.write_table(index.TABLE_BUCHUNGSBLATT, values)
764class _PartIndexer(_Indexer):
765 """Indexer for Parts (Nutzung, Festlegung, Bewertung).
767 Reads all object types with a geometry in the Part categories and
768 intersects their geometries with the most recent Flurstueck geometries.
769 Intersections smaller than ``MIN_PART_AREA`` are skipped.
770 """
772 CACHE_KEY = 'obj_part'
773 MIN_PART_AREA = 1
774 """Minimum area of an intersection."""
776 parts: list[dt.Part] = []
777 """Computed parts."""
779 fs_list = []
780 """Flurstuecke, in the order of ``fs_geom``."""
781 fs_geom = []
782 """Most recent Flurstueck geometries."""
784 stree: shapely.strtree.STRtree
785 """Spatial index of ``fs_geom``."""
787 def collect(self):
789 for kind in dt.Part.KIND:
790 self.collect_kind(kind)
792 for fs in self.rr.fsdata.om.Flurstueck:
793 self.fs_list.append(fs)
794 # NB take only the most recent fs geometry into account
795 self.fs_geom.append(_geom_of(fs.recs[-1]))
797 self.stree = shapely.strtree.STRtree(self.fs_geom)
799 with ProgressIndicator(f'ALKIS: parts', len(self.om.Part)) as progress:
800 for pa in self.om.Part:
801 self.compute_intersections(pa)
802 progress.update(1)
804 self.parts.sort(key=_sortkey_part)
806 for pa in self.parts:
807 pa.isHistoric = all(r.isHistoric for r in pa.recs)
809 def collect_kind(self, kind):
810 """Collect source objects of all object types of a Part kind.
812 Args:
813 kind: Part kind.
814 """
816 _, key = dt.Part.KIND[kind]
817 classes = [
818 getattr(gid, meta['name'])
819 for meta in gid.METADATA.values()
820 if (
821 meta['kind'] == 'object'
822 and meta['geom']
823 and re.search(key + r'/\w+/', meta['key'])
824 )
825 ]
827 for cls in classes:
828 self.collect_class(kind, cls)
830 def collect_class(self, kind, cls):
831 """Collect source objects of one object type.
833 Args:
834 kind: Part kind.
835 cls: GeoInfoDok class from ``gid6``.
836 """
838 meta = gid.METADATA[cls.__name__]
839 atts = _meta_attributes(meta)
841 for uid, axs in self.rr.read_grouped(cls):
842 pa = self.om.Part.add(uid, [
843 _from_ax(
844 dt.PartRecord,
845 ax,
846 props=self.rr.props_from(ax, atts),
847 )
848 for ax in axs
849 ])
850 pa.kind = kind
851 pa.name = dt.EnumPair(meta['uid'], meta['title'])
853 def compute_intersections(self, pa: dt.Part):
854 """Intersect a source object with the Flurstuecke and add the resulting parts.
856 Args:
857 pa: Source object as a Part.
858 """
860 parts_map = {}
862 for r in pa.recs:
863 geom = _geom_of(r)
864 if not geom:
865 continue
867 for i in self.stree.query(geom):
868 part_geom = shapely.intersection(self.fs_geom[i], geom)
869 part_area = round(part_geom.area, 2)
870 if part_area < self.MIN_PART_AREA:
871 continue
873 fs = self.fs_list[i]
875 part = parts_map.setdefault(fs.uid, dt.Part(
876 uid=pa.uid,
877 recs=[],
878 kind=pa.kind,
879 name=pa.name,
880 fs=fs.uid,
881 ))
883 # computed area corrected with respect to FS's "amtlicheFlaeche"
884 part_area_corrected = round(
885 fs.recs[-1].amtlicheFlaeche * (part_area / fs.recs[-1].geomFlaeche),
886 2)
888 part.recs.append(dt.PartRecord(
889 uid=r.uid,
890 beginnt=r.beginnt,
891 endet=r.endet,
892 anlass=r.anlass,
893 props=r.props,
894 geomFlaeche=part_area,
895 amtlicheFlaeche=part_area_corrected,
896 isHistoric=r.endet is not None,
897 ))
899 part.geom = shapely.wkb.dumps(part_geom, srid=self.ix.crs.srid, hex=True)
900 part.geomFlaeche = part_area
901 part.amtlicheFlaeche = part_area_corrected
903 self.parts.extend(parts_map.values())
905 def write(self):
906 values = []
908 for n, pa in enumerate(self.parts, 1):
909 geom = _pop(pa, 'geom')
910 data = index.serialize(pa)
911 pa.geom = geom
912 values.append(dict(
913 n=n,
914 fs=pa.fs,
915 uid=pa.uid,
916 beginnt=pa.recs[-1].beginnt,
917 endet=pa.recs[-1].endet,
918 kind=pa.kind,
919 name=pa.name.text,
920 parthistoric=pa.isHistoric,
921 data=data,
922 geom=geom,
923 ))
925 self.write_table(index.TABLE_PART, values)
928class _FsDataIndexer(_Indexer):
929 """Indexer for Flurstuecke.
931 Collects current and historic Flurstuecke and links them to their Lage,
932 Gebaeude and Buchung data. Flurstuecke without a geometry or with an unknown
933 Gemarkung or Gemeinde are skipped and counted in ``counts``. Also fills in
934 the predecessor lists from the successor lists.
935 """
937 CACHE_KEY = 'obj_flurstueck'
939 def __init__(self, runner: '_Runner'):
940 super().__init__(runner)
941 self.counts = {
942 'ok': 0,
943 'no_geometry': 0,
944 'excluded': 0,
945 }
947 def collect(self):
948 for uid, axs in self.rr.read_grouped(gid.AX_Flurstueck):
949 recs = gws.u.compact(self.record(ax) for ax in axs)
950 if recs:
951 self.om.Flurstueck.add(uid, recs)
953 for uid, axs in self.rr.read_grouped(gid.AX_HistorischesFlurstueck):
954 recs = gws.u.compact(self.record(ax) for ax in axs)
955 if not recs:
956 continue
957 # For a historic FS, 'beginnt' is basically when the history beginnt
958 # (see comments for AX_HistorischesFlurstueck in gid6).
959 # we set F.endet=F.beginnt to designate this one as 'historic'
960 for r in recs:
961 r.endet = r.beginnt
962 r.isHistoric = True
963 self.om.Flurstueck.add(uid, recs)
965 for fs in self.om.Flurstueck:
966 fs.flurstueckskennzeichen = fs.recs[-1].flurstueckskennzeichen
967 fs.fsnummer = index.make_fsnummer(fs.recs[-1])
968 fs.x = fs.recs[-1].x
969 fs.y = fs.recs[-1].y
970 self.process_lage(fs)
971 self.process_gebaeude(fs)
972 self.process_buchung(fs)
974 # check the 'nachfolgerFlurstueckskennzeichen' array
975 # and mark each referenced FS as a "vorgaenger" FS
976 # It is a M:N relation, therefore 'vorgaengerFlurstueckskennzeichen' is also an array
978 knz_to_fs = {
979 fs.flurstueckskennzeichen: fs
980 for fs in self.om.Flurstueck
981 }
982 for fs in self.om.Flurstueck:
983 nfs = fs.recs[-1].nachfolgerFlurstueckskennzeichen
984 if not nfs:
985 continue
986 for nf_knz in nfs:
987 nf_fs = knz_to_fs.get(nf_knz)
988 if not nf_fs:
989 gws.log.warning(f'ALKIS: nachfolgerFlurstueck missing {fs.flurstueckskennzeichen!r}->{nf_knz!r}')
990 continue
991 if not nf_fs.recs[-1].vorgaengerFlurstueckskennzeichen:
992 nf_fs.recs[-1].vorgaengerFlurstueckskennzeichen = []
993 nf_fs.recs[-1].vorgaengerFlurstueckskennzeichen.append(fs.flurstueckskennzeichen)
995 def record(self, ax):
996 """Create a Flurstueck record from a source object.
998 Args:
999 ax: Source Flurstueck object.
1001 Returns:
1002 The record, or ``None`` if the Flurstueck has no geometry or an unknown Gemarkung or Gemeinde.
1003 """
1005 r: dt.FlurstueckRecord = _from_ax(
1006 dt.FlurstueckRecord,
1007 ax,
1008 amtlicheFlaeche=ax.amtlicheFlaeche or 0,
1009 flurnummer=_str(ax.flurnummer),
1010 zaehler=_str(ax.flurstuecksnummer.zaehler),
1011 nenner=_str(ax.flurstuecksnummer.nenner),
1012 zustaendigeStelle=[self.rr.place.get_dienststelle(p) for p in (ax.zustaendigeStelle or [])],
1014 _weistAuf=ax.weistAuf,
1015 _zeigtAuf=ax.zeigtAuf,
1016 _istGebucht=ax.istGebucht,
1017 _buchung=ax.buchung,
1018 )
1020 # basic data
1022 r.gemarkung = self.rr.place.get_gemarkung(ax.gemarkung)
1023 r.gemeinde = self.rr.place.get_gemeinde(ax.gemeindezugehoerigkeit)
1024 r.regierungsbezirk = self.rr.place.get_regierungsbezirk(ax.gemeindezugehoerigkeit)
1025 r.kreis = self.rr.place.get_kreis(ax.gemeindezugehoerigkeit)
1026 r.land = self.rr.place.get_land(ax.gemeindezugehoerigkeit)
1028 if self.rr.place.is_empty(r.gemarkung) or self.rr.place.is_empty(r.gemeinde):
1029 # exclude Flurstücke that refer to Gemeinde/Gemarkung
1030 # which do not exist in the reference AX tables
1031 self.counts['excluded'] += 1
1032 return None
1034 # geometry
1036 geom = _geom_of(r)
1037 if not geom:
1038 self.counts['no_geometry'] += 1
1039 return None
1041 r.geomFlaeche = round(geom.area, 2)
1042 r.x = round(geom.centroid.x, 2)
1043 r.y = round(geom.centroid.y, 2)
1045 self.counts['ok'] += 1
1046 return r
1048 def process_lage(self, fs: dt.Flurstueck):
1049 """Set the Lage list of a Flurstueck.
1051 Args:
1052 fs: Flurstueck.
1053 """
1055 fs.lageList = []
1057 # AX_Flurstueck.weistAuf -> AX_LagebezeichnungMitHausnummer
1058 # AX_Flurstueck.zeigtAuf -> AX_LagebezeichnungOhneHausnummer
1059 fs.lageList.extend(self.rr.lage.om.Lage.get_from_ptr(fs, '_weistAuf'))
1060 fs.lageList.extend(self.rr.lage.om.Lage.get_from_ptr(fs, '_zeigtAuf'))
1062 def process_gebaeude(self, fs: dt.Flurstueck):
1063 """Set the Gebaeude list and total building areas of a Flurstueck.
1065 Args:
1066 fs: Flurstueck, with its Lage list set.
1067 """
1069 ge_map = {}
1071 for la in fs.lageList:
1072 for ge in la.gebaeudeList:
1073 ge_map[ge.uid] = ge
1075 fs.gebaeudeList = list(ge_map.values())
1076 fs.gebaeudeList.sort(key=_sortkey_gebaeude)
1078 fs.gebaeudeAmtlicheFlaeche = sum(ge.recs[-1].amtlicheFlaeche for ge in fs.gebaeudeList if not ge.recs[-1].endet)
1079 fs.gebaeudeGeomFlaeche = sum(ge.recs[-1].geomFlaeche for ge in fs.gebaeudeList if not ge.recs[-1].endet)
1081 def process_buchung(self, fs: dt.Flurstueck):
1082 """Set the Buchung list of a Flurstueck.
1084 Collects the Buchungsstellen of all Flurstueck records, including their
1085 parents and children, and groups them by Buchungsblatt.
1087 Args:
1088 fs: Flurstueck.
1090 Returns:
1091 The Flurstueck.
1092 """
1094 bs_historic_map = {}
1095 bs_seen = set()
1096 buchung_map = {}
1098 # for each Flurstück record, we collect all related Buchungsstellen (with respect to parent-child relations)
1099 # then group Buchungsstellen by their Buchungsblatt
1100 # and create Buchung objects for a FS
1102 for r in fs.recs:
1103 hist_buchung = _pop(r, '_buchung')
1104 if hist_buchung:
1105 bs_list = self.historic_buchungsstelle_list(r, hist_buchung)
1106 else:
1107 bs_list = self.buchungsstelle_list(r)
1109 for bs in bs_list:
1110 # a Buchungsstelle referred to by an expired Flurstück might not be expired itself,
1111 # so we have to track its state separately by wrapping it in a BuchungsstelleReference
1112 # a BuchungsstelleReference is historic if its Flurstück is
1113 bs_historic_map[bs.uid] = r.isHistoric
1115 if bs.uid in bs_seen:
1116 continue
1117 bs_seen.add(bs.uid)
1119 # populate Flurstück references in a Buchungsstelle
1120 if fs.uid not in bs.fsUids:
1121 bs.fsUids.append(fs.uid)
1122 bs.flurstueckskennzeichenList.append(fs.flurstueckskennzeichen)
1124 # create Buchung records by grouping Buchungsstellen
1125 for bb_uid in bs.buchungsblattUids:
1126 bu = buchung_map.setdefault(bb_uid, dt.Buchung(recs=[], buchungsblattUid=bb_uid))
1127 bu.recs.append(dt.BuchungsstelleReference(buchungsstelle=bs))
1129 fs.buchungList = list(buchung_map.values())
1131 for bu in fs.buchungList:
1132 for ref in bu.recs:
1133 ref.isHistoric = bs_historic_map[ref.buchungsstelle.uid]
1134 bu.isHistoric = all(ref.isHistoric for ref in bu.recs)
1136 return fs
1138 def historic_buchungsstelle_list(self, r: dt.FlurstueckRecord, hist_buchung):
1139 """Create historic Buchungsstellen for a historic Flurstueck record.
1141 Args:
1142 r: Historic Flurstueck record.
1143 hist_buchung: ``buchung`` structs of the historic Flurstueck.
1145 Returns:
1146 A list of Buchungsstellen for the Buchungsblaetter that are found.
1147 """
1149 # an AX_HistorischesFlurstueck with a special 'buchung' reference
1151 bs_list = []
1153 for bu in hist_buchung:
1154 bb = self.rr.buchung.buchungsblattkennzeichenMap.get(bu.buchungsblattkennzeichen)
1155 if not bb:
1156 continue
1157 # create a fake historic Buchungstelle
1158 bs_list.append(dt.Buchungsstelle(
1159 uid=bb.uid + '_' + bu.laufendeNummerDerBuchungsstelle,
1160 recs=[
1161 dt.BuchungsstelleRecord(
1162 endet=r.endet,
1163 laufendeNummer=bu.laufendeNummerDerBuchungsstelle,
1164 isHistoric=True,
1165 )
1166 ],
1167 buchungsblattUids=[bb.uid],
1168 buchungsblattkennzeichenList=[bb.buchungsblattkennzeichen],
1169 parentUids=[],
1170 childUids=[],
1171 fsUids=[],
1172 parentkennzeichenList=[],
1173 flurstueckskennzeichenList=[],
1174 laufendeNummer=bu.laufendeNummerDerBuchungsstelle,
1175 isHistoric=True,
1176 ))
1178 return bs_list
1180 def buchungsstelle_list(self, r: dt.FlurstueckRecord):
1181 """Return the Buchungsstelle of a Flurstueck record with its parents and children.
1183 Args:
1184 r: Flurstueck record.
1186 Returns:
1187 A list of Buchungsstellen: the parents, the Buchungsstelle itself and the children.
1188 """
1190 # AX_Flurstueck.istGebucht -> AX_Buchungsstelle
1192 this_bs = self.rr.buchung.om.Buchungsstelle.get(_pop(r, '_istGebucht'))
1193 if not this_bs:
1194 return []
1196 bs_list = []
1198 # A Flurstück points to a Buchungsstelle (F.istGebucht -> B).
1199 # A Buchungsstelle can have parent (B.an -> parent.uid) and child (child.an -> B.uid) Buchungsstellen
1200 # (these references are populated in _BuchungIndexer above).
1201 # Our task here, given F.istGebucht -> B, collect B's parents and children
1202 # These are Buchungsstellen that directly or indirectly mention the current Flurstück.
1204 queue: list[dt.Buchungsstelle] = [this_bs]
1205 while queue:
1206 bs = queue.pop(0)
1207 bs_list.insert(0, bs)
1208 for uid in bs.parentUids:
1209 queue.append(self.rr.buchung.om.Buchungsstelle.get(uid))
1211 # remove this_bs
1212 bs_list.pop()
1214 queue: list[dt.Buchungsstelle] = [this_bs]
1215 while queue:
1216 bs = queue.pop(0)
1217 bs_list.append(bs)
1218 # sort related (child) Buchungsstellen by their BB-Kennzeichen
1219 child_bs_list = self.rr.buchung.om.Buchungsstelle.get_many(bs.childUids)
1220 child_bs_list.sort(key=_sortkey_buchungsstelle_by_bblatt)
1221 queue.extend(child_bs_list)
1223 # if len(bs_list) > 1:
1224 # gws.log.debug(f'bs chain: {r.uid=} {this_bs.uid=} {[bs.uid for bs in bs_list]}')
1226 return bs_list
1228 def write(self):
1229 values = []
1231 for fs in self.om.Flurstueck:
1232 geoms = [_pop(r, 'geom') for r in fs.recs]
1233 data = index.serialize(fs)
1234 for r, g in zip(fs.recs, geoms):
1235 r.geom = g
1237 values.append(dict(
1238 uid=fs.uid,
1239 rc=len(fs.recs),
1240 fshistoric=fs.isHistoric,
1241 data=data,
1242 geom=geoms[-1],
1243 ))
1245 self.write_table(index.TABLE_FLURSTUECK, values)
1248class _FsIndexIndexer(_Indexer):
1249 """Indexer for the flat search tables (``index*``)."""
1251 entries = {
1252 index.TABLE_INDEXFLURSTUECK: [],
1253 index.TABLE_INDEXLAGE: [],
1254 index.TABLE_INDEXBUCHUNGSBLATT: [],
1255 index.TABLE_INDEXPERSON: [],
1256 index.TABLE_INDEXGEOM: [],
1257 }
1258 """Rows by table id."""
1260 def collect(self):
1261 with ProgressIndicator(f'ALKIS: creating indexes', len(self.rr.fsdata.om.Flurstueck)) as progress:
1262 for fs in self.rr.fsdata.om.Flurstueck:
1263 for r in fs.recs:
1264 self.add(fs, r)
1265 progress.update(1)
1267 def add(self, fs: dt.Flurstueck, r: dt.FlurstueckRecord):
1268 """Add search table rows for a Flurstueck record.
1270 Args:
1271 fs: Flurstueck.
1272 r: Flurstueck record.
1273 """
1275 base = dict(
1276 fs=r.uid,
1277 fshistoric=r.isHistoric,
1278 )
1280 places = dict(
1281 land=r.land.text,
1282 land_t=index.text_key(r.land.text),
1283 landcode=r.land.code,
1285 regierungsbezirk=r.regierungsbezirk.text,
1286 regierungsbezirk_t=index.text_key(r.regierungsbezirk.text),
1287 regierungsbezirkcode=r.regierungsbezirk.code,
1289 kreis=r.kreis.text,
1290 kreis_t=index.text_key(r.kreis.text),
1291 kreiscode=r.kreis.code,
1293 gemeinde=r.gemeinde.text,
1294 gemeinde_t=index.text_key(r.gemeinde.text),
1295 gemeindecode=r.gemeinde.code,
1297 gemarkung=r.gemarkung.text,
1298 gemarkung_t=index.text_key(r.gemarkung.text),
1299 gemarkungcode=r.gemarkung.code,
1301 )
1303 self.entries[index.TABLE_INDEXFLURSTUECK].append(dict(
1304 **base,
1305 **places,
1307 amtlicheflaeche=r.amtlicheFlaeche,
1308 geomflaeche=r.geomFlaeche,
1310 flurnummer=r.flurnummer,
1311 zaehler=r.zaehler,
1312 nenner=r.nenner,
1313 flurstuecksfolge=r.flurstuecksfolge,
1314 flurstueckskennzeichen=r.flurstueckskennzeichen,
1316 x=r.x,
1317 y=r.y,
1318 ))
1320 self.entries[index.TABLE_INDEXGEOM].append(dict(
1321 **base,
1322 geomflaeche=r.geomFlaeche,
1323 x=r.x,
1324 y=r.y,
1325 geom=r.geom,
1326 ))
1328 for la in fs.lageList:
1329 for la_r in la.recs:
1330 self.entries[index.TABLE_INDEXLAGE].append(dict(
1331 **base,
1332 **places,
1333 lageuid=la_r.uid,
1334 lagehistoric=la_r.isHistoric,
1335 strasse=la_r.strasse,
1336 strasse_t=index.strasse_key(la_r.strasse),
1337 hausnummer=la_r.hausnummer,
1338 x=la.x or r.x,
1339 y=la.y or r.y,
1340 ))
1342 for bu in fs.buchungList:
1343 bb = self.rr.buchung.om.Buchungsblatt.get(bu.buchungsblattUid)
1345 for bb_r in bb.recs:
1346 self.entries[index.TABLE_INDEXBUCHUNGSBLATT].append(dict(
1347 **base,
1348 buchungsblattuid=bb_r.uid,
1349 buchungsblattkennzeichen=bb_r.buchungsblattkennzeichen,
1350 buchungsblatthistoric=bu.isHistoric,
1351 ))
1353 pe_uids = set()
1355 for nn in bb.namensnummerList:
1356 for pe in nn.personList:
1357 if pe.uid in pe_uids:
1358 continue
1359 pe_uids.add(pe.uid)
1360 for pe_r in pe.recs:
1361 self.entries[index.TABLE_INDEXPERSON].append(dict(
1362 **base,
1363 personuid=pe_r.uid,
1364 personhistoric=pe_r.isHistoric,
1365 name=pe_r.nachnameOderFirma,
1366 name_t=index.text_key(pe_r.nachnameOderFirma),
1367 vorname=pe_r.vorname,
1368 vorname_t=index.text_key(pe_r.vorname),
1369 ))
1371 def write(self):
1372 for table_id, values in self.entries.items():
1373 if not self.ix.has_table(table_id):
1374 for n, v in enumerate(values, 1):
1375 v['n'] = n
1376 self.write_table(table_id, values)
1379class _Runner:
1380 """Runs all indexers in order and writes the index tables."""
1382 def __init__(self, ix: index.Object, reader: dt.Reader, with_cache=False):
1383 """Create a runner.
1385 Args:
1386 ix: Index to build.
1387 reader: Source data reader.
1388 with_cache: Whether to cache source data and collected objects.
1389 """
1391 self.ix: index.Object = ix
1392 self.reader: dt.Reader = reader
1394 self.withCache = with_cache
1395 self.cacheDir = gws.c.CACHE_DIR + '/alkis'
1396 if self.withCache:
1397 gws.u.ensure_dir(self.cacheDir)
1399 self.place = _PlaceIndexer(self)
1400 self.lage = _LageIndexer(self)
1401 self.buchung = _BuchungIndexer(self)
1402 self.part = _PartIndexer(self)
1403 self.fsdata = _FsDataIndexer(self)
1404 self.fsindex = _FsIndexIndexer(self)
1406 self.initMemory = gws.lib.osx.process_rss_size()
1408 def run(self):
1409 """Collect all data and write the index tables."""
1411 with self.ix.db.connect() as conn:
1412 with ProgressIndicator(f'ALKIS: indexing'):
1413 self.place.load_or_collect()
1414 self.memory_info()
1416 self.buchung.load_or_collect()
1417 self.memory_info()
1419 self.lage.load_or_collect()
1420 self.memory_info()
1422 self.fsdata.load_or_collect()
1423 gws.log.info(f'ALKIS: fs counts: {self.fsdata.counts}')
1424 self.memory_info()
1426 self.part.load_or_collect()
1427 self.memory_info()
1429 self.fsindex.collect()
1430 self.memory_info()
1432 self.place.write()
1433 self.buchung.write()
1434 self.lage.write()
1435 self.fsdata.write()
1436 self.part.write()
1437 self.fsindex.write()
1439 def memory_info(self):
1440 """Log the memory used since the runner was created."""
1442 v = gws.lib.osx.process_rss_size() - self.initMemory
1443 if v > 0:
1444 gws.log.info(f'ALKIS: memory used: {v:.2f} MB', stacklevel=2)
1446 def read_flat(self, cls):
1447 """Read all source objects of a type.
1449 Args:
1450 cls: GeoInfoDok class from ``gid6``.
1452 Returns:
1453 A list of objects.
1454 """
1456 cpath = self.cacheDir + '/flat_' + cls.__name__
1457 if self.withCache and gws.u.is_file(cpath):
1458 return gws.u.unserialize_from_path(cpath)
1460 rs = self._read_flat(cls)
1461 if self.withCache:
1462 gws.u.serialize_to_path(rs, cpath)
1464 return rs
1466 def _read_flat(self, cls):
1467 """Read all source objects of a type, without the cache."""
1469 cnt = self.reader.count(cls)
1470 if cnt <= 0:
1471 gws.log.warning(f'ALKIS: read {cls.__name__}: empty table')
1472 return []
1474 rs = []
1475 with ProgressIndicator(f'ALKIS: read {cls.__name__}', cnt) as progress:
1476 for ax in self.reader.read_all(cls):
1477 rs.append(ax)
1478 progress.update(1)
1479 return rs
1481 def read_grouped(self, cls):
1482 """Read all source objects of a type, grouped by identifier.
1484 Args:
1485 cls: GeoInfoDok class from ``gid6``.
1487 Returns:
1488 A list of ``(identifier, objects)`` tuples, objects sorted by start date.
1489 """
1491 cpath = self.cacheDir + '/grouped_' + cls.__name__
1492 if self.withCache and gws.u.is_file(cpath):
1493 return gws.u.unserialize_from_path(cpath)
1495 rs = self._read_grouped(cls)
1496 if self.withCache:
1497 gws.u.serialize_to_path(rs, cpath)
1499 return rs
1501 def _read_grouped(self, cls):
1502 """Read and group source objects of a type, without the cache."""
1504 cnt = self.reader.count(cls)
1505 if cnt <= 0:
1506 gws.log.warning(f'ALKIS: read {cls.__name__}: empty table')
1507 return []
1509 groups = {}
1510 with ProgressIndicator(f'ALKIS: read {cls.__name__}', cnt) as progress:
1511 for ax in self.reader.read_all(cls):
1512 groups.setdefault(ax.identifikator, []).append(ax)
1513 progress.update(1)
1514 for g in groups.values():
1515 g.sort(key=_sortkey_lebenszeitintervall)
1517 return list(groups.items())
1519 def props_from(self, ax, atts):
1520 """Extract descriptive properties from a source object.
1522 Object values and empty values are skipped, dates are formatted as ``DD.MM.YYYY``.
1524 Args:
1525 ax: Source object.
1526 atts: Attribute metadata.
1528 Returns:
1529 An object with the properties.
1530 """
1532 d = {}
1534 for a in atts:
1535 v = getattr(ax, a['name'], None)
1536 if isinstance(v, gid.Object):
1537 # @TODO handle properties which are objects
1538 continue
1539 if dtx.is_date(v):
1540 v = dtx.to_string('%d.%m.%Y', v)
1541 if not gws.u.is_empty(v):
1542 d[a['name']] = v
1544 return dt.Object(**d)
1547def _from_ax(cls, ax, **kwargs):
1548 """Create a record from a source object, its life span and extra values."""
1550 d = {}
1552 if ax:
1553 for k in cls.__annotations__:
1554 v = getattr(ax, k, None)
1555 if v:
1556 d[k] = v
1558 d['uid'] = ax.identifikator
1559 d['beginnt'] = ax.lebenszeitintervall.beginnt
1560 d['endet'] = ax.lebenszeitintervall.endet
1561 d['isHistoric'] = d['endet'] is not None
1562 if ax.anlass and ax.anlass[0].code != '000000':
1563 d['anlass'] = ax.anlass[0]
1564 if ax.geom:
1565 d['geom'] = ax.geom
1567 d.update(kwargs)
1568 return cls(**d)
1571def _anteil(ax):
1572 """Format the share of a source object as a fraction string."""
1574 try:
1575 z = float(ax.anteil.zaehler)
1576 z = str(int(z) if z.is_integer() else z)
1577 n = float(ax.anteil.nenner)
1578 n = str(int(n) if n.is_integer() else n)
1579 return z + '/' + n
1580 except (AttributeError, ValueError, TypeError):
1581 pass
1584def _meta_attributes(meta):
1585 """Return the attributes of a class that are known properties, sorted by title."""
1587 return sorted(
1588 [a for a in meta['attributes'] if a['name'] in dt.PROPS],
1589 key=lambda a: a['title']
1590 )
1593def _geom_of(o):
1594 """Return the shapely geometry of an object, or ``None`` with a warning."""
1596 if not o.geom:
1597 gws.log.warning(f'{o.__class__.__name__}:{o.uid}: no geometry')
1598 return
1599 return shapely.wkb.loads(o.geom, hex=True)
1602def _pop(obj, attr):
1603 """Remove an attribute from an object and return its value."""
1605 v = getattr(obj, attr, None)
1606 try:
1607 delattr(obj, attr)
1608 except AttributeError:
1609 pass
1610 return v
1613def _sortkey_beginnt(o):
1614 return o.beginnt
1617def _sortkey_lebenszeitintervall(o):
1618 return o.lebenszeitintervall.beginnt
1621def _sortkey_namensnummer(nn: dt.Namensnummer):
1622 return _natkey(nn.recs[-1].laufendeNummerNachDIN1421), nn.recs[-1].beginnt
1625def _sortkey_buchungsstelle(bs: dt.Buchungsstelle):
1626 return _natkey(bs.recs[-1].laufendeNummer), bs.recs[-1].beginnt
1629def _sortkey_buchungsstelle_by_bblatt(bs: dt.Buchungsstelle):
1630 return bs.buchungsblattkennzeichenList[0], bs.recs[-1].beginnt
1633def _sortkey_part(pa: dt.Part):
1634 return pa.name.text, -pa.geomFlaeche
1637def _sortkey_gebaeude(ge: dt.Gebaeude):
1638 # sort Gebaeude by area (big->small)
1639 return ge.recs[-1].beginnt, -ge.recs[-1].geomFlaeche
1642def _natkey(v):
1643 """Return a key for natural sorting of strings with numbers."""
1645 if not v:
1646 return []
1647 return [
1648 '{:080d}'.format(int(digits)) if digits else chars.lower()
1649 for digits, chars in re.findall(r'(\d+)|(\D+)', v.strip())
1650 ]
1653def _comma(a):
1654 """Join values with commas, ``None`` becomes an empty string."""
1656 return ','.join(str(s) if s is not None else '' for s in a)
1659def _str(x):
1660 """Convert to a string, keeping ``None``."""
1662 return None if x is None else str(x)