backup

exclude from all for fast_float
backup
2026-06-04 22:14:24 +08:00 · 2025-12-03 16:22:44 +01:00 · 2025-11-27 09:05:51 +01:00 · 2025-11-26 16:35:30 +01:00 · 2025-11-26 13:46:55 +01:00 · 2025-11-26 13:11:54 +01:00
58 changed files with 6921 additions and 3272 deletions
--- a/.clang-format
+++ b/.clang-format
@@ -0,0 +1,22 @@
+BasedOnStyle: LLVM
+UseTab: AlignWithSpaces
+IndentWidth: 4
+TabWidth: 4
+BreakBeforeBraces: Allman
+ColumnLimit: 0
+NamespaceIndentation: Inner
+FixNamespaceComments: true
+AccessModifierOffset: -2
+AllowShortCaseLabelsOnASingleLine: true
+IndentCaseLabels: true
+BreakConstructorInitializers: BeforeComma
+BraceWrapping:
+  BeforeLambdaBody: false
+AlignAfterOpenBracket: DontAlign
+Cpp11BracedListStyle: false
+IncludeBlocks: Regroup
+LambdaBodyIndentation: Signature
+AllowShortLambdasOnASingleLine: Inline
+EmptyLineBeforeAccessModifier: LogicalBlock
+IndentPPDirectives: AfterHash
+PPIndentWidth: 1
--- a/.github/workflows/build-documentation.yml
+++ b/.github/workflows/build-documentation.yml
@@ -19,7 +19,7 @@ jobs:
  docs:
    runs-on: ubuntu-latest
    steps:
-    - uses: actions/checkout@v1
+    - uses: actions/checkout@v4

    - name: Set reusable strings
      # Turn repeated input strings (such as the build output directory) into step outputs. These step outputs can be used throughout the workflow file.
@@ -47,7 +47,7 @@ jobs:
        ls -l ${{ steps.strings.outputs.build-output-dir }}/docs/sphinx

    - name: Upload artifact
-      uses: actions/upload-pages-artifact@v2
+      uses: actions/upload-pages-artifact@v3
      with:
        path: ${{ steps.strings.outputs.build-output-dir }}/docs/sphinx

@@ -62,4 +62,4 @@ jobs:
    steps:
      - name: Deploy to GitHub Pages
        id: deployment
-        uses: actions/deploy-pages@v2
+        uses: actions/deploy-pages@v4
--- a/.github/workflows/cmake-multi-platform.yml
+++ b/.github/workflows/cmake-multi-platform.yml
@@ -33,13 +33,18 @@ jobs:

    - name: Install dependencies Ubuntu
      if: matrix.os == 'ubuntu-latest'
-      run: sudo apt-get update && sudo apt-get install mrc
+      run: sudo apt-get update && sudo apt-get install mrc catch2

    - name: Install dependencies Window
      if: matrix.os == 'windows-latest'
      run: ./tools/depends.cmd
      shell: cmd

+    - name: Install Catch2 macOS
+      if: matrix.os == 'macos-latest'
+      run: >
+        brew install catch2
+
    - name: Configure CMake
      run: >
        cmake -B ${{ steps.strings.outputs.build-output-dir }}
--- a/.gitignore
+++ b/.gitignore
@@ -13,3 +13,6 @@ docs/api
 docs/conf.py
 build_ci/
 data/components.cif
+perf.data*
+.cache/
+
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -24,29 +24,26 @@

 cmake_minimum_required(VERSION 3.23)

+if(CMAKE_SOURCE_DIR STREQUAL CMAKE_CURRENT_SOURCE_DIR AND NOT CMAKE_BUILD_TYPE AND NOT CMAKE_CONFIGURATION_TYPES)
+	set(CMAKE_BUILD_TYPE Release CACHE STRING "Build type" FORCE)
+	set_property(CACHE CMAKE_BUILD_TYPE PROPERTY STRINGS "Debug" "Release" "MinSizeRel" "RelWithDebInfo")
+endif()
+
 # set the project name
 project(
 	libcifpp
-	VERSION 8.0.0
-	LANGUAGES CXX)
+	VERSION 9.0.5
+	LANGUAGES CXX C)

 list(PREPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake")

 include(FindAtomic)
-include(CheckFunctionExists)
-include(CheckIncludeFiles)
-include(CheckLibraryExists)
 include(CMakePackageConfigHelpers)
-include(CheckCXXSourceCompiles)
 include(GenerateExportHeader)
 include(CTest)
-include(FetchContent)
 include(ExternalProject)
-
-# FindBoost, take care of it now.
-if(CMAKE_VERSION VERSION_GREATER_EQUAL 3.30)
-	cmake_policy(SET CMP0167 NEW)
-endif()
+include(FetchContent)
+include(VersionString)

 # When building with ninja-multiconfig, build both debug and release by default
 if(CMAKE_GENERATOR STREQUAL "Ninja Multi-Config")
@@ -63,25 +60,23 @@ elseif(MSVC)
 endif()

 # Build documentation?
-option(BUILD_DOCUMENTATION "Build the documentation" OFF)
+set(BUILD_DOCUMENTATION OFF CACHE BOOL "Build the documentation")

 # Optionally build a version to be installed inside CCP4
-option(BUILD_FOR_CCP4 "Build a version to be installed in CCP4")
+set(BUILD_FOR_CCP4 OFF CACHE BOOL "Build a version to be installed in CCP4")

 # Building shared libraries?
 if(NOT(BUILD_FOR_CCP4 AND WIN32))
-	option(BUILD_SHARED_LIBS "Build a shared library instead of a static one" OFF)
+	set(BUILD_SHARED_LIBS OFF CACHE BOOL "Build a shared library instead of a static one")
 endif()

 if(PROJECT_IS_TOP_LEVEL AND NOT BUILD_FOR_CCP4)
 	# Lots of code depend on the availability of the components.cif file
-	option(CIFPP_DOWNLOAD_CCD
-		"Download the CCD file components.cif during installation" ON)
+	set(CIFPP_DOWNLOAD_CCD ON CACHE BOOL "Download the CCD file components.cif during installation")

 	# An optional cron script can be installed to keep the data files up-to-date
 	if(UNIX AND NOT APPLE)
-		option(CIFPP_INSTALL_UPDATE_SCRIPT
-			"Install the script to update CCD and dictionary files" ON)
+		set(CIFPP_INSTALL_UPDATE_SCRIPT ON CACHE BOOL "Install the script to update CCD and dictionary files")
 	endif()
 else()
 	unset(CIFPP_DOWNLOAD_CCD)
@@ -91,14 +86,13 @@ endif()
 # When CCP4 is sourced in the environment, we can recreate the symmetry
 # operations table
 if(EXISTS "$ENV{CCP4}/lib/data/syminfo.lib")
-	option(CIFPP_RECREATE_SYMOP_DATA
-		"Recreate SymOp data table in case it is out of date" ON)
+	set(CIFPP_RECREATE_SYMOP_DATA ON CACHE BOOL "Recreate SymOp data table in case it is out of date")
 endif()

 # CCP4 build
 if(BUILD_FOR_CCP4)
 	if("$ENV{CCP4}" STREQUAL "" OR NOT EXISTS $ENV{CCP4})
-		message(FATAL_ERROR "A CCP4 built was requested but CCP4 was not sourced")
+		message(FATAL_ERROR "cifpp: A CCP4 built was requested but CCP4 was not sourced")
 	else()
 		list(PREPEND CMAKE_MODULE_PATH "$ENV{CCP4}")
 		list(PREPEND CMAKE_PREFIX_PATH "$ENV{CCP4}")
@@ -128,9 +122,6 @@ if(WIN32)
 		add_definitions(-D _WIN32_WINNT=0x0501)
 	endif()

-	# Man, this is 2024 we're living in...
-	add_definitions(-DNOMINMAX)
-
 	# We do not want to write an export file for all our symbols...
 	set(CMAKE_WINDOWS_EXPORT_ALL_SYMBOLS ON)
 endif()
@@ -150,51 +141,8 @@ endif()

 # Libraries

-# Start by finding out if std:regex is usable. Note that the current
-# implementation in GCC is not acceptable, it crashes on long lines. The
-# implementation in libc++ (clang) and MSVC seem to be OK.
-check_cxx_source_compiles(
-	"
-#include <iostream>
-#ifndef __GLIBCXX__
-#error
-#endif
-int main(int argc, char *argv[]) { return 0; }"
-	GXX_LIBSTDCPP)
-
-if(GXX_LIBSTDCPP)
-	message(
-		STATUS "Testing for known regex bug, since you're using GNU libstdc++")
-
-	try_run(STD_REGEX_RUNNING STD_REGEX_COMPILING
-		${CMAKE_CURRENT_BINARY_DIR}/test
-		${CMAKE_CURRENT_SOURCE_DIR}/cmake/test-rx.cpp)
-
-	if(STD_REGEX_RUNNING STREQUAL FAILED_TO_RUN)
-		message(
-			STATUS
-			"You are probably trying to compile using the g++ standard library which contains a crashing std::regex implementation. Will use boost::regex instead"
-		)
-
-		find_package(Boost 1.80 QUIET COMPONENTS regex)
-
-		if(NOT Boost_FOUND)
-			set(BOOST_REGEX_STANDALONE ON)
-
-			FetchContent_Declare(
-				boost-rx
-				GIT_REPOSITORY https://github.com/boostorg/regex
-				GIT_TAG boost-1.83.0)
-
-			FetchContent_MakeAvailable(boost-rx)
-		endif()
-
-		set(BOOST_REGEX ON)
-	endif()
-endif()
-
 if(MSVC)
-	# Avoid linking the shared library of zlib Search ZLIB_ROOT first if it is
+	# Avoid linking the shared library of zlib. Search ZLIB_ROOT first if it is
 	# set.
 	if(ZLIB_ROOT)
 		set(_ZLIB_SEARCH_ROOT PATHS ${ZLIB_ROOT} NO_DEFAULT_PATH)
@@ -221,11 +169,34 @@ if(MSVC)
 	endforeach()
 endif()

-find_package(ZLIB QUIET)
+# Using fast_float for float parsing, but only if needed
+try_compile(STD_CHARCONV_COMPILING
+	SOURCES ${CMAKE_CURRENT_SOURCE_DIR}/cmake/test-charconv.cpp)
+
+if(NOT STD_CHARCONV_COMPILING)
+	message(NOTICE "libcifpp: Using fast_float for std::from_chars")
+	FetchContent_Declare(fastfloat
+		GIT_REPOSITORY "https://github.com/fastfloat/fast_float"
+		GIT_TAG v8.0.2
+		EXCLUDE_FROM_ALL)
+	FetchContent_MakeAvailable(fastfloat)
+endif()
+
 find_package(Threads)
+find_package(ZLIB QUIET)

 if(NOT ZLIB_FOUND)
-	message(FATAL_ERROR "The zlib development files were not found you this system, please install them and try again (hint: on debian/ubuntu use apt-get install zlib1g-dev)")
+	message(FATAL_ERROR "cifpp: The zlib development files were not found you this system, please install them and try again (hint: on debian/ubuntu use apt-get install zlib1g-dev)")
+endif()
+
+include(FindPkgConfig)
+
+if(PKG_CONFIG_FOUND)
+	pkg_check_modules(PCRE2 IMPORTED_TARGET libpcre2-8)
+endif()
+
+if(NOT PCRE2_FOUND)
+	add_subdirectory(pcre2-simple)
 endif()

 # Using Eigen3 is a bit of a thing. We don't want to build it completely since
@@ -239,18 +210,16 @@ if(Eigen3_FOUND AND TARGET Eigen3::Eigen)
 else()
 	# Use ExternalProject since FetchContent always tries to install the result...
 	ExternalProject_Add(my-eigen3
-		GIT_REPOSITORY https://gitlab.com/libeigen/eigen.git
-		GIT_TAG 3.4.0
+		URL https://gitlab.com/libeigen/eigen/-/archive/3.4.0/eigen-3.4.0.zip
+		DOWNLOAD_EXTRACT_TIMESTAMP TRUE
+		CONFIGURE_COMMAND ""
+		BUILD_COMMAND ""
 		INSTALL_COMMAND "")
-	
+
 	ExternalProject_Get_Property(my-eigen3 SOURCE_DIR)
 	set(EIGEN_INCLUDE_DIR ${SOURCE_DIR})
 endif()

-# Create a revision file, containing the current git version info
-include(VersionString)
-write_version_header(${CMAKE_CURRENT_SOURCE_DIR}/src/ LIB_NAME "LibCIFPP")
-
 # SymOp data table
 if(CIFPP_RECREATE_SYMOP_DATA)
 	# The tool to create the table
@@ -272,6 +241,9 @@ if(CIFPP_RECREATE_SYMOP_DATA)
 		"$ENV{CLIBD}/symop.lib")
 endif()

+# Create a revision file, containing the current git version info
+write_version_header("${CMAKE_CURRENT_SOURCE_DIR}/src/" LIB_NAME "LibCIFPP")
+
 # Sources
 set(project_sources
 	${CMAKE_CURRENT_SOURCE_DIR}/src/category.cpp
@@ -290,6 +262,7 @@ set(project_sources
 	${CMAKE_CURRENT_SOURCE_DIR}/src/point.cpp
 	${CMAKE_CURRENT_SOURCE_DIR}/src/symmetry.cpp
 	${CMAKE_CURRENT_SOURCE_DIR}/src/model.cpp
+	${CMAKE_CURRENT_SOURCE_DIR}/src/cql/transaction.cpp
 	${CMAKE_CURRENT_SOURCE_DIR}/src/pdb/cif2pdb.cpp
 	${CMAKE_CURRENT_SOURCE_DIR}/src/pdb/pdb2cif.cpp
 	${CMAKE_CURRENT_SOURCE_DIR}/src/pdb/pdb_record.hpp
@@ -326,6 +299,7 @@ set(project_headers
 	include/cif++/row.hpp
 	include/cif++/symmetry.hpp
 	include/cif++/text.hpp
+	include/cif++/cql/transaction.hpp
 	include/cif++/utilities.hpp
 	include/cif++/validate.hpp
 )
@@ -349,19 +323,9 @@ target_sources(cifpp
 # The code now really requires C++20
 target_compile_features(cifpp PUBLIC cxx_std_20)

-set(CMAKE_DEBUG_POSTFIX d)
-set_target_properties(cifpp PROPERTIES DEBUG_POSTFIX "d")
-
 generate_export_header(cifpp EXPORT_FILE_NAME
 	${CMAKE_CURRENT_SOURCE_DIR}/include/cif++/exports.hpp)

-if(BOOST_REGEX)
-	target_compile_definitions(cifpp PRIVATE USE_BOOST_REGEX=1
-		BOOST_REGEX_STANDALONE=1)
-	get_target_property(BOOST_REGEX_INCLUDE_DIR Boost::regex
-		INTERFACE_INCLUDE_DIRECTORIES)
-endif()
-
 if(MSVC)
 	target_compile_definitions(cifpp PUBLIC NOMINMAX=1)
 endif()
@@ -372,9 +336,21 @@ target_include_directories(
 	cifpp
 	PUBLIC "$<BUILD_INTERFACE:${CMAKE_CURRENT_SOURCE_DIR}/include>"
 	"$<INSTALL_INTERFACE:${CMAKE_INSTALL_INCLUDEDIR}>"
-	PRIVATE "${BOOST_REGEX_INCLUDE_DIR}" "${EIGEN_INCLUDE_DIR}")
+	PRIVATE "${EIGEN_INCLUDE_DIR}")

-target_link_libraries(cifpp PUBLIC Threads::Threads ZLIB::ZLIB $<$<TARGET_EXISTS:std::atomic>:std::atomic>)
+target_link_libraries(cifpp
+	PUBLIC Threads::Threads ZLIB::ZLIB $<$<TARGET_EXISTS:std::atomic>:std::atomic>)
+
+if(PCRE2_FOUND)
+	target_include_directories(cifpp PRIVATE ${PCRE2_INCLUDE_DIRS})
+	target_link_libraries(cifpp PRIVATE ${PCRE2_LINK_LIBRARIES})
+else()
+	target_link_libraries(cifpp PRIVATE $<BUILD_INTERFACE:pcre2s>)
+endif()
+
+if(NOT STD_CHARCONV_COMPILING)
+	target_link_libraries(cifpp PUBLIC FastFloat::fast_float)
+endif()

 if(CMAKE_CXX_COMPILER_ID STREQUAL "AppleClang")
 	target_link_options(cifpp PRIVATE -undefined dynamic_lookup)
@@ -388,7 +364,7 @@ if(CIFPP_DOWNLOAD_CCD)
 		file(SIZE ${COMPONENTS_CIF} CCD_FILE_SIZE)

 		if(CCD_FILE_SIZE EQUAL 0)
-			message(STATUS "Removing empty ${COMPONENTS_CIF} file")
+			message(STATUS "cifpp: Removing empty ${COMPONENTS_CIF} file")
 			file(REMOVE "${COMPONENTS_CIF}")
 		endif()
 	endif()
@@ -427,7 +403,7 @@ if(CIFPP_DOWNLOAD_CCD)

 		if(CCD_FETCH_STATUS_CODE)
 			message(
-				FATAL_ERROR "Error trying to download CCD file: ${CCD_FETCH_STATUS}")
+				FATAL_ERROR "cifpp: Error trying to download CCD file: ${CCD_FETCH_STATUS}")
 		endif()
 	endif()
 endif()
@@ -491,7 +467,7 @@ file(GLOB OLD_CONFIG_FILES

 if(OLD_CONFIG_FILES)
 	message(
-		STATUS "Installation will remove old config files: ${OLD_CONFIG_FILES}")
+		STATUS "cifpp: Installation will remove old config files: ${OLD_CONFIG_FILES}")
 	install(CODE "file(REMOVE ${OLD_CONFIG_FILES})")
 endif()

@@ -557,7 +533,7 @@ if(CIFPP_INSTALL_UPDATE_SCRIPT)
 			PERMISSIONS OWNER_EXECUTE OWNER_READ GROUP_EXECUTE GROUP_READ WORLD_EXECUTE
 			WORLD_READ)
 	else()
-		message(FATAL_ERROR "Don't know where to install the update script")
+		message(FATAL_ERROR "cifpp: Don't know where to install the update script")
 	endif()

 	# a config file, to make it complete
@@ -571,7 +547,7 @@ if(CIFPP_INSTALL_UPDATE_SCRIPT)
 		install(FILES ${CMAKE_CURRENT_BINARY_DIR}/libcifpp.conf
 			DESTINATION ${CMAKE_INSTALL_SYSCONFDIR})
 		install(
-			CODE "message(\"A configuration file has been written to ${CIFPP_ETC_DIR}/libcifpp.conf, please edit this file to enable automatic updates\")"
+			CODE "message(\"cifpp: A configuration file has been written to ${CIFPP_ETC_DIR}/libcifpp.conf, please edit this file to enable automatic updates\")"
 		)

 		install(DIRECTORY DESTINATION ${CMAKE_INSTALL_SYSCONFDIR}/libcifpp/cache-update.d)
--- a/README.md
+++ b/README.md
@@ -117,12 +117,8 @@ Other libraries you might want to install beforehand are:
  `libeigen3-dev`
 - [zlib](https://github.com/madler/zlib), the development version of this
  library. On Debian/Ubuntu this is the package `zlib1g-dev`.
- [boost](https://www.boost.org), in Debian/Ubuntu this is `libboost-dev`.
-  
-  The Boost libraries are only needed in case you are using GCC due to a long
-  standing bug in GNU's implementation of std::regex. It simply crashes
-  on the regular expressions used in the mmcif_pdbx dictionary and so
-  we use the boost regex implementation instead.
+- [pcre2](https://www.pcre.org/), the Perl Compatible Regular Expression
+  library. On Debian/Ubuntu this is the package `libpcre2-dev`.

 ### Building

--- a/36
+++ b/36
@@ -1,3 +1,39 @@
+Version 9.0.5
+- Added exists to compound_factory
+- Added sub_matrix, fix and extend determinant calculation
+- Added yet another structure::create_non_poly
+- Remove revision.hpp file in make clean (new VersionString.cmake)
+
+Version 9.0.4
+- Fix various stopping and reconstruction errors
+
+Version 9.0.3
+- Reconstruction fixed when some entity ids are missing
+
+Version 9.0.2
+- Fix code that reconstructs sequences, could throw a map::at
+- Many optimisations in validation and reconstruction code.
+
+Version 9.0.1
+- Use pcre2 from pkg-config if available, if not
+  build a version from the original code.
+
+Version 9.0.0
+- Rename fields of cif::mm::polymer to match the naming
+  in mmcif_pdbx.dic. Also, related, fix building mm::structure
+  using the correct mapping between atom_site and residues.
+- _atom_site.auth_alt_id does not exist, it should be
+  _atom_site.pdbx_auth_alt_id of course.
+- Added a more lightweight fixup for mmcif_pdbx files
+  that lack certain categories.
+
+Version 8.0.1
+- Fix cif::mm::structure::cleanup_empty_categories, removed too much
+- Add default value for B_iso_or_equiv in residue::create_new_atom
+- Reconstruct some branch records in bare pdbx files
+- Fix parsing PDB files (bug due to missing validator in dest. cat.)
+- Do not fail conversion of PDB files when compound info is missing
+
 Version 8.0.0
 - A dictionary is for a datablock and a file can have
  datablocks with differing dictionaries.
--- a/cmake/FindPCRE2.cmake
+++ b/cmake/FindPCRE2.cmake
@@ -0,0 +1,12 @@
+# The problem is, find_package(PCRE2) does not work
+# and using pkg-config results in linking to a shared library
+# causing all kinds of trouble later on
+
+find_path(PCRE2_INCLUDEDIR NAMES pcre2.h HINTS "C:/Program Files (x86)/PCRE2/include" REQUIRED)
+find_library(PCRE2_LIBRARY NAMES pcre2-8-static libpcre2-8.a HINTS "C:/Program Files (x86)/PCRE2/lib" REQUIRED)
+
+add_library(pcre2-8 IMPORTED STATIC)
+target_include_directories(pcre2-8 INTERFACE ${PCRE2_INCLUDEDIR})
+target_compile_definitions(pcre2-8 INTERFACE PCRE2_STATIC)
+set_target_properties(pcre2-8 PROPERTIES IMPORTED_LOCATION ${PCRE2_LIBRARY})
+set_target_properties(pcre2-8 PROPERTIES IMPORTED_IMPLIB ${PCRE2_LIBRARY})
--- a/cmake/VersionString.cmake
+++ b/cmake/VersionString.cmake
@@ -238,7 +238,7 @@ function(write_version_header dir)
 		if(res EQUAL 0)
 			set(REVISION_STRING "${out}")
 		else()
-			message(STATUS "Git hash not found, does this project has a 'build' tag?")
+			message(STATUS "Git hash not found, does this project have a 'build' tag?")
 		endif()
 	else()
 		message(STATUS "Git hash not found")
--- a/cmake/test-charconv.cpp
+++ b/cmake/test-charconv.cpp
@@ -0,0 +1,17 @@
+#include <charconv>
+#include <cassert>
+#include <cstring>
+
+int main()
+{
+	float v;
+	char s[] = "1.0";
+
+	auto r = std::from_chars(s, s + strlen(s), v);
+
+	assert(r.ec == std::errc{});
+	assert(r.ptr = s + strlen(s));
+	assert(v == 1.0f);
+
+	return 0;
+}
--- a/cmake/test-rx.cpp
+++ b/cmake/test-rx.cpp
@@ -1,18 +0,0 @@
-// See: https://gcc.gnu.org/bugzilla/show_bug.cgi?id=86164
-
-#include <iostream>
-#include <regex>
-
-int main()
-{
-	std::string s(100'000, '*');
-	std::smatch m;
-	std::regex r("^(.*?)$");
-
-	std::regex_search(s, m, r);
-
-	std::cout << s.substr(0, 10) << '\n';
-	std::cout << m.str(1).substr(0, 10) << '\n';
-
-	return 0;
-}
--- a/include/cif++/category.hpp
+++ b/include/cif++/category.hpp
@@ -157,7 +157,7 @@ class category
 		emplace(std::forward<row_initializer>(rows));
 	}

-	category(const category &rhs);   ///< Copy constructor
+	category(const category &rhs); ///< Copy constructor

 	category(category &&rhs) noexcept ///< Move constructor
 	{
@@ -223,6 +223,11 @@ class category
 	/// @return Returns true is all validations pass
 	bool validate_links() const;

+	/**
+	 * @brief Strip removes items from this category that are invalid according to the assigned validator
+	 */
+	void strip();
+
 	/// @brief Equality operator, returns true if @a rhs is equal to this
 	/// @param rhs The object to compare with
 	/// @return True if the data contained is equal
@@ -327,8 +332,16 @@ class category
 	// --------------------------------------------------------------------
 	// A category can have a key, as defined by the validator/dictionary

+	/// @brief The type of an element of the key_type
+	struct key_element_type
+	{
+		std::string name;         ///< Name of the item
+		std::string value;        ///< Value to be found
+		bool may_be_null = false; ///< If true, value should be same or empty
+	};
+
 	/// @brief The key type
-	using key_type = row_initializer;
+	using key_type = std::vector<key_element_type>;

 	/// @brief Return a row_handle for the row specified by \a key
 	/// @param key The value for the key, items specified in the dictionary should have a value
@@ -1244,7 +1257,7 @@ class category
 		{
 		}

-// TODO: NEED TO FIX THIS!
+		// TODO: NEED TO FIX THIS!
 		category *linked;
 		const link_validator *v;
 	};
--- a/include/cif++/compound.hpp
+++ b/include/cif++/compound.hpp
@@ -179,8 +179,8 @@ class compound
 	friend class compound_factory_impl;
 	friend class local_compound_factory_impl;

-	compound(cif::datablock &db);
-	
+	compound(datablock &db);
+
 	std::string m_id;
 	std::string m_name;
 	std::string m_type;
@@ -270,11 +270,15 @@ class compound_factory
 		return is_std_base(res_name) or is_std_peptide(res_name);
 	}

+	/// Return whether @a res_name is water
 	bool is_water(std::string_view res_name) const
 	{
 		return res_name == "HOH" or res_name == "H2O" or res_name == "WAT";
 	}

+	/// Return whether @a res_name already exists, without creating it.
+	bool exists(std::string_view res_name) const;
+
 	/// \brief Create the compound object for \a id
 	///
 	/// This will create the compound instance for \a id if it doesn't exist already.
@@ -290,6 +294,13 @@ class compound_factory

 	void report_missing_compound(std::string_view compound_id);

+	bool get_report_missing() const { return m_report_missing; }
+
+	void set_report_missing(bool report)
+	{
+		m_report_missing = report;
+	}
+
  private:
 	compound_factory();

@@ -301,6 +312,7 @@ class compound_factory
 	static bool s_use_thread_local_instance;

 	std::shared_ptr<compound_factory_impl> m_impl;
+	bool m_report_missing = true;
 };

 // --------------------------------------------------------------------
@@ -323,14 +335,14 @@ class compound_factory
 class compound_source
 {
  public:
-	compound_source(const cif::file &file)
+	compound_source(const file &file)
 	{
-		cif::compound_factory::instance().push_dictionary(file);
+		compound_factory::instance().push_dictionary(file);
 	}

 	~compound_source()
 	{
-		cif::compound_factory::instance().pop_dictionary();
+		compound_factory::instance().pop_dictionary();
 	}
 };

--- a/include/cif++/condition.hpp
+++ b/include/cif++/condition.hpp
@@ -27,6 +27,7 @@
 #pragma once

 #include "cif++/row.hpp"
+#include "cif++/format.hpp"

 #include <cassert>
 #include <concepts>
@@ -49,49 +50,49 @@
 * @code {.cpp}
 * cif::condition c = cif::key("id") == 1;
 * @endcode
- * 
+ *
 * That will find rows where the ID item contains the number 1. If
 * using cif::key is a bit too much typing, you can also write:
- * 
+ *
 * @code{.cpp}
 * using namespace cif::literals;
- * 
+ *
 * cif::condition c2 = "id"_key == 1;
 * @endcode
- * 
+ *
 * Now if you want both ID = 1 and ID = 2 in the result:
- * 
+ *
 * @code{.cpp}
 * auto c3 = "id"_key == 1 or "id"_key == 2;
 * @endcode
- * 
+ *
 * There are some special values you can use. To find rows with item that
 * do not have a value:
- * 
+ *
 * @code{.cpp}
 * auto c4 = "type"_key == cif::null;
- * @endcode 
- * 
+ * @endcode
+ *
 * Of if it should not be NULL:
- * 
+ *
 * @code{.cpp}
 * auto c5 = "type"_key != cif::null;
- * @endcode 
- * 
+ * @endcode
+ *
 * There's even a way to find all records:
- * 
+ *
 * @code{.cpp}
 * auto c6 = cif::all;
 * @endcode
- * 
+ *
 * And when you want to search for any item containing the value 'foo':
- * 
+ *
 * @code{.cpp}
 * auto c7 = cif::any == "foo";
- * @endcode 
- * 
+ * @endcode
+ *
 * All these conditions can be chained together again:
- * 
+ *
 * @code{.cpp}
 * auto c8 = std::move(c3) and std::move(c5);
 * @endcode
@@ -106,7 +107,7 @@ namespace cif

 /**
 * @brief Get the items that can be used as key in conditions for a category
- * 
+ *
 * @param cat The category whose items to return
 * @return iset The set of key item names
 */
@@ -115,7 +116,7 @@ iset get_category_fields(const category &cat);

 /**
 * @brief Get the items that can be used as key in conditions for a category
- * 
+ *
 * @param cat The category whose items to return
 * @return iset The set of key field names
 */
@@ -123,7 +124,7 @@ iset get_category_items(const category &cat);

 /**
 * @brief Get the item index for item @a col in category @a cat
- * 
+ *
 * @param cat The category
 * @param col The name of the item
 * @return uint16_t The index
@@ -132,7 +133,7 @@ uint16_t get_item_ix(const category &cat, std::string_view col);

 /**
 * @brief Return whether the item @a col in category @a cat has a primitive type of *uchar*
- * 
+ *
 * @param cat The category
 * @param col The item name
 * @return true If the primitive type is of type *uchar*
@@ -175,14 +176,13 @@ namespace detail
 class condition
 {
  public:
-
 	/** @cond */
 	using condition_impl = detail::condition_impl;
 	/** @endcond */

 	/**
 	 * @brief Construct a new, empty condition object
-	 * 
+	 *
 	 */
 	condition()
 		: m_impl(nullptr)
@@ -191,7 +191,7 @@ class condition

 	/**
 	 * @brief Construct a new condition object with implementation @a impl
-	 * 
+	 *
 	 * @param impl The implementation to use
 	 */
 	explicit condition(condition_impl *impl)
@@ -230,15 +230,15 @@ class condition
 	/**
 	 * @brief Prepare the condition to be used on category @a c. This will
 	 * take care of setting the correct indices for items e.g.
-	 * 
+	 *
 	 * @param c The category this query should act upon
 	 */
 	void prepare(const category &c);

 	/**
-	 * @brief This operator returns true if the row referenced by @a r is 
+	 * @brief This operator returns true if the row referenced by @a r is
 	 * a match for this condition.
-	 * 
+	 *
 	 * @param r The reference to a row.
 	 * @return true If there is a match
 	 * @return false If there is no match
@@ -263,7 +263,7 @@ class condition
 	/**
 	 * @brief If the prepare step found out there is only one hit
 	 * this single hit can be returned by this method.
-	 * 
+	 *
 	 * @return std::optional<row_handle> The result will contain
 	 * a row reference if there is a single hit, it will be empty otherwise
 	 */
@@ -292,7 +292,7 @@ class condition

 	/**
 	 * @brief Operator to use to write out a condition to @a os, for debugging purposes
-	 * 
+	 *
 	 * @param os The std::ostream to write to
 	 * @param cond The condition to write
 	 * @return std::ostream& The same as @a os
@@ -752,28 +752,9 @@ namespace detail
 				delete sub;
 		}

-		condition_impl *prepare(const category &c) override
-		{
-			for (auto &sub : m_sub)
-				sub = sub->prepare(c);
-			return this;
-		}
+		condition_impl *prepare(const category &c) override;

-		bool test(row_handle r) const override
-		{
-			bool result = true;
-
-			for (auto sub : m_sub)
-			{
-				if (sub->test(r))
-					continue;
-
-				result = false;
-				break;
-			}
-
-			return result;
-		}
+		bool test(row_handle r) const override;

 		void str(std::ostream &os) const override
 		{
@@ -820,6 +801,7 @@ namespace detail
 		static condition_impl *combine_equal(std::vector<and_condition_impl *> &subs, or_condition_impl *oc);

 		std::vector<condition_impl *> m_sub;
+		std::optional<row_handle> m_single; // Potential result of index lookup
 	};

 	struct or_condition_impl : public condition_impl
@@ -977,9 +959,9 @@ inline condition operator or(condition &&a, condition &&b)
 			if (ci->m_item_name == ce->m_item_name)
 				return condition(new detail::key_equals_or_empty_condition_impl(ci));
 		}
-		
+
 		if (typeid(*b.m_impl) == typeid(detail::key_equals_condition_impl) and
-				 typeid(*a.m_impl) == typeid(detail::key_is_empty_condition_impl))
+			typeid(*a.m_impl) == typeid(detail::key_is_empty_condition_impl))
 		{
 			auto ci = static_cast<detail::key_equals_condition_impl *>(b.m_impl);
 			auto ce = static_cast<detail::key_is_empty_condition_impl *>(a.m_impl);
@@ -997,9 +979,9 @@ inline condition operator or(condition &&a, condition &&b)
 			if (ci->m_item_name == ce->m_item_name)
 				return condition(new detail::key_equals_number_or_empty_condition_impl(ci));
 		}
-		
+
 		if (typeid(*b.m_impl) == typeid(detail::key_equals_number_condition_impl) and
-				 typeid(*a.m_impl) == typeid(detail::key_is_empty_condition_impl))
+			typeid(*a.m_impl) == typeid(detail::key_is_empty_condition_impl))
 		{
 			auto ci = static_cast<detail::key_equals_number_condition_impl *>(b.m_impl);
 			auto ce = static_cast<detail::key_is_empty_condition_impl *>(a.m_impl);
@@ -1019,7 +1001,7 @@ inline condition operator or(condition &&a, condition &&b)

 /**
 * @brief A helper class to make it possible to search for empty items (NULL)
- * 
+ *
 * @code{.cpp}
 * "id"_key == cif::empty_type();
 * @endcode
@@ -1031,7 +1013,7 @@ struct empty_type

 /**
 * @brief A helper to make it possible to have conditions like
- * 
+ *
 * @code{.cpp}
 * "id"_key == cif::null;
 * @endcode
@@ -1041,14 +1023,14 @@ inline constexpr empty_type null = empty_type();

 /**
 * @brief Class to use in creating conditions, creates a reference to a item or item
- * 
+ *
 */
 struct key
 {
 	/**
 	 * @brief Construct a new key object using @a item_name as name
-	 * 
-	 * @param item_name 
+	 *
+	 * @param item_name
 	 */
 	explicit key(const std::string &item_name)
 		: m_item_name(item_name)
@@ -1057,8 +1039,8 @@ struct key

 	/**
 	 * @brief Construct a new key object using @a item_name as name
-	 * 
-	 * @param item_name 
+	 *
+	 * @param item_name
 	 */
 	explicit key(const char *item_name)
 		: m_item_name(item_name)
@@ -1067,8 +1049,8 @@ struct key

 	/**
 	 * @brief Construct a new key object using @a item_name as name
-	 * 
-	 * @param item_name 
+	 *
+	 * @param item_name
 	 */
 	explicit key(std::string_view item_name)
 		: m_item_name(item_name)
@@ -1090,7 +1072,8 @@ concept Numeric = ((std::is_floating_point_v<T> or std::is_integral_v<T>) and no
 template <Numeric T>
 condition operator==(const key &key, const T &v)
 {
-	return condition(new detail::key_equals_number_condition_impl(key.m_item_name, v));
+	// TODO: change key_equals_etc... to use std::variant<double,int64_t> or something
+	return condition(new detail::key_equals_number_condition_impl(key.m_item_name, static_cast<double>(v)));
 }

 /**
@@ -1137,13 +1120,10 @@ inline condition operator!=(const key &key, std::string_view value)
 template <Numeric T>
 condition operator>(const key &key, const T &v)
 {
-	std::ostringstream s;
-	s << " > " << v;
-
 	return condition(new detail::key_compare_condition_impl(
 		key.m_item_name, [item_name = key.m_item_name, v](row_handle r, bool icase)
 		{ return r[item_name].compare(v) > 0; },
-		s.str()));
+		cif::format(" > {}", v)));
 }

 /**
@@ -1152,13 +1132,10 @@ condition operator>(const key &key, const T &v)
 template <Numeric T>
 condition operator>=(const key &key, const T &v)
 {
-	std::ostringstream s;
-	s << " >= " << v;
-
 	return condition(new detail::key_compare_condition_impl(
 		key.m_item_name, [item_name = key.m_item_name, v](row_handle r, bool icase)
 		{ return r[item_name].compare(v) >= 0; },
-		s.str()));
+		cif::format(" >= {}", v)));
 }

 /**
@@ -1167,13 +1144,10 @@ condition operator>=(const key &key, const T &v)
 template <Numeric T>
 condition operator<(const key &key, const T &v)
 {
-	std::ostringstream s;
-	s << " < " << v;
-
 	return condition(new detail::key_compare_condition_impl(
 		key.m_item_name, [item_name = key.m_item_name, v](row_handle r, bool icase)
 		{ return r[item_name].compare(v) < 0; },
-		s.str()));
+		cif::format(" < {}", v)));
 }

 /**
@@ -1182,13 +1156,10 @@ condition operator<(const key &key, const T &v)
 template <Numeric T>
 condition operator<=(const key &key, const T &v)
 {
-	std::ostringstream s;
-	s << " <= " << v;
-
 	return condition(new detail::key_compare_condition_impl(
 		key.m_item_name, [item_name = key.m_item_name, v](row_handle r, bool icase)
 		{ return r[item_name].compare(v) <= 0; },
-		s.str()));
+		cif::format(" <= {}", v)));
 }

 /**
@@ -1196,13 +1167,10 @@ condition operator<=(const key &key, const T &v)
 */
 inline condition operator>(const key &key, std::string_view v)
 {
-	std::ostringstream s;
-	s << " > " << v;
-
 	return condition(new detail::key_compare_condition_impl(
 		key.m_item_name, [item_name = key.m_item_name, v](row_handle r, bool icase)
 		{ return r[item_name].compare(v, icase) > 0; },
-		s.str()));
+		cif::format(" > {}", v)));
 }

 /**
@@ -1210,13 +1178,10 @@ inline condition operator>(const key &key, std::string_view v)
 */
 inline condition operator>=(const key &key, std::string_view v)
 {
-	std::ostringstream s;
-	s << " >= " << v;
-
 	return condition(new detail::key_compare_condition_impl(
 		key.m_item_name, [item_name = key.m_item_name, v](row_handle r, bool icase)
 		{ return r[item_name].compare(v, icase) >= 0; },
-		s.str()));
+		cif::format(" >= {}", v)));
 }

 /**
@@ -1224,13 +1189,10 @@ inline condition operator>=(const key &key, std::string_view v)
 */
 inline condition operator<(const key &key, std::string_view v)
 {
-	std::ostringstream s;
-	s << " < " << v;
-
 	return condition(new detail::key_compare_condition_impl(
 		key.m_item_name, [item_name = key.m_item_name, v](row_handle r, bool icase)
 		{ return r[item_name].compare(v, icase) < 0; },
-		s.str()));
+		cif::format(" < {}", v)));
 }

 /**
@@ -1238,13 +1200,10 @@ inline condition operator<(const key &key, std::string_view v)
 */
 inline condition operator<=(const key &key, std::string_view v)
 {
-	std::ostringstream s;
-	s << " <= " << v;
-
 	return condition(new detail::key_compare_condition_impl(
 		key.m_item_name, [item_name = key.m_item_name, v](row_handle r, bool icase)
 		{ return r[item_name].compare(v, icase) <= 0; },
-		s.str()));
+		cif::format(" <= {}", v)));
 }

 /**
@@ -1345,7 +1304,7 @@ namespace literals
 {
 	/**
 	 * @brief Return a cif::key for the item name @a text
-	 * 
+	 *
 	 * @param text The name of the item
 	 * @param length The length of @a text
 	 * @return key The cif::key created
--- a/include/cif++/cql/transaction.hpp
+++ b/include/cif++/cql/transaction.hpp
@@ -0,0 +1,451 @@
+/*-
+ * SPDX-License-Identifier: BSD-2-Clause
+ *
+ * Copyright (c) 2025  NKI/AVL, Netherlands Cancer Institute
+ *
+ * Redistribution and use in source and binary forms, with or without
+ * modification, are permitted provided that the following conditions are met:
+ *
+ * 1. Redistributions of source code must retain the above copyright notice, this
+ *    list of conditions and the following disclaimer
+ * 2. Redistributions in binary form must reproduce the above copyright notice,
+ *    this list of conditions and the following disclaimer in the documentation
+ *    and/or other materials provided with the distribution.
+ *
+ * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
+ * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
+ * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
+ * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+ * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
+ * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
+ * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
+ * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
+ * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
+ */
+
+#pragma once
+
+#include "cif++/category.hpp"
+#include "cif++/condition.hpp"
+#include "cif++/datablock.hpp"
+#include "cif++/item.hpp"
+#include "cif++/row.hpp"
+#include "cif++/validate.hpp"
+
+#include <algorithm>
+#include <iterator>
+#include <memory>
+#include <stdexcept>
+#include <string>
+#include <vector>
+
+// --------------------------------------------------------------------
+
+namespace cif::cql
+{
+
+class result;
+class row;
+class transaction;
+class view;
+
+// --------------------------------------------------------------------
+
+struct column
+{
+	std::string name;
+	size_t index;
+};
+
+using column_list = std::vector<column>;
+
+// --------------------------------------------------------------------
+
+class field_ref
+{
+  public:
+	std::string_view name() const &
+	{
+		return m_col->name;
+	}
+
+	constexpr size_t num() const noexcept
+	{
+		return m_col->index;
+	}
+
+	std::string_view text() const &
+	{
+		return m_row[m_col->index].text();
+	}
+
+	/** Return the contents of this item as type @tparam T */
+	template <typename T = std::string>
+	auto as() const -> T
+	{
+		return m_row[m_col->index].as<T>();
+	}
+
+	/** Return the contents of this item as type @tparam T or, if not
+	 * set, use @a dv as the default value.
+	 */
+	template <typename T>
+	auto value_or(const T &dv) const
+	{
+		return m_row[m_col->index].value_or(dv);
+	}
+
+	field_ref(row_handle rh, column_list::const_iterator col)
+		: m_row(rh)
+		, m_col(col)
+	{
+	}
+
+	field_ref(const field_ref &) = default;
+	field_ref(field_ref &&) = default;
+
+	field_ref &operator=(const field_ref &) = default;
+	field_ref &operator=(field_ref &&) = default;
+
+  private:
+	row_handle m_row;
+	column_list::const_iterator m_col;
+};
+
+// --------------------------------------------------------------------
+
+class row_ref final
+{
+  public:
+	class const_field_iterator
+	{
+	  public:
+		friend class result;
+
+		using iterator_category = std::forward_iterator_tag;
+		using value_type = const field_ref;
+		using difference_type = std::ptrdiff_t;
+		using pointer = value_type *;
+		using reference = value_type &;
+
+		const_field_iterator(const const_field_iterator &) = default;
+		const_field_iterator(const_field_iterator &&) = default;
+
+		const_field_iterator &operator=(const const_field_iterator &) = default;
+		const_field_iterator &operator=(const_field_iterator &&) = default;
+
+		reference operator*()
+		{
+			return m_current;
+		}
+
+		pointer operator->()
+		{
+			return &m_current;
+		}
+
+		const_field_iterator &operator++()
+		{
+			if (m_row)
+			{
+				++m_col;
+				m_current = field_ref(m_row, m_col);
+			}
+
+			return *this;
+		}
+
+		const_field_iterator operator++(int)
+		{
+			const_field_iterator result(*this);
+			this->operator++();
+			return result;
+		}
+
+		bool operator==(const const_field_iterator &rhs) const
+		{
+			return m_row == rhs.m_row and m_col == rhs.m_col;
+		}
+
+		bool operator!=(const const_field_iterator &rhs) const
+		{
+			return m_row != rhs.m_row or m_col != rhs.m_col;
+		}
+
+	  private:
+		friend class row_ref;
+
+		const_field_iterator(const row_handle &row, column_list::const_iterator col)
+			: m_row(row)
+			, m_col(col)
+			, m_current(m_row, m_col)
+		{
+		}
+
+		row_handle m_row;
+		column_list::const_iterator m_col;
+		field_ref m_current;
+	};
+
+	// --------------------------------------------------------------------
+
+	row_ref() = default;
+
+	row_ref(row_handle rh, const column_list &cols)
+		: m_row(rh)
+		, m_cols(&cols)
+	{
+	}
+
+	row_ref(row_ref r, const column_list &cols)
+		: m_row(r.m_row)
+		, m_cols(&cols)
+	{
+	}
+
+	row_ref(const row_ref &) = default;
+	row_ref &operator=(const row_ref &) = default;
+
+	// --------------------------------------------------------------------
+
+	const_field_iterator cbegin() const noexcept { return const_field_iterator(m_row, m_cols->cbegin()); }
+	const_field_iterator begin() const noexcept { return const_field_iterator(m_row, m_cols->cbegin()); }
+	const_field_iterator cend() const noexcept { return const_field_iterator(m_row, m_cols->cend()); }
+	const_field_iterator end() const noexcept { return const_field_iterator(m_row, m_cols->cend()); }
+
+	field_ref front() const noexcept { return field_ref(m_row, m_cols->cbegin()); }
+	field_ref back() const noexcept { return field_ref(m_row, m_cols->cend()); }
+
+	size_t size() const noexcept { return m_cols->size(); }
+	bool empty() const noexcept { return m_cols->empty(); }
+
+	field_ref operator[](size_t index) const noexcept;
+	field_ref operator[](std::string_view name) const noexcept;
+
+	// --------------------------------------------------------------------
+
+	bool operator==(const row_ref &rhs) const { return m_row == rhs.m_row and m_cols == rhs.m_cols; }
+	bool operator!=(const row_ref &rhs) const { return m_row != rhs.m_row or m_cols != rhs.m_cols; }
+
+  private:
+	row_handle m_row;
+	const column_list *m_cols = nullptr;
+};
+
+// --------------------------------------------------------------------
+
+class view : public std::enable_shared_from_this<view>
+{
+  public:
+	virtual ~view() = default;
+
+	class const_row_iterator
+	{
+	  public:
+		friend class view;
+
+		using iterator_category = std::forward_iterator_tag;
+		using value_type = const row_ref;
+		using difference_type = std::ptrdiff_t;
+		using pointer = value_type *;
+		using reference = value_type &;
+
+		// const_row_iterator() = default;
+
+		const_row_iterator(const const_row_iterator &) = default;
+		const_row_iterator(const_row_iterator &&) = default;
+
+		// const_row_iterator &operator=(const const_row_iterator &) = default;
+		// const_row_iterator &operator=(const_row_iterator &&) = default;
+
+		reference operator*()
+		{
+			return m_current;
+		}
+
+		pointer operator->()
+		{
+			return &m_current;
+		}
+
+		const_row_iterator &operator++()
+		{
+			++m_index;
+			if (m_index < m_data.size())
+				m_current = m_data.at(m_index);
+			return *this;
+		}
+
+		const_row_iterator operator++(int)
+		{
+			const_row_iterator result(*this);
+			this->operator++();
+			return result;
+		}
+
+		bool operator==(const const_row_iterator &rhs) const
+		{
+			return &m_data == &rhs.m_data and m_index == rhs.m_index;
+		}
+
+		bool operator!=(const const_row_iterator &rhs) const
+		{
+			return &m_data != &rhs.m_data or m_index != rhs.m_index;
+		}
+
+	  private:
+		const_row_iterator(const view &result, size_t index, row_ref current)
+			: m_data(result)
+			, m_index(index)
+			, m_current(current)
+		{
+		}
+
+		const view &m_data;
+		size_t m_index = 0;
+		row_ref m_current;
+	};
+
+	// --------------------------------------------------------------------
+
+	const_row_iterator begin() const noexcept { return const_row_iterator(*this, 0, at(0)); }
+	const_row_iterator cbegin() const noexcept { return const_row_iterator(*this, 0, at(0)); }
+
+	const_row_iterator end() const noexcept { return const_row_iterator(*this, size(), row_ref{}); }
+	const_row_iterator cend() const noexcept { return const_row_iterator(*this, size(), row_ref{}); }
+
+	virtual row_ref front() const noexcept = 0;
+	virtual row_ref back() const noexcept = 0;
+
+	virtual size_t size() const noexcept = 0;
+	bool empty() const noexcept { return size() == 0; }
+
+	virtual row_ref at(size_t index) const = 0;
+
+	// --------------------------------------------------------------------
+
+	std::vector<std::string> columns() const
+	{
+		std::vector<std::string> result;
+		for (const auto &[name, ignore] : m_columns)
+			result.emplace_back(name);
+		return result;
+	}
+
+  protected:
+	friend class const_row_iterator;
+
+	view(const column_list &cols)
+		: m_columns(cols)
+	{
+	}
+
+	view(column_list &&cols)
+		: m_columns(std::forward<column_list>(cols))
+	{
+	}
+
+	column_list m_columns;
+};
+
+// --------------------------------------------------------------------
+
+class simple_view : public view
+{
+  public:
+	simple_view(const category &cat)
+		: view(get_column_list_for_category(cat))
+		, m_cat(cat)
+	{
+	}
+
+	simple_view(const simple_view &) = default;
+	simple_view(simple_view &&) = default;
+
+	virtual size_t size() const noexcept override { return m_cat.size(); }
+
+	virtual row_ref front() const noexcept override;
+	virtual row_ref back() const noexcept override;
+
+	virtual row_ref at(size_t index) const override;
+
+  protected:
+
+	static column_list get_column_list_for_category(const category &cat);
+
+  const category &m_cat;
+};
+
+// --------------------------------------------------------------------
+
+class result
+{
+  public:
+	// --------------------------------------------------------------------
+
+	using const_row_iterator = view::const_row_iterator;
+
+	// --------------------------------------------------------------------
+
+	result();
+	result(result const &rhs) noexcept = default;
+	result(result &&rhs) noexcept = default;
+	result &operator=(result const &rhs) noexcept = default;
+	result &operator=(result &&rhs) noexcept = default;
+
+	result(view &vw, const std::string &query = "");
+
+	row_ref one_row() const;
+	field_ref one_field() const;
+
+	// --------------------------------------------------------------------
+
+	const_row_iterator begin() const noexcept;
+	const_row_iterator cbegin() const noexcept;
+
+	const_row_iterator end() const noexcept;
+	const_row_iterator cend() const noexcept;
+
+	row_ref front() const noexcept;
+	row_ref back() const noexcept;
+
+	size_t size() const noexcept;
+	bool empty() const noexcept;
+
+	size_t column_count() const;
+
+  private:
+	friend class transaction;
+	friend class SelectStatement;
+
+	result expect_columns(size_t cols) const
+	{
+		if (auto actual = column_count(); cols != actual)
+			throw std::runtime_error("Unexpected number of columns");
+		return *this;
+	}
+
+	row_ref at(size_t index) const;
+
+	std::string m_query;
+	std::shared_ptr<view> m_view;
+};
+
+// --------------------------------------------------------------------
+
+class transaction
+{
+  public:
+	transaction(const datablock &db)
+		: m_db(const_cast<datablock &>(db))
+	{
+	}
+
+	result exec(std::string_view query);
+
+  private:
+	datablock &m_db;
+};
+
+} // namespace cif::cql
--- a/include/cif++/datablock.hpp
+++ b/include/cif++/datablock.hpp
@@ -43,7 +43,7 @@ namespace cif

 /**
 * @brief A datablock is a list of category objects with some additional features
- * 
+ *
 */

 class datablock : public std::list<category>
@@ -53,7 +53,7 @@ class datablock : public std::list<category>

 	/**
 	 * @brief Construct a new datablock object with name @a name
-	 * 
+	 *
 	 * @param name The name for the new datablock
 	 */
 	datablock(std::string_view name)
@@ -80,7 +80,7 @@ class datablock : public std::list<category>
 	{
 		std::swap(a.m_name, b.m_name);
 		std::swap(a.m_validator, b.m_validator);
-		std::swap(static_cast<std::list<category>&>(a), static_cast<std::list<category>&>(b));
+		std::swap(static_cast<std::list<category> &>(a), static_cast<std::list<category> &>(b));
 	}

 	// --------------------------------------------------------------------
@@ -92,7 +92,7 @@ class datablock : public std::list<category>

 	/**
 	 * @brief Set the name of this datablock to @a name
-	 * 
+	 *
 	 * @param name The new name
 	 */
 	void set_name(std::string_view name)
@@ -102,56 +102,55 @@ class datablock : public std::list<category>

 	/**
 	 * @brief Attempt to load the dictionary specified in audit_conform category
-	 * 
+	 *
 	 */
 	void load_dictionary();

 	/**
 	 * @brief Set the validator object to @a v
-	 * 
+	 *
 	 * @param v The new validator object, may be null
 	 */
 	void set_validator(const validator *v);

 	/**
 	 * @brief Get the validator object
-	 * 
+	 *
 	 * @return const validator* The validator or nullptr if there is none
 	 */
 	const validator *get_validator() const;

 	/**
 	 * @brief Validates the content of this datablock and all its content
-	 * 
+	 *
 	 * @return true If the content is valid
 	 * @return false If the content is not valid
 	 */
 	bool is_valid() const;

-	/**
-	 * @brief Validates the content of this datablock and all its content
-	 * and updates or removes the audit_conform category to match the result.
-	 * 
-	 * @return true If the content is valid
-	 * @return false If the content is not valid
-	 */
-	bool is_valid();
-
 	/**
 	 * @brief Validates all contained data for valid links between parents and children
 	 * as defined in the validator
-	 * 
+	 *
 	 * @return true If all links are valid
 	 * @return false If all links are not valid
 	 */
 	bool validate_links() const;

+	/**
+	 * @brief Strip removes all categories and items that are invalid according
+	 * to the assigned validator. Will also add a valid audit_conform block.
+	 *
+	 * @return true if the remaining datablock is valid
+	 */
+	bool strip();
+
 	// --------------------------------------------------------------------

 	/**
 	 * @brief Return the category named @a name, will create a new and empty
 	 * category named @a name if it does not exist.
-	 * 
+	 *
 	 * @param name The name of the category to return
 	 * @return category& Reference to the named category
 	 */
@@ -160,7 +159,7 @@ class datablock : public std::list<category>
 	/**
 	 * @brief Return the const category named @a name, will return a reference
 	 * to a static empty category if it was not found.
-	 * 
+	 *
 	 * @param name The name of the category to return
 	 * @return category& Reference to the named category
 	 */
@@ -169,7 +168,7 @@ class datablock : public std::list<category>
 	/**
 	 * @brief Return a pointer to the category named @a name or nullptr if
 	 * it does not exist.
-	 * 
+	 *
 	 * @param name The name of the category
 	 * @return category* Pointer to the category found or nullptr
 	 */
@@ -178,18 +177,26 @@ class datablock : public std::list<category>
 	/**
 	 * @brief Return a pointer to the category named @a name or nullptr if
 	 * it does not exist.
-	 * 
+	 *
 	 * @param name The name of the category
 	 * @return category* Pointer to the category found or nullptr
 	 */
 	const category *get(std::string_view name) const;

+	/**
+	 * @brief Return true if this datablock contains a non-empty category
+	 */
+	bool contains(std::string_view name) const
+	{
+		return get(name) != nullptr;
+	}
+
 	/**
 	 * @brief Tries to find a category with name @a name and will create a
 	 * new one if it is not found. The result is a tuple of an iterator
 	 * pointing to the category and a boolean indicating whether the category
 	 * was created or not.
-	 * 
+	 *
 	 * @param name The name for the category
 	 * @return std::tuple<iterator, bool> A tuple containing an iterator pointing
 	 * at the category and a boolean indicating whether the category was newly
--- a/include/cif++/format.hpp
+++ b/include/cif++/format.hpp
@@ -26,138 +26,28 @@

 #pragma once

+#if __has_include(<format>)
+#include <format>
+#define USE_STD_FORMAT 1
+#else
+#include <fmt/format.h>
+#endif
+
 #include <string>

 /**  \file format.hpp
 * 
- * File containing a basic reimplementation of boost::format
- * but then a bit more simplistic. Still this allowed me to move my code
- * from using boost::format to something without external dependency easily.
+ * Now using cif::format instead of a home grown rip off
 */

 namespace cif
 {

-namespace detail
-{
-	template <typename T>
-	struct to_varg
-	{
-		using type = T;
-
-		to_varg(const T &v)
-			: m_value(v)
-		{
-		}
-
-		type operator*() { return m_value; }
-
-		T m_value;
-	};
-
-	template <>
-	struct to_varg<const char *>
-	{
-		using type = const char *;
-
-		to_varg(const char *v)
-			: m_value(v)
-		{
-		}
-
-		type operator*() { return m_value.c_str(); }
-
-		std::string m_value;
-	};
-
-	template <>
-	struct to_varg<std::string>
-	{
-		using type = const char *;
-
-		to_varg(const std::string &v)
-			: m_value(v)
-		{
-		}
-
-		type operator*() { return m_value.c_str(); }
-
-		std::string m_value;
-	};
-
-} // namespace
-
-/** @cond */
-
-template <typename... Args>
-class format_plus_arg
-{
-  public:
-	using args_vector_type = std::tuple<detail::to_varg<Args>...>;
-	using vargs_vector_type = std::tuple<typename detail::to_varg<Args>::type...>;
-
-	format_plus_arg(const format_plus_arg &) = delete;
-	format_plus_arg &operator=(const format_plus_arg &) = delete;
-
-
-	format_plus_arg(std::string_view fmt, Args... args)
-		: m_fmt(fmt)
-		, m_args(std::forward<Args>(args)...)
-	{
-		auto ix = std::make_index_sequence<sizeof...(Args)>();
-		copy_vargs(ix);
-	}
-
-	std::string str()
-	{
-		char buffer[1024];
-		std::string::size_type r = std::apply(snprintf, std::tuple_cat(std::make_tuple(buffer, sizeof(buffer), m_fmt.c_str()), m_vargs));
-		return { buffer, r };
-	}
-
-	friend std::ostream &operator<<(std::ostream &os, const format_plus_arg &f)
-	{
-		char buffer[1024];
-		std::string::size_type r = std::apply(snprintf, std::tuple_cat(std::make_tuple(buffer, sizeof(buffer), f.m_fmt.c_str()), f.m_vargs));
-		os.write(buffer, r);
-		return os;
-	}
-
-  private:
-
-	template <std::size_t... I>
-	void copy_vargs(std::index_sequence<I...>)
-	{
-		((std::get<I>(m_vargs) = *std::get<I>(m_args)), ...);
-	}
-
-	std::string m_fmt;
-	args_vector_type m_args;
-	vargs_vector_type m_vargs;
-};
-
-/** @endcond */
-
-/**
- * @brief A simplistic reimplementation of boost::format, in fact it is
- * actually a way to call the C function snprintf to format the arguments
- * in @a args into the format string @a fmt
- * 
- * The string in @a fmt should thus be a C style format string.
- * 
- * TODO: Move to C++23 style of printing.
- * 
- * @tparam Args The types of the arguments
- * @param fmt The format string
- * @param args The arguments
- * @return An object that can be written out to a std::ostream using operator<<
- */
-
-template <typename... Args>
-constexpr auto format(std::string_view fmt, Args... args)
-{
-	return format_plus_arg(fmt, std::forward<Args>(args)...);
-}
+#if USE_STD_FORMAT
+using std::format;
+#else
+using fmt::format;
+#endif

 // --------------------------------------------------------------------
 /// A streambuf that fills out lines with spaces up until a specified width
--- a/include/cif++/gzio.hpp
+++ b/include/cif++/gzio.hpp
@@ -1,7 +1,33 @@
-//          Copyright Maarten L. Hekkelman, 2022
-// Distributed under the Boost Software License, Version 1.0.
-//    (See accompanying file LICENSE_1_0.txt or copy at
-//          http://www.boost.org/LICENSE_1_0.txt)
+/*-
+ * SPDX-License-Identifier: BSD-2-Clause
+ * 
+ * Copyright (c) 2025 NKI/AVL, Netherlands Cancer Institute
+ * 
+ * Redistribution and use in source and binary forms, with or without
+ * modification, are permitted provided that the following conditions are met:
+ * 
+ * 1. Redistributions of source code must retain the above copyright notice, this
+ *    list of conditions and the following disclaimer
+ * 2. Redistributions in binary form must reproduce the above copyright notice,
+ *    this list of conditions and the following disclaimer in the documentation
+ *    and/or other materials provided with the distribution.
+ * 
+ * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
+ * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
+ * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
+ * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+ * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
+ * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
+ * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
+ * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
+ * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
+ */
+
+/*
+	Original code comes from libgxrio at https://github.com/mhekkel/gxrio
+	This is a stripped down version.
+*/

 #pragma once

--- a/include/cif++/item.hpp
+++ b/include/cif++/item.hpp
@@ -53,12 +53,12 @@ namespace cif
 // --------------------------------------------------------------------
 /** @brief item is a transient class that is used to pass data into rows
 * but it also takes care of formatting data.
- * 
- * 
- * 
+ *
+ *
+ *
 * The class cif::item is often used implicitly when creating a row in a category
 * using the emplace function.
- * 
+ *
 * @code{.cpp}
 * cif::category cat("my-cat");
 * cat.emplace({
@@ -68,12 +68,12 @@ namespace cif
 *   { "item-4", std::make_optional<int>(42) },   // <- stores an item with value 42
 *   { "item-5" }                                 // <- stores an item with value .
 * });
- * 
+ *
 * std::cout << cat << '\n';
 * @endcode
- * 
+ *
 * Will result in:
- * 
+ *
 * @code{.txt}
 * _my-cat.item-1 1
 * _my-cat.item-2 1.00
@@ -176,7 +176,7 @@ class item

 	/// \brief constructor for an item with name \a name and as
 	/// content value \a value
-	template<typename T, std::enable_if_t<std::is_same_v<T, std::string>, int> = 0>
+	template <typename T, std::enable_if_t<std::is_same_v<T, std::string>, int> = 0>
 	item(const std::string_view name, T &&value)
 		: m_name(name)
 		, m_value(std::move(value))
@@ -221,8 +221,8 @@ class item
 	item &operator=(item &&rhs) noexcept = default;
 	/** @endcond */

-	std::string_view name() const { return m_name; }   ///< Return the name of the item
-	std::string_view value() const & { return m_value; } ///< Return the value of the item
+	std::string_view name() const { return m_name; }            ///< Return the name of the item
+	std::string_view value() const & { return m_value; }        ///< Return the value of the item
 	std::string value() const && { return std::move(m_value); } ///< Return the value of the item

 	/// \brief replace the content of the stored value with \a v
@@ -250,6 +250,8 @@ class item
 			return value();
 	}

+	auto operator<=>(const item &rhs) const = default;
+
  private:
 	std::string_view m_name;
 	std::string m_value;
@@ -558,7 +560,9 @@ struct item_handle::item_value_as<T, std::enable_if_t<std::is_arithmetic_v<T> an
 			auto b = txt.data();
 			auto e = txt.data() + txt.size();

-			std::from_chars_result r = (b + 1 < e and *b == '+' and std::isdigit(b[1])) ? selected_charconv<value_type>::from_chars(b + 1, e, result) : selected_charconv<value_type>::from_chars(b, e, result);
+			std::from_chars_result r = (b + 1 < e and *b == '+' and std::isdigit(b[1])) //
+			                               ? from_chars(b + 1, e, result)
+			                               : from_chars(b, e, result);

 			if ((bool)r.ec or r.ptr != e)
 			{
@@ -593,7 +597,9 @@ struct item_handle::item_value_as<T, std::enable_if_t<std::is_arithmetic_v<T> an
 			auto b = txt.data();
 			auto e = txt.data() + txt.size();

-			std::from_chars_result r = (b + 1 < e and *b == '+' and std::isdigit(b[1])) ? selected_charconv<value_type>::from_chars(b + 1, e, v) : selected_charconv<value_type>::from_chars(b, e, v);
+			std::from_chars_result r = (b + 1 < e and *b == '+' and std::isdigit(b[1]))
+			                               ? from_chars(b + 1, e, v)
+			                               : from_chars(b, e, v);

 			if ((bool)r.ec or r.ptr != e)
 			{
--- a/include/cif++/matrix.hpp
+++ b/include/cif++/matrix.hpp
@@ -124,6 +124,23 @@ class matrix_expression

 		return os;
 	}
+
+	template <typename M2>
+	constexpr bool operator==(const matrix_expression<M2> &m) const
+	{
+		bool same = false;
+		if (dim_m() == m.dim_m() and dim_n() == m.dim_n())
+		{
+			same = true;
+			for (std::size_t i = 0; same and i < m.dim_m(); ++i)
+			{
+				for (std::size_t j = 0; same and j < m.dim_n(); ++j)
+					same = operator()(i, j) == m(i, j);
+			}
+		}
+
+		return same;
+	}
 };

 // --------------------------------------------------------------------
@@ -594,6 +611,35 @@ auto operator*(const matrix_expression<M1> &m1, const matrix_expression<M2> &m2)

 // --------------------------------------------------------------------

+template <typename M2>
+class sub_matrix : public matrix_expression<sub_matrix<M2>>
+{
+  public:
+	sub_matrix(const M2 &m, int i, int j)
+		: m_m(m)
+		, m_i(i)
+		, m_j(j)
+	{
+	}
+
+	constexpr std::size_t dim_m() const { return m_m.dim_m() - 1; } ///< Return dimension m
+	constexpr std::size_t dim_n() const { return m_m.dim_n() - 1; } ///< Return dimension n
+
+	/** Access to the value of element [ @a i, @a j ] */
+	constexpr auto operator()(std::size_t i, std::size_t j) const
+	{
+		return m_m(
+			i >= m_i ? i + 1 : i,
+			j >= m_j ? j + 1 : j);
+	}
+
+  private:
+	const M2 &m_m;
+	std::size_t m_i, m_j;
+};
+
+// --------------------------------------------------------------------
+
 /** Generic routine to calculate the determinant of a matrix
 * 
 * @note This is currently only implemented for fixed matrices of size 3x3
@@ -605,11 +651,23 @@ auto determinant(const M &m);
 template <typename F = float>
 auto determinant(const matrix3x3<F> &m)
 {
-	return (m(0, 0) * (m(1, 1) * m(2, 2) - m(1, 2) * m(2, 1)) +
-			m(0, 1) * (m(1, 2) * m(2, 0) - m(1, 0) * m(2, 2)) +
-			m(0, 2) * (m(1, 0) * m(2, 1) - m(1, 1) * m(2, 0)));
+	return (m(0, 0) * ((m(1, 1) * m(2, 2) - m(1, 2) * m(2, 1))) +
+			m(0, 1) * ((m(1, 2) * m(2, 0) - m(1, 0) * m(2, 2))) +
+			m(0, 2) * ((m(1, 0) * m(2, 1) - m(1, 1) * m(2, 0))));
 }

+/** Implementation of the determinant function for fixed size matrices of size 4x4 */
+template <typename F = float>
+F determinant(const matrix4x4<F> &m)
+{
+	return m(0, 0) * determinant(matrix3x3<F>(sub_matrix<decltype(m)>(m, 0, 0))) -
+	       m(0, 1) * determinant(matrix3x3<F>(sub_matrix<decltype(m)>(m, 0, 1))) +
+	       m(0, 2) * determinant(matrix3x3<F>(sub_matrix<decltype(m)>(m, 0, 2))) -
+	       m(0, 3) * determinant(matrix3x3<F>(sub_matrix<decltype(m)>(m, 0, 3)));
+}
+
+// --------------------------------------------------------------------
+
 /** Generic routine to calculate the inverse of a matrix
 * 
 * @note This is currently only implemented for fixed matrices of size 3x3
--- a/include/cif++/model.hpp
+++ b/include/cif++/model.hpp
@@ -29,12 +29,13 @@
 #include "cif++/atom_type.hpp"
 #include "cif++/datablock.hpp"
 #include "cif++/point.hpp"
+#include "cif++/row.hpp"

 #include <memory>
 #include <numeric>

 #if __cpp_lib_format
-#include <format>
+# include <format>
 #endif

 /** @file model.hpp
@@ -106,8 +107,6 @@ class atom

 		atom_impl(const atom_impl &i) = default;

-		void prefetch();
-
 		int compare(const atom_impl &b) const;

 		// bool getAnisoU(float anisou[6]) const;
@@ -136,14 +135,20 @@ class atom

 		row_handle row_aniso()
 		{
+			row_handle result{};
 			auto cat = m_db.get("atom_site_anisotrop");
-			return cat ? cat->operator[]({ { "id", m_id } }) : row_handle{};
+			if (cat)
+				result = cat->operator[]({ { "id", m_id } });
+			return result;
 		}

 		const row_handle row_aniso() const
 		{
+			row_handle result{};
 			auto cat = m_db.get("atom_site_anisotrop");
-			return cat ? cat->operator[]({ { "id", m_id } }) : row_handle{};
+			if (cat)
+				result = cat->operator[]({ { "id", m_id } });
+			return result;
 		}

 		const datablock &m_db;
@@ -345,7 +350,7 @@ class atom
 	std::string get_auth_asym_id() const { return get_property("auth_asym_id"); }      ///< Return the auth_asym_id property
 	std::string get_auth_seq_id() const { return get_property("auth_seq_id"); }        ///< Return the auth_seq_id property
 	std::string get_auth_atom_id() const { return get_property("auth_atom_id"); }      ///< Return the auth_atom_id property
-	std::string get_auth_alt_id() const { return get_property("auth_alt_id"); }        ///< Return the auth_alt_id property
+	std::string get_auth_alt_id() const { return get_property("pdbx_auth_alt_id"); }   ///< Return the auth_alt_id property
 	std::string get_auth_comp_id() const { return get_property("auth_comp_id"); }      ///< Return the auth_comp_id property
 	std::string get_pdb_ins_code() const { return get_property("pdbx_PDB_ins_code"); } ///< Return the pdb_ins_code property

@@ -481,8 +486,8 @@ class residue
 		, m_compound_id(compoundID)
 		, m_asym_id(asymID)
 		, m_seq_id(seqID)
-		, m_auth_asym_id(authAsymID)
-		, m_auth_seq_id(authSeqID)
+		, m_pdb_strand_id(authAsymID)
+		, m_pdb_seq_num(authSeqID)
 		, m_pdb_ins_code(pdbInsCode)
 	{
 	}
@@ -509,9 +514,9 @@ class residue
 	const std::string &get_asym_id() const { return m_asym_id; } ///< Return the asym_id
 	int get_seq_id() const { return m_seq_id; }                  ///< Return the seq_id

-	const std::string get_auth_asym_id() const { return m_auth_asym_id; } ///< Return the auth_asym_id
-	const std::string get_auth_seq_id() const { return m_auth_seq_id; }   ///< Return the auth_seq_id
-	std::string get_pdb_ins_code() const { return m_pdb_ins_code; }       ///< Return the pdb_ins_code
+	const std::string get_pdb_strand_id() const { return m_pdb_strand_id; } ///< Return the pdb_strand_id
+	const std::string get_pdb_seq_num() const { return m_pdb_seq_num; }     ///< Return the pdb_seq_num
+	std::string get_pdb_ins_code() const { return m_pdb_ins_code; }         ///< Return the pdb_ins_code

 	const std::string &get_compound_id() const { return m_compound_id; } ///< Return the compound_id
 	void set_compound_id(const std::string &id) { m_compound_id = id; }  ///< Set the compound_id to @a id
@@ -540,6 +545,9 @@ class residue
 	/// \brief Return the atom with atom_id @a atomID
 	atom get_atom_by_atom_id(const std::string &atomID) const;

+	/// \brief Return the atom with atom_id @a atomID and alternate_id @a altID
+	atom get_atom_by_atom_id(const std::string &atomID, const std::string &altID) const;
+
 	/// \brief Return the list of atoms having ID \a atomID
 	///
 	/// This includes all alternate atoms with this ID
@@ -577,7 +585,7 @@ class residue
 								   m_seq_id == rhs.m_seq_id and
 								   m_asym_id == rhs.m_asym_id and
 								   m_compound_id == rhs.m_compound_id and
-								   m_auth_seq_id == rhs.m_auth_seq_id);
+								   m_pdb_seq_num == rhs.m_pdb_seq_num);
 	}

 	/// @brief Create a new atom and add it to the list
@@ -591,7 +599,7 @@ class residue
 	structure *m_structure = nullptr;
 	std::string m_compound_id, m_asym_id;
 	int m_seq_id = 0;
-	std::string m_auth_asym_id, m_auth_seq_id, m_pdb_ins_code;
+	std::string m_pdb_strand_id, m_pdb_seq_num, m_pdb_ins_code;
 	std::vector<atom> m_atoms;
 	/** @endcond */
 };
@@ -622,6 +630,9 @@ class monomer : public residue
 	bool is_first_in_chain() const; ///< Return if this residue is the first residue in the chain
 	bool is_last_in_chain() const;  ///< Return if this residue is the last residue in the chain

+	const monomer &prev() const; // Return previous monomer in polymer
+	const monomer &next() const; // Return next monomer in polymer
+
 	// convenience
 	bool has_alpha() const; ///< Return if a alpha value can be calculated (depends on location in chain)
 	bool has_kappa() const; ///< Return if a kappa value can be calculated (depends on location in chain)
@@ -708,15 +719,15 @@ class polymer : public std::vector<monomer>

 	structure *get_structure() const { return m_structure; } ///< Return the structure

-	std::string get_asym_id() const { return m_asym_id; }           ///< Return the asym_id
-	std::string get_auth_asym_id() const { return m_auth_asym_id; } ///< Return the PDB chain ID, actually
-	std::string get_entity_id() const { return m_entity_id; }       ///< Return the entity_id
+	std::string get_asym_id() const { return m_asym_id; }             ///< Return the asym_id
+	std::string get_pdb_strand_id() const { return m_pdb_strand_id; } ///< Return the PDB chain ID, actually
+	std::string get_entity_id() const { return m_entity_id; }         ///< Return the entity_id

  private:
 	structure *m_structure;
 	std::string m_entity_id;
 	std::string m_asym_id;
-	std::string m_auth_asym_id;
+	std::string m_pdb_strand_id;
 };

 // --------------------------------------------------------------------
@@ -754,7 +765,7 @@ class sugar : public residue
 	int num() const
 	{
 		int result;
-		auto r = std::from_chars(m_auth_seq_id.data(), m_auth_seq_id.data() + m_auth_seq_id.length(), result);
+		auto r = std::from_chars(m_pdb_seq_num.data(), m_pdb_seq_num.data() + m_pdb_seq_num.length(), result);
 		if ((bool)r.ec)
 			throw std::runtime_error("The auth_seq_id should be a number for a sugar");
 		return result;
@@ -853,19 +864,38 @@ class branch : public std::vector<sugar>
 	std::string m_asym_id, m_entity_id;
 };

-// --------------------------------------------------------------------
-
-/// \brief A still very limited set of options for reading structures
-enum class StructureOpenOptions
+/** @brief Enumeration for controlling atom selection based on occupancy. */
+enum class occupancy_policy
 {
-	SkipHydrogen = 1 << 0 ///< Do not include hydrogen atoms in the structure object
+	/** @brief Include all atoms regardless of their occupancy factor. */
+	ALL = 0,
+
+	/** @brief Select only alternate atoms with the maximum occupancy factor.
+	 * If multiple atoms have the same maximum occupancy, choose the one with the minimum B-factor.
+	 * If multiple atoms share both the maximum occupancy and the minimum B-factor, select the first encountered atom.
+	 */
+	MAX = 1,
+
+	/** @brief Select only alternate atoms with the minimum occupancy factor.
+	 * Similar to MAX, if multiple atoms have the same minimum occupancy, choose the one with the minimum B-factor.
+	 * If multiple atoms share both the minimum occupancy and the minimum B-factor, select the first encountered atom.
+	 */
+	MIN = 2,
+
+	/** @brief Exclude all atoms with an occupancy factor greater than zero. */
+	UNOCCUPIED = 3
 };

-/// \brief A way to combine two options. Not very useful as there is only one...
-constexpr inline bool operator&(StructureOpenOptions a, StructureOpenOptions b)
+struct structure_open_options
 {
-	return static_cast<int>(a) bitand static_cast<int>(b);
-}
+	bool skip_hydrogen = false;                              ///< Do not include hydrogen atoms in the structure object
+	bool skip_hetatom = false;                               ///< Do not include HET atoms in the structure object
+	bool skip_water = false;                                 ///< Do not include water atoms in the structure object
+	occupancy_policy occupancy_mode = occupancy_policy::ALL; ///< By default, the occupancy policy is set to occupancy_policy::ALL
+	std::vector<std::string> asyms;                          ///< The asyms to load, if empty load all
+	std::optional<float> min_b_factor;                       ///< Only load atoms with at least this b_factor
+	std::optional<float> max_b_factor;                       ///< Only load atoms with at most this b_factor
+};

 // --------------------------------------------------------------------

@@ -879,10 +909,10 @@ class structure
 {
  public:
 	/// \brief Read the structure from cif::file @a p
-	structure(file &p, std::size_t modelNr = 1, StructureOpenOptions options = {});
+	structure(file &p, std::size_t modelNr = 1, structure_open_options options = {});

 	/// \brief Load the structure from already parsed mmCIF data in @a db
-	structure(datablock &db, std::size_t modelNr = 1, StructureOpenOptions options = {});
+	structure(datablock &db, std::size_t modelNr = 1, structure_open_options options = {});

 	/** @cond */
 	structure(structure &&s) = default;
@@ -992,18 +1022,18 @@ class structure
 	/**
 	 * @brief Change residue @a res to a new compound ID optionally
 	 * remapping atoms.
-	 * 
+	 *
 	 * A new chem_comp entry as well as an entity is created if needed and
 	 * if the list of @a remappedAtoms is not empty it is used to remap.
-	 * 
+	 *
 	 * The array in @a remappedAtoms contains tuples of strings, both
 	 * strings contain an atom_id. The first is the one in the current
 	 * residue and the second is the atom_id that should be used instead.
 	 * If the second string is empty, the atom is removed from the residue.
-	 * 
-	 * @param res 
-	 * @param newcompound 
-	 * @param remappedAtoms 
+	 *
+	 * @param res
+	 * @param newcompound
+	 * @param remappedAtoms
 	 */
 	void change_residue(residue &res, const std::string &newcompound,
 		const std::vector<std::tuple<std::string, std::string>> &remappedAtoms);
@@ -1036,12 +1066,30 @@ class structure
 	/// \return				The newly create asym ID
 	std::string create_non_poly(const std::string &entity_id, std::vector<row_initializer> atoms);

+	/// \brief Create a new NonPolymer struct_asym for a compound of type \a compound_id, returns asym_id.
+	/// This method creates new atom records filled with info from the CCD compound info.
+	///
+	/// \param compound_id	 The compound ID of the new nonpoly
+	/// \param skip_hydrogen Do not create hydrogen atoms when true
+	/// \return				 The newly create asym ID
+	std::string create_non_poly(const std::string &compound_id, bool skip_hydrogen);
+
 	/// \brief Create a new water with atom constructed from info in \a atom_info
 	/// This method creates a new atom record filled with info from the info.
 	///
 	/// \param atom			The set of item data containing the data for the atoms.
 	void create_water(row_initializer atom);

+	/// \brief Create a link, a struct_conn record for two atoms.
+	///
+	/// \param a1			Atom 1
+	/// \param a2			Atom 2
+	/// \param link_type	The struct_conn_type ID for the link
+	/// \param role			The pdbx_role field value
+	/// \return 			The ID of the struct_conn record created
+
+	std::string create_link(atom a1, atom a2, const std::string &link_type, const std::string &role);
+
 	/// \brief Create a new and empty (sugar) branch
 	branch &create_branch();

@@ -1112,7 +1160,7 @@ class structure
 	friend polymer;
 	friend residue;

-	void load_atoms_for_model(StructureOpenOptions options);
+	void load_atoms_for_model(structure_open_options options);

 	std::string insert_compound(const std::string &compoundID, bool is_entity);

--- a/include/cif++/pdb.hpp
+++ b/include/cif++/pdb.hpp
@@ -104,6 +104,27 @@ inline void write(const std::filesystem::path &p, const file &f)

 // --------------------------------------------------------------------

+/**
+ * @brief Quickly fix a PDB file that lacks some often needed categories
+ * 
+ * This differs from reconstruct_pdbx which does a much more thorough job
+ * 
+ * \param pdbx_file The cif::file that hopefully contains some valid data
+ */
+
+void fixup_pdbx(file &pdbx_file);
+
+/**
+ * @brief Quickly fix a PDB file that lacks some often needed categories
+ * 
+ * This differs from reconstruct_pdbx which does a much more thorough job
+ * 
+ * \param pdbx_file The cif::file that hopefully contains some valid data
+ * \param v The validator to use
+ */
+
+void fixup_pdbx(file &pdbx_file, const validator &v);
+
 /** \brief Reconstruct all missing categories for an assumed PDBx file.
 *
 * Some people believe that simply dumping some atom records is enough.
--- a/include/cif++/point.hpp
+++ b/include/cif++/point.hpp
@@ -30,7 +30,9 @@
 #include <cmath>
 #include <complex>
 #include <cstdint>
+#include <format>
 #include <functional>
+#include <optional>
 #include <valarray>

 #if __has_include(<clipper/core/coords.h>)
@@ -365,11 +367,18 @@ class quaternion_type
 	}

 	/// \brief test for all zero values
-	constexpr operator bool() const
+	constexpr explicit operator bool() const
 	{
 		return a != 0 or b != 0 or c != 0 or d != 0;
 	}

+	/// \brief for debugging e.g.
+	friend std::ostream &operator<<(std::ostream &os, const quaternion_type &rhs)
+	{
+		os << std::format("{{ a: {}, b: {}, c: {}, d: {} }}", rhs.a, rhs.b, rhs.c, rhs.d);
+		return os;
+	}
+
  private:
 	value_type a, b, c, d;
 };
@@ -743,6 +752,55 @@ inline constexpr auto cross_product(const point_type<F1> &a, const point_type<F2
 		a.m_x * b.m_y - b.m_x * a.m_y);
 }

+/// \brief return the squared norm of point @a p
+template <typename F>
+constexpr F norm_squared(const point_type<F> &p)
+{
+	return p.m_x * p.m_x + p.m_y * p.m_y + p.m_z * p.m_z;
+}
+
+/// \brief return the norm of point @a p
+template <typename F>
+constexpr point_type<F> norm(const point_type<F> &p)
+{
+	return std::sqrt(norm_squared(p));
+}
+
+/// \brief return the point where two lines intersect, or an empty value if they don't intersect at all
+template <typename F>
+std::optional<cif::point> line_line_intersection(const point_type<F> &p1,
+	const point_type<F> &p2, const point_type<F> &p3, const point_type<F> &p4)
+{
+	auto p13 = p1 - p3;
+	auto p43 = p4 - p3;
+	if (std::abs(p43.m_x) < std::numeric_limits<F>::epsilon() and std::abs(p43.m_y) < std::numeric_limits<F>::epsilon() and std::abs(p43.m_z) < std::numeric_limits<F>::epsilon())
+		return {};
+
+	auto p21 = p2 - p1;
+	if (std::abs(p21.m_x) < std::numeric_limits<F>::epsilon() and std::abs(p21.m_y) < std::numeric_limits<F>::epsilon() and std::abs(p21.m_z) < std::numeric_limits<F>::epsilon())
+		return {};
+
+	auto d1343 = cif::dot_product(p43, p13);
+	auto d4321 = cif::dot_product(p43, p21);
+	auto d1321 = cif::dot_product(p13, p21);
+	auto d4343 = cif::dot_product(p43, p43);
+	auto d2121 = cif::dot_product(p21, p21);
+
+	auto denom = d2121 * d4343 - d4321 * d4321;
+	if (std::abs(denom) < std::numeric_limits<F>::epsilon())
+		return {};
+
+	auto numer = d1343 * d4321 - d1321 * d4343;
+
+	auto mua = numer / denom;
+	auto mub = (d1343 + d4321 * mua) / d4343;
+
+	auto pa = p1 + mua * p21;
+	auto pb = p3 + mub * p43;
+
+	return { (pa + pb) / 2 };
+}
+
 /// \brief return the angle in degrees between the vectors from point @a p2 to @a p1 and @a p2 to @a p3
 template <typename F>
 constexpr auto angle(const point_type<F> &p1, const point_type<F> &p2, const point_type<F> &p3)
@@ -806,6 +864,9 @@ constexpr auto distance_point_to_line(const point_type<F> &l1, const point_type<
 	return cross.length() / line.length();
 }

+/// \brief return the smallest sphere around the points in @a pts
+std::tuple<point, float> smallest_sphere_around_points(std::vector<point> pts);
+
 // --------------------------------------------------------------------
 /**
 * @brief For e.g. simulated annealing, returns a new point that is moved in
--- a/include/cif++/text.hpp
+++ b/include/cif++/text.hpp
@@ -355,279 +355,35 @@ std::string cif_id_for_number(int number);
 std::vector<std::string> word_wrap(const std::string &text, std::size_t width);

 // --------------------------------------------------------------------
-/// \brief std::from_chars for floating point types.
-///
-/// These are optional, there's a selected_charconv class below that selects
-/// the best option to use based on support by the stl library.
-///
-/// I.e. that in case of GNU < 12 (or something) the cif implementation will
-/// be used, all other cases will use the stl version.

-template <typename FloatType, std::enable_if_t<std::is_floating_point_v<FloatType>, int> = 0>
-std::from_chars_result from_chars(const char *first, const char *last, FloatType &value)
-{
-	std::from_chars_result result{ first, {} };
-
-	enum State
-	{
-		IntegerSign,
-		Integer,
-		Fraction,
-		ExponentSign,
-		Exponent
-	} state = IntegerSign;
-	int sign = 1;
-	unsigned long long vi = 0;
-	int fl = 0, tz = 0;
-	int exponent_sign = 1;
-	int exponent = 0;
-	bool done = false;
-
-	while (not done and not (bool)result.ec)
-	{
-		char ch = result.ptr != last ? *result.ptr : 0;
-		++result.ptr;
-
-		switch (state)
-		{
-			case IntegerSign:
-				if (ch == '-')
-				{
-					sign = -1;
-					state = Integer;
-				}
-				else if (ch == '+')
-					state = Integer;
-				else if (ch >= '0' and ch <= '9')
-				{
-					vi = ch - '0';
-					state = Integer;
-				}
-				else if (ch == '.')
-					state = Fraction;
-				else
-					result.ec = std::errc::invalid_argument;
-				break;
-
-			case Integer:
-				if (ch >= '0' and ch <= '9')
-					vi = 10 * vi + (ch - '0');
-				else if (ch == 'e' or ch == 'E')
-					state = ExponentSign;
-				else if (ch == '.')
-					state = Fraction;
-				else
-				{
-					done = true;
-					--result.ptr;
-				}
-				break;
-
-			case Fraction:
-				if (ch >= '0' and ch <= '9')
-				{
-					vi = 10 * vi + (ch - '0');
-
-					if (ch == '0')
-						tz += 1;
-					else
-					{
-						fl += tz + 1;
-						tz = 0;
-					}
-				}
-				else if (ch == 'e' or ch == 'E')
-					state = ExponentSign;
-				else
-				{
-					done = true;
-					--result.ptr;
-				}
-				break;
-
-			case ExponentSign:
-				if (ch == '-')
-				{
-					exponent_sign = -1;
-					state = Exponent;
-				}
-				else if (ch == '+')
-					state = Exponent;
-				else if (ch >= '0' and ch <= '9')
-				{
-					exponent = ch - '0';
-					state = Exponent;
-				}
-				else
-					result.ec = std::errc::invalid_argument;
-				break;
-
-			case Exponent:
-				if (ch >= '0' and ch <= '9')
-					exponent = 10 * exponent + (ch - '0');
-				else
-				{
-					done = true;
-					--result.ptr;
-				}
-				break;
-		}
-	}
-
-	if (not (bool)result.ec)
-	{
-		while (tz-- > 0)
-			vi /= 10;
-
-		long double v = std::pow(10, -fl) * vi * sign;
-		if (exponent != 0)
-			v *= std::pow(10, exponent * exponent_sign);
-
-		if (std::isnan(v))
-			result.ec = std::errc::invalid_argument;
-		else if (std::abs(v) > std::numeric_limits<FloatType>::max())
-			result.ec = std::errc::result_out_of_range;
-
-		value = static_cast<FloatType>(v);
-	}
-
-	return result;
-}
-
-/// \brief duplication of std::chars_format for deficient STL implementations
-enum class chars_format
-{
-	scientific = 1,
-	fixed = 2,
-	// hex,
-	general = fixed | scientific
-};
-
-/// \brief a simplistic implementation of std::to_chars for old STL implementations
-template <typename FloatType, std::enable_if_t<std::is_floating_point_v<FloatType>, int> = 0>
-std::to_chars_result to_chars(char *first, char *last, FloatType &value, chars_format fmt)
-{
-	int size = static_cast<int>(last - first);
-	int r = 0;
-
-	switch (fmt)
-	{
-		case chars_format::scientific:
-			if constexpr (std::is_same_v<FloatType, long double>)
-				r = snprintf(first, last - first, "%le", value);
-			else
-				r = snprintf(first, last - first, "%e", value);
-			break;
-
-		case chars_format::fixed:
-			if constexpr (std::is_same_v<FloatType, long double>)
-				r = snprintf(first, last - first, "%lf", value);
-			else
-				r = snprintf(first, last - first, "%f", value);
-			break;
-
-		case chars_format::general:
-			if constexpr (std::is_same_v<FloatType, long double>)
-				r = snprintf(first, last - first, "%lg", value);
-			else
-				r = snprintf(first, last - first, "%g", value);
-			break;
-	}
-
-	std::to_chars_result result;
-	if (r < 0 or r >= size)
-		result = { first, std::errc::value_too_large };
-	else
-		result = { first + r, std::errc() };
-
-	return result;
-}
-
-/// \brief a simplistic implementation of std::to_chars for old STL implementations
-template <typename FloatType, std::enable_if_t<std::is_floating_point_v<FloatType>, int> = 0>
-std::to_chars_result to_chars(char *first, char *last, FloatType &value, chars_format fmt, int precision)
-{
-	int size = static_cast<int>(last - first);
-	int r = 0;
-
-	switch (fmt)
-	{
-		case chars_format::scientific:
-			if constexpr (std::is_same_v<FloatType, long double>)
-				r = snprintf(first, last - first, "%.*le", precision, value);
-			else
-				r = snprintf(first, last - first, "%.*e", precision, value);
-			break;
-
-		case chars_format::fixed:
-			if constexpr (std::is_same_v<FloatType, long double>)
-				r = snprintf(first, last - first, "%.*lf", precision, value);
-			else
-				r = snprintf(first, last - first, "%.*f", precision, value);
-			break;
-
-		case chars_format::general:
-			if constexpr (std::is_same_v<FloatType, long double>)
-				r = snprintf(first, last - first, "%.*lg", precision, value);
-			else
-				r = snprintf(first, last - first, "%.*g", precision, value);
-			break;
-	}
-
-	std::to_chars_result result;
-	if (r < 0 or r >= size)
-		result = { first, std::errc::value_too_large };
-	else
-		result = { first + r, std::errc() };
-
-	return result;
-}
-
-/// \brief class that uses our implementation of std::from_chars and std::to_chars
 template <typename T>
-struct my_charconv
-{
-	/// @brief Simply call our version of std::from_chars
-	static std::from_chars_result from_chars(const char *a, const char *b, T &d)
-	{
-		return cif::from_chars(a, b, d);
-	}
+using from_chars_function = decltype(std::from_chars(std::declval<const char *>(), std::declval<const char *>(), std::declval<T &>()));

-	/// @brief Simply call our version of std::to_chars
-	static std::to_chars_result to_chars(char *first, char *last, T &value, chars_format fmt)
-	{
-		return cif::to_chars(first, last, value, fmt);
-	}
-};
-
-/// \brief class that uses the STL implementation of std::from_chars and std::to_chars
 template <typename T>
 struct std_charconv
 {
-	/// @brief Simply call std::from_chars
 	static std::from_chars_result from_chars(const char *a, const char *b, T &d)
 	{
 		return std::from_chars(a, b, d);
 	}
-
-	/// @brief Simply call std::to_chars
-	static std::to_chars_result to_chars(char *first, char *last, T &value, chars_format fmt)
-	{
-		return std::to_chars(first, last, value, fmt);
-	}
 };

-/// \brief helper to find a from_chars function
-template <typename T>
-using from_chars_function = decltype(std::from_chars(std::declval<const char *>(), std::declval<const char *>(), std::declval<T &>()));
+template <typename T, typename = void>
+struct ff_charconv;

-/**
- * @brief Helper to select the best implementation of charconv based on availability of the
- * function in the std:: namespace
- *
- * @tparam T The type for which we want to find a from_chars/to_chars function
- */
 template <typename T>
-using selected_charconv = typename std::conditional_t<std_experimental::is_detected_v<from_chars_function, T>, std_charconv<T>, my_charconv<T>>;
+struct ff_charconv<T, typename std::enable_if_t<std::is_floating_point_v<T>>>
+{
+	static std::from_chars_result from_chars(const char *a, const char *b, T &v);
+};
+
+template <typename T>
+using charconv = typename std::conditional_t<std_experimental::is_detected_v<from_chars_function, T>, std_charconv<T>, ff_charconv<T>>;
+
+template <typename T>
+constexpr auto from_chars(const char *s, const char *e, T &v)
+{
+	return charconv<T>::from_chars(s, e, v);
+}

 } // namespace cif
--- a/include/cif++/utilities.hpp
+++ b/include/cif++/utilities.hpp
@@ -53,6 +53,7 @@
 #pragma warning(disable : 4068) // unknown pragma
 #pragma warning(disable : 4100) // unreferenced formal parameter
 #pragma warning(disable : 4101) // unreferenced local variable
+#pragma warning(disable : 4702) // unreachable code (too bad, this one. Happens in for loops)
 #define _SILENCE_CXX17_CODECVT_HEADER_DEPRECATION_WARNING 1
 #endif

@@ -295,6 +296,11 @@ class progress_bar
 	 */
 	void message(const std::string &inMessage);

+	/**
+	 * @brief Flush the progress bar to the output stream
+	 */
+	void flush();
+
  private:
 	progress_bar(const progress_bar &) = delete;
 	progress_bar &operator=(const progress_bar &) = delete;
--- a/pcre2-simple/CMakeLists.txt
+++ b/pcre2-simple/CMakeLists.txt
@@ -0,0 +1,316 @@
+# SPDX-License-Identifier: BSD-2-Clause
+# 
+# Copyright (c) 2025 Maarten L. Hekkelman
+# 
+# Redistribution and use in source and binary forms, with or without
+# modification, are permitted provided that the following conditions are met:
+# 
+# 1. Redistributions of source code must retain the above copyright notice, this
+#    list of conditions and the following disclaimer
+# 2. Redistributions in binary form must reproduce the above copyright notice,
+#    this list of conditions and the following disclaimer in the documentation
+#    and/or other materials provided with the distribution.
+# 
+# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
+# ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
+# WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
+# DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
+# ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+# (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
+# LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
+# ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
+# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
+# SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
+
+# A simplified wrapper CMakeLists.txt file for PCRE2
+#
+# This will generate an OBJECT library so it can be linked into another library
+
+cmake_minimum_required(VERSION 3.25)
+
+include(FetchContent)
+
+project(pcre2s VERSION 1.0.0 LANGUAGES C CXX)
+
+# The original code:
+
+file(DOWNLOAD https://github.com/PCRE2Project/pcre2/releases/download/pcre2-10.46/pcre2-10.46.tar.gz
+    ${CMAKE_CURRENT_BINARY_DIR}/pcre2-code.tgz
+    EXPECTED_HASH SHA256=8d28d7f2c3b970c3a4bf3776bcbb5adfc923183ce74bc8df1ebaad8c1985bd07)
+file(ARCHIVE_EXTRACT INPUT ${CMAKE_CURRENT_BINARY_DIR}/pcre2-code.tgz
+    DESTINATION ${CMAKE_CURRENT_BINARY_DIR})
+set(PCRE2_SOURCE_DIR ${CMAKE_CURRENT_BINARY_DIR}/pcre2-10.46)
+set(PCRE2_MAJOR 10)
+set(PCRE2_MINOR 46)
+set(PCRE2_VERSION "${PCRE2_MAJOR}.${PCRE2_MINOR}")
+set(PCRE2_DATE "2024-06-09")
+
+# Some needed configuration options
+
+# option(PCRE2_BUILD_PCRE2_8 "Build 8 bit PCRE2 library" ON)
+# option(PCRE2_BUILD_PCRE2_16 "Build 16 bit PCRE2 library" OFF)
+# option(PCRE2_BUILD_PCRE2_32 "Build 32 bit PCRE2 library" OFF)
+
+option(PCRE2_STATIC_PIC "Build the static library with the option position independent code enabled." OFF)
+
+set(PCRE2_NEWLINE "LF" CACHE STRING "What to recognize as a newline (one of CR, LF, CRLF, ANY, ANYCRLF, NUL)." FORCE)
+set_property(CACHE PCRE2_NEWLINE PROPERTY STRINGS "CR" "LF" "CRLF" "ANY" "ANYCRLF" "NUL")
+
+set(PCRE2_LINK_SIZE "2" CACHE STRING "Internal link size (2, 3 or 4 allowed). See LINK_SIZE in config.h.in for details.")
+set_property(CACHE PCRE2_LINK_SIZE PROPERTY STRINGS "2" "3" "4")
+
+set(PCRE2_PARENS_NEST_LIMIT "250" CACHE STRING "Default nested parentheses limit. See PARENS_NEST_LIMIT in config.h.in for details.")
+set(PCRE2_HEAP_LIMIT "20000000" CACHE STRING "Default limit on heap memory (kibibytes). See HEAP_LIMIT in config.h.in for details.")
+set(PCRE2_MAX_VARLOOKBEHIND "255" CACHE STRING "Default limit on variable lookbehinds.")
+set(PCRE2_MATCH_LIMIT "10000000" CACHE STRING "Default limit on internal looping. See MATCH_LIMIT in config.h.in for details.")
+set(PCRE2_MATCH_LIMIT_DEPTH "MATCH_LIMIT" CACHE STRING "Default limit on internal depth of search. See MATCH_LIMIT_DEPTH in config.h.in for details.")
+set(PCRE2GREP_BUFSIZE "20480" CACHE STRING "Buffer starting size parameter for pcre2grep. See PCRE2GREP_BUFSIZE in config.h.in for details.")
+set(PCRE2GREP_MAX_BUFSIZE "1048576" CACHE STRING "Buffer maximum size parameter for pcre2grep. See PCRE2GREP_MAX_BUFSIZE in config.h.in for details.")
+set(PCRE2_SUPPORT_JIT OFF CACHE BOOL "Enable support for Just-in-time compiling.")
+
+if(${CMAKE_SYSTEM_NAME} MATCHES Linux|NetBSD)
+    set(PCRE2_SUPPORT_JIT_SEALLOC OFF CACHE BOOL "Enable SELinux compatible execmem allocator in JIT (experimental).")
+else()
+    set(PCRE2_SUPPORT_JIT_SEALLOC IGNORE)
+endif()
+
+set(PCRE2GREP_SUPPORT_JIT ON CACHE BOOL "Enable use of Just-in-time compiling in pcre2grep.")
+set(PCRE2GREP_SUPPORT_CALLOUT ON CACHE BOOL "Enable callout string support in pcre2grep.")
+set(PCRE2GREP_SUPPORT_CALLOUT_FORK ON CACHE BOOL "Enable callout string fork support in pcre2grep.")
+set(PCRE2_SUPPORT_UNICODE ON CACHE BOOL "Enable support for Unicode and UTF-8/UTF-16/UTF-32 encoding.")
+set(PCRE2_SUPPORT_BSR_ANYCRLF OFF CACHE BOOL "ON=Backslash-R matches only LF CR and CRLF, OFF=Backslash-R matches all Unicode Linebreaks")
+set(PCRE2_NEVER_BACKSLASH_C OFF CACHE BOOL "If ON, backslash-C (upper case C) is locked out.")
+set(PCRE2_SUPPORT_VALGRIND OFF CACHE BOOL "Enable Valgrind support.")
+
+if(MINGW)
+    option(NON_STANDARD_LIB_PREFIX "ON=Shared libraries built in mingw will be named pcre2.dll, etc., instead of libpcre2.dll, etc." OFF)
+    option(NON_STANDARD_LIB_SUFFIX "ON=Shared libraries built in mingw will be named libpcre2-0.dll, etc., instead of libpcre2.dll, etc." OFF)
+endif()
+
+# 
+
+set(NEWLINE_DEFAULT "")
+
+if(PCRE2_NEWLINE STREQUAL "CR")
+    set(NEWLINE_DEFAULT "1")
+elseif(PCRE2_NEWLINE STREQUAL "LF")
+    set(NEWLINE_DEFAULT "2")
+elseif(PCRE2_NEWLINE STREQUAL "CRLF")
+    set(NEWLINE_DEFAULT "3")
+elseif(PCRE2_NEWLINE STREQUAL "ANY")
+    set(NEWLINE_DEFAULT "4")
+elseif(PCRE2_NEWLINE STREQUAL "ANYCRLF")
+    set(NEWLINE_DEFAULT "5")
+elseif(PCRE2_NEWLINE STREQUAL "NUL")
+    set(NEWLINE_DEFAULT "6")
+else()
+    message(FATAL_ERROR "The PCRE2_NEWLINE variable must be set to one of the following values: \"LF\", \"CR\", \"CRLF\", \"ANY\", \"ANYCRLF\".")
+endif()
+
+# Some tests
+
+include(CheckCSourceCompiles)
+include(CheckFunctionExists)
+include(CheckSymbolExists)
+include(CheckIncludeFile)
+
+check_include_file(assert.h HAVE_ASSERT_H)
+check_include_file(dirent.h HAVE_DIRENT_H)
+check_include_file(sys/stat.h HAVE_SYS_STAT_H)
+check_include_file(sys/types.h HAVE_SYS_TYPES_H)
+check_include_file(unistd.h HAVE_UNISTD_H)
+check_include_file(windows.h HAVE_WINDOWS_H)
+
+check_symbol_exists(bcopy "strings.h" HAVE_BCOPY)
+check_symbol_exists(memfd_create "sys/mman.h" HAVE_MEMFD_CREATE)
+check_symbol_exists(memmove "string.h" HAVE_MEMMOVE)
+check_symbol_exists(secure_getenv "stdlib.h" HAVE_SECURE_GETENV)
+check_symbol_exists(strerror "string.h" HAVE_STRERROR)
+
+check_c_source_compiles(
+  "int main(void) { char buf[128] __attribute__((uninitialized)); (void)buf; return 0; }"
+  HAVE_ATTRIBUTE_UNINITIALIZED
+)
+
+check_c_source_compiles(
+  [=[
+  extern __attribute__ ((visibility ("default"))) int f(void);
+  int main(void) { return f(); }
+  int f(void) { return 42; }
+  ]=]
+  HAVE_VISIBILITY
+)
+
+if(HAVE_VISIBILITY)
+  set(PCRE2_EXPORT [=[__attribute__ ((visibility ("default")))]=])
+else()
+  set(PCRE2_EXPORT)
+endif()
+
+check_c_source_compiles("int main(void) { __assume(1); return 0; }" HAVE_BUILTIN_ASSUME)
+
+check_c_source_compiles(
+  [=[
+  #include <stddef.h>
+  int main(void) { int a,b; size_t m; __builtin_mul_overflow(a,b,&m); return 0; }
+  ]=]
+  HAVE_BUILTIN_MUL_OVERFLOW
+)
+
+check_c_source_compiles(
+  "int main(int c, char *v[]) { if (c) __builtin_unreachable(); return (int)(*v[0]); }"
+  HAVE_BUILTIN_UNREACHABLE
+)
+
+# # Check whether Intel CET is enabled, and if so, adjust compiler flags. This
+# # code was written by PH, trying to imitate the logic from the autotools
+# # configuration.
+
+# check_c_source_compiles(
+#   [=[
+#   #ifndef __CET__
+#   #error CET is not enabled
+#   #endif
+#   int main() { return 0; }
+#   ]=]
+#   INTEL_CET_ENABLED
+# )
+
+# if(INTEL_CET_ENABLED)
+#   set(CMAKE_C_FLAGS "${CMAKE_C_FLAGS} -mshstk")
+# endif()
+
+# Set up some dependencies first
+
+configure_file(
+    ${PCRE2_SOURCE_DIR}/src/pcre2_chartables.c.dist
+    ${CMAKE_CURRENT_BINARY_DIR}/pcre2_chartables.c
+    COPYONLY
+)
+
+configure_file(
+    ${PCRE2_SOURCE_DIR}/config-cmake.h.in
+    ${CMAKE_CURRENT_BINARY_DIR}/interface/config.h
+    @ONLY
+)
+
+configure_file(
+    ${PCRE2_SOURCE_DIR}/src/pcre2.h.in
+    ${CMAKE_CURRENT_BINARY_DIR}/interface/pcre2.h
+    @ONLY
+)
+
+# Define our library
+
+list(APPEND PCRE2_HEADERS
+    ${CMAKE_CURRENT_BINARY_DIR}/interface/pcre2.h)
+
+list(APPEND PCRE2_SOURCES
+    ${PCRE2_SOURCE_DIR}/src/pcre2_auto_possess.c
+    ${CMAKE_CURRENT_BINARY_DIR}/pcre2_chartables.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_chkdint.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_compile.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_compile_class.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_config.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_context.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_convert.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_dfa_match.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_error.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_extuni.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_find_bracket.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_jit_compile.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_maketables.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_match.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_match_data.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_newline.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_ord2utf.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_pattern_info.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_script_run.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_serialize.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_string_utils.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_study.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_substitute.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_substring.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_tables.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_ucd.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_valid_utf.c
+    ${PCRE2_SOURCE_DIR}/src/pcre2_xclass.c
+)
+
+add_library(pcre2s OBJECT)
+
+target_sources(pcre2s
+    PRIVATE ${PCRE2_SOURCES}
+    PUBLIC
+    FILE_SET pcre2_headers TYPE HEADERS
+    BASE_DIRS ${PCRE2_SOURCE_DIR}/include ${CMAKE_CURRENT_BINARY_DIR}/interface
+    FILES ${PCRE2_HEADERS}
+)
+
+target_compile_definitions(pcre2s PUBLIC PCRE2_CODE_UNIT_WIDTH=8 HAVE_CONFIG_H)
+if(NOT BUILD_SHARED_LIBS)
+    target_compile_definitions(pcre2s PUBLIC PCRE2_STATIC)
+endif()
+
+target_include_directories(pcre2s PRIVATE ${CMAKE_CURRENT_BINARY_DIR}/interface ${PCRE2_SOURCE_DIR}/src)
+
+if(PCRE2_STATIC_PIC)
+    set_target_properties(pcre2s PROPERTIES POSITION_INDEPENDENT_CODE 1)
+endif()
+
+# # Installation and config files
+
+# include(CMakePackageConfigHelpers)
+# include(GenerateExportHeader)
+
+# # Install rules
+# install(TARGETS pcre2s
+#     EXPORT pcre2s
+#     FILE_SET pcre2_headers DESTINATION ${CMAKE_INSTALL_INCLUDEDIR})
+
+# if(MSVC AND BUILD_SHARED_LIBS)
+#     install(
+#         FILES $<TARGET_PDB_FILE:pcre2s>
+#         DESTINATION ${CMAKE_INSTALL_LIBDIR}
+#         OPTIONAL)
+# endif()
+
+# install(EXPORT pcre2s
+#     NAMESPACE pcre2s::
+#     FILE "pcre2s-targets.cmake"
+#     DESTINATION lib/cmake/pcre2s)
+
+# configure_package_config_file(
+#     ${CMAKE_CURRENT_SOURCE_DIR}/pcre2s-config.cmake.in ${CMAKE_CURRENT_BINARY_DIR}/pcre2s/pcre2s-config.cmake
+#     INSTALL_DESTINATION lib/cmake/pcre2s)
+
+# install(
+#     FILES "${CMAKE_CURRENT_BINARY_DIR}/pcre2s/pcre2s-config.cmake"
+#     "${CMAKE_CURRENT_BINARY_DIR}/pcre2s/pcre2s-config-version.cmake"
+#     DESTINATION lib/cmake/pcre2s)
+
+# set_target_properties(
+#     pcre2s
+#     PROPERTIES VERSION ${PCRE2_VERSION}
+#     SOVERSION ${PCRE2_VERSION}
+#     INTERFACE_pcre2s_MAJOR_VERSION ${PCRE2_MAJOR})
+
+# set_property(
+#     TARGET pcre2s
+#     APPEND
+#     PROPERTY COMPATIBLE_INTERFACE_STRING pcre2s_MAJOR_VERSION)
+
+# write_basic_package_version_file(
+#     "${CMAKE_CURRENT_BINARY_DIR}/pcre2s/pcre2s-config-version.cmake"
+#     VERSION "${PCRE2_VERSION}"
+#     COMPATIBILITY AnyNewerVersion)
+
+# # Testing
+
+# if(PROJECT_IS_TOP_LEVEL)
+#     include(CTest)
+
+#     if(BUILD_TESTING)
+#         add_subdirectory(test)
+#     endif()
+# endif()
--- a/rsrc/mmcif_pdbx.dic
+++ b/rsrc/mmcif_pdbx.dic
--- a/src/category.cpp
+++ b/src/category.cpp
@@ -92,7 +92,7 @@ class row_comparator
 		return d;
 	}

-	int operator()(const category &cat, const row_initializer &a, const row *b) const
+	int operator()(const category &cat, const category::key_type &a, const row *b) const
 	{
 		assert(b);

@@ -105,10 +105,11 @@ class row_comparator
 		{
 			assert(ai != a.end());

-			std::string_view ka = ai->value();
+			std::string_view ka = ai->value;
 			std::string_view kb = rhb[k].text();

-			d = f(ka, kb);
+			if (not (ai->may_be_null and rhb[k].empty()))
+				d = f(ka, kb);

 			if (d != 0)
 				break;
@@ -142,7 +143,7 @@ class category_index
 	}

 	row *find(const category &cat, row *k) const;
-	row *find_by_value(const category &cat, row_initializer k) const;
+	row *find_by_value(const category &cat, const category::key_type &k) const;

 	void insert(category &cat, row *r);
 	void erase(category &cat, row *r);
@@ -352,19 +353,19 @@ row *category_index::find(const category &cat, row *k) const
 	return r ? r->m_row : nullptr;
 }

-row *category_index::find_by_value(const category &cat, row_initializer k) const
+row *category_index::find_by_value(const category &cat, const category::key_type &k) const
 {
 	// sort the values in k first

-	row_initializer k2;
+	category::key_type k2;
 	for (auto &f : cat.key_item_indices())
 	{
 		auto fld = cat.get_item_name(f);

 		auto ki = find_if(k.begin(), k.end(), [&fld](auto &i)
-			{ return i.name() == fld; });
+			{ return i.name == fld; });
 		if (ki == k.end())
-			k2.emplace_back(fld, "");
+			k2.emplace_back(std::string{ fld }, "");
 		else
 			k2.emplace_back(*ki);
 	}
@@ -914,6 +915,24 @@ bool category::validate_links() const
 	return result;
 }

+void category::strip()
+{
+	std::vector<std::string> to_be_removed;
+
+	for (auto &item : m_items)
+	{
+		if (item.m_validator == nullptr)
+			to_be_removed.push_back(item.m_name);
+	}
+
+	for (auto item : to_be_removed)
+	{
+		if (cif::VERBOSE > 0)
+			std::clog << "Dropping item " << m_name << '.' << item << '\n';
+		remove_item(item);
+	}
+}
+
 // --------------------------------------------------------------------

 row_handle category::operator[](const key_type &key)
@@ -1320,8 +1339,7 @@ std::string category::get_unique_value(std::string_view item_name)
 		// brain-dead implementation
 		for (std::size_t ix = 0; ix < size(); ++ix)
 		{
-			// result = m_name + "-" + std::to_string(ix);
-			result = cif_id_for_number(ix);
+			result = cif_id_for_number(static_cast<int>(ix));
 			if (not contains(key(item_name) == result))
 				break;
 		}
--- a/src/compound.cpp
+++ b/src/compound.cpp
@@ -24,14 +24,38 @@
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
 */

-#include "cif++.hpp"
+#include "cif++/compound.hpp" // for compound_atom, compound_bond, compoun...

-#include <filesystem>
-#include <fstream>
-#include <map>
-#include <mutex>
-#include <numeric>
-#include <shared_mutex>
+#include "cif++/atom_type.hpp" // for atom_type_traits
+#include "cif++/category.hpp"  // for category
+#include "cif++/datablock.hpp" // for datablock
+#include "cif++/file.hpp"      // for file
+#include "cif++/item.hpp"      // for item
+#include "cif++/iterator.hpp"  // for iterator_proxy
+#include "cif++/parser.hpp"    // for parser
+#include "cif++/point.hpp"     // for distance, point
+#include "cif++/row.hpp"       // for tie, row_initializer, tie_wrap
+#include "cif++/text.hpp"      // for iequals, replace_all, iset
+#include "cif++/utilities.hpp" // for load_resource, VERBOSE, colour_type
+
+#include <algorithm>    // for find_if
+#include <cstddef>      // for size_t
+#include <exception>    // for exception, throw_with_nested
+#include <filesystem>   // for path, exists
+#include <fstream>      // for char_traits, basic_ostream, operator<<
+#include <iomanip>      // for operator<<, quoted
+#include <iostream>     // for clog, cout, cerr
+#include <limits>       // for numeric_limits
+#include <list>         // for _List_iterator
+#include <map>          // for allocator, map, _Rb_tree_iterator
+#include <memory>       // for shared_ptr, unique_ptr, __shared_ptr_...
+#include <optional>     // for optional
+#include <shared_mutex> // for shared_lock, shared_timed_mutex
+#include <stdexcept>    // for runtime_error, invalid_argument, out_...
+#include <string>       // for basic_string, string, operator==, ope...
+#include <string_view>  // for string_view, basic_string_view
+#include <utility>      // for pair, exchange, move
+#include <vector>       // for vector

 namespace fs = std::filesystem;

@@ -140,7 +164,7 @@ compound::compound(cif::datablock &db)

 	cif::tie(m_id, m_name, m_type, m_formula, m_formula_weight, m_formal_charge, one_letter_code, m_parent_id) =
 		chemComp.front().get("id", "name", "type", "formula", "formula_weight", "pdbx_formal_charge", "one_letter_code", "mon_nstd_parent_comp_id");
-	
+
 	if (one_letter_code.length() == 1)
 		m_one_letter_code = one_letter_code.front();

@@ -159,7 +183,7 @@ compound::compound(cif::datablock &db)
 		if (stereo_config.empty())
 			atom.stereo_config = stereo_config_type::N;
 		else
-		atom.stereo_config = parse_stereo_config_from_string(stereo_config);
+			atom.stereo_config = parse_stereo_config_from_string(stereo_config);
 		m_atoms.push_back(std::move(atom));
 	}

@@ -172,7 +196,7 @@ compound::compound(cif::datablock &db)
 		if (valueOrder.empty())
 			bond.type = bond_type::sing;
 		else
-		bond.type = parse_bond_type_from_string(valueOrder);
+			bond.type = parse_bond_type_from_string(valueOrder);
 		m_bonds.push_back(std::move(bond));
 	}
 }
@@ -231,12 +255,12 @@ float compound::bond_length(const std::string &atomId_1, const std::string &atom

 bool compound::is_peptide() const
 {
-	return iequals(m_type, "l-peptide linking")	or iequals(m_type, "peptide linking");
+	return iequals(m_type, "l-peptide linking") or iequals(m_type, "peptide linking");
 }

 bool compound::is_base() const
 {
-	return iequals(m_type, "dna linking")	or iequals(m_type, "rna linking");
+	return iequals(m_type, "dna linking") or iequals(m_type, "rna linking");
 }

 // --------------------------------------------------------------------
@@ -294,12 +318,31 @@ class compound_factory_impl : public std::enable_shared_from_this<compound_facto
 			delete c;
 	}

+	virtual bool exists_self(const std::string &id) const
+	{
+		if (m_missing.contains(id))
+			return false;
+
+		if (std::find_if(m_compounds.begin(), m_compounds.end(), [id](compound *c)
+				{ return c->id() == id; }) != m_compounds.end())
+			return true;
+
+		return m_next and m_next->exists_self(id);
+	}
+
+	bool exists(std::string_view id)
+	{
+		std::shared_lock lock(mMutex);
+
+		return exists_self(std::string{ id });
+	}
+
 	compound *get(std::string id)
 	{
 		std::shared_lock lock(mMutex);

 		compound *result = nullptr;
-		
+
 		for (auto impl = shared_from_this(); impl; impl = impl->m_next)
 		{
 			result = impl->create(id);
@@ -363,7 +406,9 @@ compound *compound_factory_impl::create(const std::string &id)
 	if (m_missing.contains(id))
 		return nullptr;

-	if (auto i = find_if(m_compounds.begin(), m_compounds.end(), [id](compound *c) { return c->id() == id; }); i != m_compounds.end())
+	if (auto i = find_if(m_compounds.begin(), m_compounds.end(), [id](compound *c)
+			{ return c->id() == id; });
+		i != m_compounds.end())
 		return *i;

 	compound *result = nullptr;
@@ -454,7 +499,6 @@ class local_compound_factory_impl : public compound_factory_impl
 	compound *create(const std::string &id) override;

  private:
-
 	compound *construct_compound(const datablock &db, const std::string &id, const std::string &name, const std::string &three_letter_code, const std::string &group);

 	cif::file m_local_file;
@@ -465,7 +509,9 @@ compound *local_compound_factory_impl::create(const std::string &id)
 	if (m_missing.contains(id))
 		return nullptr;

-	if (auto i = find_if(m_compounds.begin(), m_compounds.end(), [id](compound *c) { return c->id() == id; }); i != m_compounds.end())
+	if (auto i = find_if(m_compounds.begin(), m_compounds.end(), [id](compound *c)
+			{ return c->id() == id; });
+		i != m_compounds.end())
 		return *i;

 	compound *result = nullptr;
@@ -480,10 +526,13 @@ compound *local_compound_factory_impl::create(const std::string &id)

 			try
 			{
-				const auto &[id, name, threeLetterCode, group] =
+				const auto &[id2, name, threeLetterCode, group] =
 					chem_comp->front().get<std::string, std::string, std::string, std::string>("id", "name", "three_letter_code", "group");

-				result = construct_compound(db, id, name, threeLetterCode, group);
+				if (id == id2)
+					result = construct_compound(db, id, name, threeLetterCode, group);
+				else
+					throw std::runtime_error("Compound ID's don't match: id 1=" + id + ", id 2=" + id2);
 			}
 			catch (const std::exception &ex)
 			{
@@ -507,12 +556,10 @@ compound *local_compound_factory_impl::construct_compound(const datablock &rdb,

 	float formula_weight = 0;
 	int formal_charge = 0;
-	std::map<std::string,std::size_t> formula_data;
+	std::map<std::string, std::size_t> formula_data;

 	for (std::size_t ord = 1; const auto &[atom_id, type_symbol, type, charge, x, y, z, xi, yi, zi] :
-		rdb["chem_comp_atom"].rows<std::string, std::string, std::string, int,
-			std::optional<float>, std::optional<float>, std::optional<float>,
-			std::optional<float>, std::optional<float>, std::optional<float>>(
+		rdb["chem_comp_atom"].rows<std::string, std::string, std::string, int, std::optional<float>, std::optional<float>, std::optional<float>, std::optional<float>, std::optional<float>, std::optional<float>>(
 			"atom_id", "type_symbol", "type", "charge",
 			"model_Cartn_x", "model_Cartn_y", "model_Cartn_z",
 			"pdbx_model_Cartn_x_ideal", "pdbx_model_Cartn_y_ideal", "pdbx_model_Cartn_z_ideal"))
@@ -522,16 +569,14 @@ compound *local_compound_factory_impl::construct_compound(const datablock &rdb,

 		formula_data[type_symbol] += 1;

-		db["chem_comp_atom"].emplace({
-			{ "comp_id", id },
+		db["chem_comp_atom"].emplace({ { "comp_id", id },
 			{ "atom_id", atom_id },
 			{ "type_symbol", type_symbol },
 			{ "charge", charge },
-			{ "model_Cartn_x",  x.has_value() ? x : xi, 3 },
-			{ "model_Cartn_y",  y.has_value() ? y : yi, 3 },
-			{ "model_Cartn_z",  z.has_value() ? z : zi, 3 },
-			{ "pdbx_ordinal", ord++ }
-		});
+			{ "model_Cartn_x", x.has_value() ? x : xi, 3 },
+			{ "model_Cartn_y", y.has_value() ? y : yi, 3 },
+			{ "model_Cartn_z", z.has_value() ? z : zi, 3 },
+			{ "pdbx_ordinal", ord++ } });

 		formal_charge += charge;
 	}
@@ -548,21 +593,19 @@ compound *local_compound_factory_impl::construct_compound(const datablock &rdb,
 		else if (cif::iequals(type, "triple") or cif::iequals(type, "trip"))
 			value_order = "TRIP";

-		db["chem_comp_bond"].emplace({
-			{ "comp_id", id },
+		db["chem_comp_bond"].emplace({ { "comp_id", id },
 			{ "atom_id_1", atom_id_1 },
 			{ "atom_id_2", atom_id_2 },
 			{ "value_order", value_order },
 			{ "pdbx_aromatic_flag", aromatic },
 			// TODO: fetch stereo_config info from chem_comp_chir
-			{ "pdbx_ordinal", ord++ }
-		});
+			{ "pdbx_ordinal", ord++ } });
 	}

 	db.emplace_back(rdb["pdbx_chem_comp_descriptor"]);

 	std::string formula;
-	for (bool first = true; const auto &[symbol, count]: formula_data)
+	for (bool first = true; const auto &[symbol, count] : formula_data)
 	{
 		if (std::exchange(first, false))
 			formula += ' ';
@@ -581,15 +624,13 @@ compound *local_compound_factory_impl::construct_compound(const datablock &rdb,
 	else
 		type = "NON-POLYMER";

-	db["chem_comp"].emplace({
-		{ "id", id },
+	db["chem_comp"].emplace({ { "id", id },
 		{ "name", name },
 		{ "type", type },
 		{ "formula", formula },
 		{ "pdbx_formal_charge", formal_charge },
 		{ "formula_weight", formula_weight },
-		{ "three_letter_code", three_letter_code }
-	});
+		{ "three_letter_code", three_letter_code } });

 	std::shared_lock lock(mMutex);

@@ -695,6 +736,11 @@ void compound_factory::pop_dictionary()
 		m_impl = m_impl->next();
 }

+bool compound_factory::exists(std::string_view id) const
+{
+	return m_impl and m_impl->exists(id);
+}
+
 const compound *compound_factory::create(std::string_view id)
 {
 	auto result = m_impl ? m_impl->get(std::string{ id }) : nullptr;
@@ -719,7 +765,7 @@ bool compound_factory::is_peptide(std::string_view res_name) const
 	bool result = is_std_peptide(res_name);
 	if (not result and m_impl)
 	{
-		auto compound = const_cast<compound_factory&>(*this).create(res_name);
+		auto compound = const_cast<compound_factory &>(*this).create(res_name);
 		result = compound != nullptr and compound->is_peptide();
 	}
 	return result;
@@ -731,7 +777,7 @@ bool compound_factory::is_base(std::string_view res_name) const
 	bool result = is_std_base(res_name);
 	if (not result and m_impl)
 	{
-		auto compound = const_cast<compound_factory&>(*this).create(res_name);
+		auto compound = const_cast<compound_factory &>(*this).create(res_name);
 		result = compound != nullptr and compound->is_base();
 	}
 	return result;
@@ -757,8 +803,7 @@ bool compound_factory::is_monomer(std::string_view res_name) const

 void compound_factory::report_missing_compound(std::string_view compound_id)
 {
-	static bool s_reported = false;
-	if (std::exchange(s_reported, true) == false)
+	if (std::exchange(m_report_missing, false))
 	{
 		using namespace cif::colour;

@@ -782,7 +827,7 @@ void compound_factory::report_missing_compound(std::string_view compound_id)
 				  << "in /var/cache/libcifpp using the following commands:\n\n"
 				  << "curl -o " << CACHE_DIR << "/components.cif https://files.wwpdb.org/pub/pdb/data/monomers/components.cif\n"
 				  << "curl -o " << CACHE_DIR << "/mmcif_pdbx.dic https://mmcif.wwpdb.org/dictionaries/ascii/mmcif_pdbx_v50.dic\n"
-				  << "curl -o " << CACHE_DIR << "/mmcif_ma.dic https://github.com/ihmwg/ModelCIF/raw/master/dist/mmcif_ma.dic\n\n";
+				  << "curl -o " << CACHE_DIR << "/mmcif_ma.dic https://mmcif.wwpdb.org/dictionaries/ascii/mmcif_ma.dic\n\n";
 #endif

 		if (m_impl)
--- a/src/condition.cpp
+++ b/src/condition.cpp
@@ -24,8 +24,8 @@
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
 */

-#include "cif++/category.hpp"
 #include "cif++/condition.hpp"
+#include "cif++/category.hpp"
 #include "cif++/validate.hpp"

 namespace cif
@@ -61,6 +61,52 @@ bool is_item_type_uchar(const category &cat, std::string_view col)

 namespace detail
 {
+	// 	// index lookup
+	// 	struct index_lookup_condition_impl : public condition_impl
+	// 	{
+	// 		index_lookup_condition_impl(row_initializer &&key_values)
+	// 			: m_key_values(std::move(key_values))
+	// 		{
+	// 		}
+	//
+	// 		condition_impl *prepare(const category &c) override
+	// 		{
+	// 			m_single_hit = c[m_key_values];
+	// 			return this;
+	// 		}
+	//
+	// 		bool test(row_handle r) const override
+	// 		{
+	// 			return m_single_hit == r;
+	// 		}
+	//
+	// 		void str(std::ostream &os) const override
+	// 		{
+	// 			os << "index scan";
+	// 		}
+	//
+	// 		virtual std::optional<row_handle> single() const override
+	// 		{
+	// 			return m_single_hit;
+	// 		}
+	//
+	// 		virtual bool equals(const condition_impl *rhs) const override
+	// 		{
+	// 			if (typeid(*rhs) == typeid(index_lookup_condition_impl))
+	// 			{
+	// 				auto ri = static_cast<const index_lookup_condition_impl *>(rhs);
+	// 				if (m_single_hit or ri->m_single_hit)
+	// 					return m_single_hit == ri->m_single_hit;
+	// 				else
+	// 					// watch out, both m_item_ix might be the same while item_names might be diffent (in case they both do not exist in the category)
+	// 					return m_key_values == ri->m_key_values;
+	// 			}
+	// 			return this == rhs;
+	// 		}
+	//
+	// 		row_initializer m_key_values;
+	// 		row_handle m_single_hit;
+	// 	};

 	condition_impl *key_equals_condition_impl::prepare(const category &c)
 	{
@@ -85,7 +131,8 @@ namespace detail
 			c.key_item_indices().contains(m_item_ix) and
 			c.key_item_indices().size() == 1)
 		{
-			m_single_hit = c[{ { m_item_name, m_value } }];
+			item v(m_item_name, m_value);
+			m_single_hit = c[{ { m_item_name, std::string{ v.value() }, false } }];
 		}

 		return this;
@@ -99,7 +146,8 @@ namespace detail
 		{
 			auto &cs = (*s)->m_sub;

-			if (find_if(cs.begin(), cs.end(), [c](const condition_impl *i) { return i->equals(c); }) == cs.end())
+			if (find_if(cs.begin(), cs.end(), [c](const condition_impl *i)
+					{ return i->equals(c); }) == cs.end())
 			{
 				result = false;
 				break;
@@ -119,7 +167,8 @@ namespace detail
 		for (size_t fc_i = 0; fc_i < fc.size();)
 		{
 			auto c = fc[fc_i];
-			if (not found_in_range(c, subs.begin() + 1, subs.end())) {
+			if (not found_in_range(c, subs.begin() + 1, subs.end()))
+			{
 				++fc_i;
 				continue;
 			}
@@ -137,11 +186,12 @@ namespace detail
 				for (size_t ssub_i = 0; ssub_i < ssub.size();)
 				{
 					auto sc = ssub[ssub_i];
-					if (not sc->equals(c)) {
+					if (not sc->equals(c))
+					{
 						++ssub_i;
 						continue;
 					}
-					
+
 					ssub.erase(ssub.begin() + ssub_i);
 					delete sc;
 					break;
@@ -158,6 +208,99 @@ namespace detail
 		return oc;
 	}

+	condition_impl *and_condition_impl::prepare(const category &c)
+	{
+		for (auto &sub : m_sub)
+			sub = sub->prepare(c);
+
+		if (auto cv = c.get_cat_validator(); cv != nullptr)
+		{
+			// See if we can collapse a search part of this and_condition into a single index lookup
+
+			cif::iset keys{ cv->m_keys.begin(), cv->m_keys.end() };
+			category::key_type lookup;
+			std::vector<condition_impl *> subs;
+			std::vector<std::string> may_be_empty;
+
+			for (auto &sub : m_sub)
+			{
+				if (auto s = dynamic_cast<const key_equals_condition_impl *>(sub); s != nullptr)
+				{
+					if (keys.contains(s->m_item_name))
+					{
+						lookup.emplace_back(s->m_item_name, s->m_value);
+						subs.emplace_back(sub);
+					}
+					continue;
+				}
+
+				if (auto s = dynamic_cast<const key_equals_number_condition_impl *>(sub); s != nullptr)
+				{
+					if (keys.contains(s->m_item_name))
+					{
+						item v{ s->m_item_name, s->m_value };
+						lookup.emplace_back(s->m_item_name, std::string{ v.value() } );
+						subs.emplace_back(sub);
+					}
+					continue;
+				}
+
+				if (auto s = dynamic_cast<const key_equals_or_empty_condition_impl *>(sub); s != nullptr)
+				{
+					if (keys.contains(s->m_item_name))
+					{
+						lookup.emplace_back(s->m_item_name, s->m_value, true);
+						subs.emplace_back(sub);
+						may_be_empty.emplace_back(s->m_item_name);
+					}
+					continue;
+				}
+
+				if (auto s = dynamic_cast<const key_equals_number_or_empty_condition_impl *>(sub); s != nullptr)
+				{
+					if (keys.contains(s->m_item_name))
+					{
+						item v{ s->m_item_name, s->m_value };
+						lookup.emplace_back(s->m_item_name, std::string{ v.value() }, true );
+						subs.emplace_back(sub);
+					}
+					continue;
+				}
+			}
+
+			if (lookup.size() == keys.size())
+			{
+				m_single = c[lookup];
+
+				for (auto s : subs)
+					m_sub.erase(std::remove(m_sub.begin(), m_sub.end(), s), m_sub.end());
+			}
+		}
+
+		return this;
+	}
+
+	bool and_condition_impl::test(row_handle r) const
+	{
+		bool result = true;
+
+		if (m_single.has_value() and *m_single != r)
+			result = false;
+		else
+		{
+			for (auto sub : m_sub)
+			{
+				if (sub->test(r))
+					continue;
+
+				result = false;
+				break;
+			}
+		}
+
+		return result;
+	}
+
 	condition_impl *or_condition_impl::prepare(const category &c)
 	{
 		std::vector<and_condition_impl *> and_conditions;
@@ -181,7 +324,7 @@ void condition::prepare(const category &c)
 {
 	if (m_impl)
 		m_impl = m_impl->prepare(c);
-	
+
 	m_prepared = true;
 }

--- a/src/cql/transaction.cpp
+++ b/src/cql/transaction.cpp
--- a/src/datablock.cpp
+++ b/src/datablock.cpp
@@ -25,8 +25,11 @@
 */

 #include "cif++/datablock.hpp"
+
 #include "cif++/validate.hpp"

+#include <exception>
+
 namespace cif
 {

@@ -42,7 +45,16 @@ datablock::datablock(const datablock &db)
 void datablock::load_dictionary()
 {
 	if (auto *audit_conform = get("audit_conform"); audit_conform and not audit_conform->empty())
-		set_validator(&validator_factory::instance().get(*audit_conform));
+	{
+		try
+		{
+			set_validator(&validator_factory::instance().get(*audit_conform));
+		}
+		catch (const std::exception &ex)
+		{
+			std::clog << ex.what() << '\n';
+		}
+	}
 }

 void datablock::set_validator(const validator *v)
@@ -78,17 +90,42 @@ bool datablock::is_valid() const
 	return result;
 }

-bool datablock::is_valid()
+bool datablock::validate_links() const
 {
-	if (m_validator == nullptr)
-		throw std::runtime_error("Validator not specified for datablock data_" + name());
-
 	bool result = true;
+
 	for (auto &cat : *this)
-		result = cat.is_valid() and result;
+		const_cast<category &>(cat).update_links(*this);
+
+	for (auto &cat : *this)
+		result = cat.validate_links() and result;
+
+	return result;
+}
+
+bool datablock::strip()
+{
+	bool result = true;
+
+	// remove all categories that have no validator
+	erase(std::remove_if(begin(), end(), [](category &c)
+			  {
+		bool result = false;
+		if (c.get_cat_validator() == nullptr)
+		{
+			if (cif::VERBOSE > 0)
+				std::clog << "Dropping category " << c.name() << '\n';
+			result = true;
+		}
+		return result; }),
+		end());
+
+	// then strip the remaining categories
+	for (auto &cat : *this)
+		cat.strip();

 	// Add or remove the audit_conform block here.
-	if (result)
+	if (is_valid())
 	{
 		// If the dictionary declares an audit_conform category, put it in,
 		// but only if it does not exist already!
@@ -101,22 +138,7 @@ bool datablock::is_valid()
 		}
 	}
 	else
-		erase(std::find_if(begin(), end(), [](category &cat)
-				  { return cat.name() == "audit_conform"; }),
-			end());
-
-	return result;
-}
-
-bool datablock::validate_links() const
-{
-	bool result = true;
-
-	for (auto &cat : *this)
-		const_cast<category &>(cat).update_links(*this);
-
-	for (auto &cat : *this)
-		result = cat.validate_links() and result;
+		result = false;

 	return result;
 }
--- a/src/dictionary_parser.cpp
+++ b/src/dictionary_parser.cpp
@@ -28,6 +28,9 @@
 #include "cif++/dictionary_parser.hpp"
 #include "cif++/file.hpp"
 #include "cif++/parser.hpp"
+#include <exception>
+#include <iomanip>
+#include <stdexcept>

 namespace cif
 {
@@ -46,7 +49,7 @@ class dictionary_parser : public parser
 	void load_dictionary()
 	{
 		std::unique_ptr<datablock> dict;
-		auto savedDatablock = m_datablock;
+		auto savedDatablock = std::exchange(m_datablock, nullptr);

 		try
 		{
@@ -75,6 +78,9 @@ class dictionary_parser : public parser
 			error(ex.what());
 		}

+		if (m_datablock == nullptr)
+			throw std::runtime_error("Dictionary file is empty?");
+
 		// store all validators
 		for (auto &ic : mCategoryValidators)
 			m_validator.add_category_validator(std::move(ic));
--- a/src/file.cpp
+++ b/src/file.cpp
@@ -25,6 +25,7 @@
 */

 #include "cif++/file.hpp"
+#include "cif++/condition.hpp"
 #include "cif++/gzio.hpp"

 namespace cif
@@ -46,8 +47,16 @@ bool file::is_valid()
 {
 	bool result = not empty();

-	for (auto &d : *this)
-		result = d.is_valid() and result;
+	for (bool first = true; auto &d : *this)
+	{
+		if (first)
+		{
+			result = d.is_valid() and result;
+			first = false;
+		}
+		else if (d.get_validator() != nullptr)
+			result = d.is_valid() and result;
+	}

 	if (result)
 		result = validate_links();
--- a/src/model.cpp
+++ b/src/model.cpp
@@ -24,13 +24,17 @@
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
 */

+#include "cif++/model.hpp"
 #include "cif++.hpp"
+#include "cif++/point.hpp"

 #include <filesystem>
 #include <fstream>
+#include <initializer_list>
 #include <iomanip>
 #include <numeric>
 #include <stack>
+#include <stdexcept>

 namespace fs = std::filesystem;

@@ -47,15 +51,10 @@ void atom::atom_impl::moveTo(const point &p)

 	auto r = row();

-#if __cpp_lib_format
-	r.assign("Cartn_x", std::format("{:.3f}", p.m_x), false, false);
-	r.assign("Cartn_y", std::format("{:.3f}", p.m_y), false, false);
-	r.assign("Cartn_z", std::format("{:.3f}", p.m_z), false, false);
-#else
-	r.assign("Cartn_x", cif::format("%.3f", p.m_x).str(), false, false);
-	r.assign("Cartn_y", cif::format("%.3f", p.m_y).str(), false, false);
-	r.assign("Cartn_z", cif::format("%.3f", p.m_z).str(), false, false);
-#endif
+	r.assign("Cartn_x", cif::format("{:.3f}", p.m_x), false, false);
+	r.assign("Cartn_y", cif::format("{:.3f}", p.m_y), false, false);
+	r.assign("Cartn_z", cif::format("{:.3f}", p.m_z), false, false);
+
 	m_location = p;
 }

@@ -102,18 +101,18 @@ void atom::atom_impl::set_property(const std::string_view name, const std::strin
 	r.assign(name, value, true, true);
 }

-// int atom::atom_impl::compare(const atom_impl &b) const
-// {
-// 	int d = m_asym_id.compare(b.m_asym_id);
-// 	if (d == 0)
-// 		d = m_seq_id - b.m_seq_id;
-// 	if (d == 0)
-// 		d = m_auth_seq_id.compare(b.m_auth_seq_id);
-// 	if (d == 0)
-// 		d = mAtom_id.compare(b.mAtom_id);
+int atom::atom_impl::compare(const atom_impl &b) const
+{
+	int d = get_property("label_asym_id").compare(b.get_property("label_asym_id"));
+	if (d == 0)
+		d = get_property_int("label_seq_id") - b.get_property_int("label_seq_id");
+	if (d == 0)
+		d = get_property_int("auth_seq_id") - b.get_property_int("auth_seq_id");
+	if (d == 0)
+		d = get_property("label_atom_id").compare(b.get_property("label_atom_id"));

-// 	return d;
-// }
+	return d;
+}

 // bool atom::atom_impl::getAnisoU(float anisou[6]) const
 // {
@@ -149,145 +148,6 @@ int atom::atom_impl::get_charge() const
 	return formalCharge.value_or(0);
 }

-// const Compound *atom::atom_impl::compound() const
-// {
-// 	if (mCompound == nullptr)
-// 	{
-// 		std::string compID = get_property("label_comp_id");
-
-// 		mCompound = compound_factory::instance().create(compID);
-// 	}
-
-// 	return mCompound;
-// }
-
-// const std::string atom::atom_impl::get_property(const std::string_view name) const
-// {
-// 	for (auto &&[item_name, ref] : mCachedRefs)
-// 	{
-// 		if (item_name == name)
-// 			return ref.as<std::string>();
-// 	}
-
-// 	mCachedRefs.emplace_back(name, const_cast<Row &>(mRow)[name]);
-// 	return std::get<1>(mCachedRefs.back()).as<std::string>();
-// }
-
-// void atom::atom_impl::set_property(const std::string_view name, const std::string &value)
-// {
-// 	for (auto &&[item_name, ref] : mCachedRefs)
-// 	{
-// 		if (item_name != name)
-// 			continue;
-
-// 		ref = value;
-// 		return;
-// 	}
-
-// 	mCachedRefs.emplace_back(name, mRow[name]);
-// 	std::get<1>(mCachedRefs.back()) = value;
-// }
-
-// const Row atom::getRowAniso() const
-// {
-// 	auto &db = m_impl->m_db;
-// 	auto cat = db.get("atom_site_anisotrop");
-// 	if (not cat)
-// 		return {};
-// 	else
-// 		return cat->find1(key("id") == m_impl->m_id);
-// }
-
-// float atom::uIso() const
-// {
-// 	float result;
-
-// 	if (not get_property<std::string>("U_iso_or_equiv").empty())
-// 		result = get_property<float>("U_iso_or_equiv");
-// 	else if (not get_property<std::string>("B_iso_or_equiv").empty())
-// 		result = get_property<float>("B_iso_or_equiv") / static_cast<float>(8 * kPI * kPI);
-// 	else
-// 		throw std::runtime_error("Missing B_iso or U_iso");
-
-// 	return result;
-// }
-
-// const Compound &atom::compound() const
-// {
-// 	auto result = impl().compound();
-
-// 	if (result == nullptr)
-// 	{
-// 		if (VERBOSE > 0)
-// 			std::cerr << "Compound not found: '" << get_property<std::string>("label_comp_id") << '\'' << '\n';
-
-// 		throw std::runtime_error("no compound");
-// 	}
-
-// 	return *result;
-// }
-
-// std::string atom::labelEntityID() const
-// {
-// 	return get_property<std::string>("label_entity_id");
-// }
-
-// std::string atom::authAtom_id() const
-// {
-// 	return get_property<std::string>("auth_atom_id");
-// }
-
-// std::string atom::authCompID() const
-// {
-// 	return get_property<std::string>("auth_comp_id");
-// }
-
-// std::string atom::get_auth_asym_id() const
-// {
-// 	return get_property<std::string>("auth_asym_id");
-// }
-
-// std::string atom::get_pdb_ins_code() const
-// {
-// 	return get_property<std::string>("pdbx_PDB_ins_code");
-// }
-
-// std::string atom::pdbxAuthAltID() const
-// {
-// 	return get_property<std::string>("pdbx_auth_alt_id");
-// }
-
-// void atom::translate(point t)
-// {
-// 	auto loc = location();
-// 	loc += t;
-// 	location(loc);
-// }
-
-// void atom::rotate(quaternion q)
-// {
-// 	auto loc = location();
-// 	loc.rotate(q);
-// 	location(loc);
-// }
-
-// void atom::translate_and_rotate(point t, quaternion q)
-// {
-// 	auto loc = location();
-// 	loc += t;
-// 	loc.rotate(q);
-// 	location(loc);
-// }
-
-// void atom::translate_rotate_and_translate(point t1, quaternion q, point t2)
-// {
-// 	auto loc = location();
-// 	loc += t1;
-// 	loc.rotate(q);
-// 	loc += t2;
-// 	location(loc);
-// }
-
 std::ostream &operator<<(std::ostream &os, const atom &atom)
 {
 	if (atom.is_water())
@@ -319,8 +179,8 @@ residue::residue(structure &structure, const std::vector<atom> &atoms)
 	m_compound_id = a.get_label_comp_id();
 	m_asym_id = a.get_label_asym_id();
 	m_seq_id = a.get_label_seq_id();
-	m_auth_asym_id = a.get_auth_asym_id();
-	m_auth_seq_id = a.get_auth_seq_id();
+	m_pdb_strand_id = a.get_auth_asym_id();
+	m_pdb_seq_num = a.get_auth_seq_id();
 	m_pdb_ins_code = a.get_pdb_ins_code();

 	for (auto atom : atoms)
@@ -371,11 +231,12 @@ atom residue::create_new_atom(atom_type inType, const std::string &inAtomID, poi
 		{ "label_alt_id", "." },
 		{ "label_comp_id", m_compound_id },
 		{ "label_seq_id", m_seq_id },
-		{ "auth_asym_id", m_auth_asym_id },
+		{ "auth_asym_id", m_pdb_strand_id },
 		{ "auth_atom_id", inAtomID },
 		{ "auth_comp_id", m_compound_id },
-		{ "auth_seq_id", m_auth_seq_id },
+		{ "auth_seq_id", m_pdb_seq_num },
 		{ "occupancy", 1.0f, 2 },
+		{ "B_iso_or_equiv", 20.0f },
 		{ "pdbx_PDB_model_num", m_structure->get_model_nr() },
 	});

@@ -450,6 +311,28 @@ atom residue::get_atom_by_atom_id(const std::string &atom_id) const
 	return result;
 }

+atom residue::get_atom_by_atom_id(const std::string &atomID, const std::string &altID) const
+{
+	if (altID.empty())
+		return get_atom_by_atom_id(atomID);
+
+	atom result;
+
+	for (auto &a : m_atoms)
+	{
+		if (auto a_alt_id = a.get_label_alt_id(); a.get_label_atom_id() == atomID and (a_alt_id.empty() or a_alt_id == altID))
+		{
+			result = a;
+			break;
+		}
+	}
+
+	if (not result and VERBOSE > 1)
+		std::cerr << "atom with atom_id " << atomID << " and alt_id " << altID << " not found in residue " << m_asym_id << ':' << m_seq_id << '\n';
+
+	return result;
+}
+
 // residue is a single entity if the atoms for the asym with m_asym_id is equal
 // to the number of atoms in this residue...  hope this is correct....
 bool residue::is_entity() const
@@ -469,17 +352,7 @@ std::tuple<point, float> residue::center_and_radius() const
 	for (auto &a : m_atoms)
 		pts.push_back(a.get_location());

-	auto center = centroid(pts);
-	float radius = 0;
-
-	for (auto &pt : pts)
-	{
-		float d = static_cast<float>(distance(pt, center));
-		if (radius < d)
-			radius = d;
-	}
-
-	return std::make_tuple(center, radius);
+	return smallest_sphere_around_points(pts);
 }

 bool residue::has_alternate_atoms() const
@@ -518,8 +391,8 @@ std::ostream &operator<<(std::ostream &os, const residue &res)
 {
 	os << res.get_compound_id() << ' ' << res.get_asym_id() << ':' << res.get_seq_id();

-	if (res.get_auth_asym_id() != res.get_asym_id() or res.get_auth_seq_id() != std::to_string(res.get_seq_id()))
-		os << " [" << res.get_auth_asym_id() << ':' << res.get_auth_seq_id() << ']';
+	if (res.get_pdb_strand_id() != res.get_asym_id() or res.get_pdb_seq_num() != std::to_string(res.get_seq_id()))
+		os << " [" << res.get_pdb_strand_id() << ':' << res.get_pdb_seq_num() << ']';

 	return os;
 }
@@ -528,7 +401,7 @@ std::ostream &operator<<(std::ostream &os, const residue &res)
 // monomer

 monomer::monomer(const polymer &polymer, std::size_t index, int seqID, const std::string &authSeqID, const std::string &pdbInsCode, const std::string &compoundID)
-	: residue(*polymer.get_structure(), compoundID, polymer.get_asym_id(), seqID, polymer.get_auth_asym_id(), authSeqID, pdbInsCode)
+	: residue(*polymer.get_structure(), compoundID, polymer.get_asym_id(), seqID, polymer.get_pdb_strand_id(), authSeqID, pdbInsCode)
 	, m_polymer(&polymer)
 	, m_index(index)
 {
@@ -562,6 +435,16 @@ bool monomer::is_last_in_chain() const
 	return m_index + 1 == m_polymer->size();
 }

+const monomer &monomer::prev() const
+{
+	return m_polymer->at(m_index - 1);
+}
+
+const monomer &monomer::next() const
+{
+	return m_polymer->at(m_index + 1);
+}
+
 bool monomer::has_alpha() const
 {
 	return m_index >= 1 and m_index + 2 < m_polymer->size();
@@ -578,7 +461,7 @@ float monomer::phi() const

 	if (m_index > 0)
 	{
-		auto &prev = m_polymer->operator[](m_index - 1);
+		auto &prev = m_polymer->at(m_index - 1);
 		if (prev.m_seq_id + 1 == m_seq_id)
 		{
 			auto a1 = prev.C();
@@ -600,7 +483,7 @@ float monomer::psi() const

 	if (m_index + 1 < m_polymer->size())
 	{
-		auto &next = m_polymer->operator[](m_index + 1);
+		auto &next = m_polymer->at(m_index + 1);
 		if (m_seq_id + 1 == next.m_seq_id)
 		{
 			auto a1 = N();
@@ -624,9 +507,9 @@ float monomer::alpha() const
 	{
 		if (m_index >= 1 and m_index + 2 < m_polymer->size())
 		{
-			auto &prev = m_polymer->operator[](m_index - 1);
-			auto &next = m_polymer->operator[](m_index + 1);
-			auto &nextNext = m_polymer->operator[](m_index + 2);
+			auto &prev = m_polymer->at(m_index - 1);
+			auto &next = m_polymer->at(m_index + 1);
+			auto &nextNext = m_polymer->at(m_index + 2);

 			result = static_cast<float>(dihedral_angle(prev.CAlpha().get_location(), CAlpha().get_location(), next.CAlpha().get_location(), nextNext.CAlpha().get_location()));
 		}
@@ -648,8 +531,8 @@ float monomer::kappa() const
 	{
 		if (m_index >= 2 and m_index + 2 < m_polymer->size())
 		{
-			auto &prevPrev = m_polymer->operator[](m_index - 2);
-			auto &nextNext = m_polymer->operator[](m_index + 2);
+			auto &prevPrev = m_polymer->at(m_index - 2);
+			auto &nextNext = m_polymer->at(m_index + 2);

 			if (prevPrev.m_seq_id + 4 == nextNext.m_seq_id)
 			{
@@ -677,7 +560,7 @@ float monomer::tco() const
 	{
 		if (m_index > 0)
 		{
-			auto &prev = m_polymer->operator[](m_index - 1);
+			auto &prev = m_polymer->at(m_index - 1);
 			if (prev.m_seq_id + 1 == m_seq_id)
 				result = static_cast<float>(cosinus_angle(C().get_location(), O().get_location(), prev.C().get_location(), prev.O().get_location()));
 		}
@@ -699,7 +582,7 @@ float monomer::omega() const
 	try
 	{
 		if (not is_last_in_chain())
-			result = omega(*this, m_polymer->operator[](m_index + 1));
+			result = omega(*this, m_polymer->at(m_index + 1));
 	}
 	catch (const std::exception &ex)
 	{
@@ -795,7 +678,7 @@ bool monomer::is_cis() const

 	if (m_index + 1 < m_polymer->size())
 	{
-		auto &next = m_polymer->operator[](m_index + 1);
+		auto &next = m_polymer->at(m_index + 1);

 		result = monomer::is_cis(*this, next);
 	}
@@ -937,7 +820,7 @@ polymer::polymer(structure &s, const std::string &entityID, const std::string &a
 	: m_structure(const_cast<structure *>(&s))
 	, m_entity_id(entityID)
 	, m_asym_id(asym_id)
-	, m_auth_asym_id(auth_asym_id)
+	, m_pdb_strand_id(auth_asym_id)
 {
 	using namespace cif::literals;

@@ -949,12 +832,8 @@ polymer::polymer(structure &s, const std::string &entityID, const std::string &a
 	for (auto r : poly_seq_scheme.find("asym_id"_key == asym_id))
 	{
 		int seqID;
-		std::optional<int> pdbSeqNum;
-		std::string compoundID, authSeqID, pdbInsCode;
-		cif::tie(seqID, authSeqID, compoundID, pdbInsCode, pdbSeqNum) = r.get("seq_id", "auth_seq_num", "mon_id", "pdb_ins_code", "pdb_seq_num");
-
-		if (authSeqID.empty() and pdbSeqNum.has_value())
-			authSeqID = std::to_string(*pdbSeqNum);
+		std::string compoundID, pdbSeqNum, pdbInsCode;
+		cif::tie(seqID, pdbSeqNum, compoundID, pdbInsCode) = r.get("seq_id", "pdb_seq_num", "mon_id", "pdb_ins_code");

 		std::size_t index = size();

@@ -962,11 +841,11 @@ polymer::polymer(structure &s, const std::string &entityID, const std::string &a
 		if (not ix.count(seqID))
 		{
 			ix[seqID] = index;
-			emplace_back(*this, index, seqID, authSeqID, pdbInsCode, compoundID);
+			emplace_back(*this, index, seqID, pdbSeqNum, pdbInsCode, compoundID);
 		}
 		else if (VERBOSE > 0)
 		{
-			monomer m{ *this, index, seqID, authSeqID, pdbInsCode, compoundID };
+			monomer m{ *this, index, seqID, pdbSeqNum, pdbInsCode, compoundID };
 			std::cerr << "Dropping alternate residue " << m << '\n';
 		}
 	}
@@ -1106,7 +985,7 @@ cif::mm::atom sugar::add_atom(row_initializer atom_info)
 	atom_info.set_value({ "label_alt_id", "." });
 	atom_info.set_value({ "auth_asym_id", m_branch->get_asym_id() });
 	atom_info.set_value({ "auth_comp_id", m_compound_id });
-	atom_info.set_value({ "auth_seq_id", m_auth_seq_id });
+	atom_info.set_value({ "auth_seq_id", m_pdb_seq_num });
 	atom_info.set_value({ "occupancy", 1.0, 2 });
 	atom_info.set_value({ "B_iso_or_equiv", 30.0, 2 });
 	atom_info.set_value({ "pdbx_PDB_model_num", 1 });
@@ -1222,12 +1101,12 @@ sugar &branch::construct_sugar(const std::string &compound_id)
 		{ "mon_id", result.get_compound_id() },

 		{ "pdb_asym_id", result.get_asym_id() },
-		{ "pdb_seq_num", result.num() },
+		{ "pdb_seq_num", result.get_pdb_seq_num() },
 		{ "pdb_mon_id", result.get_compound_id() },

-		{ "auth_asym_id", result.get_auth_asym_id() },
+		{ "auth_asym_id", result.get_pdb_strand_id() },
 		{ "auth_mon_id", result.get_compound_id() },
-		{ "auth_seq_num", result.get_auth_seq_id() },
+		{ "auth_seq_num", result.get_pdb_seq_num() },

 		{ "hetero", "n" } });

@@ -1270,7 +1149,7 @@ std::string branch::name(const sugar &s) const

 	for (auto &sn : *this)
 	{
-		if (not sn.get_link() or sn.get_link().get_auth_seq_id() != s.get_auth_seq_id())
+		if (not sn.get_link() or sn.get_link().get_auth_seq_id() != s.get_pdb_seq_num())
 			continue;

 		auto n = name(sn) + "-(1-" + sn.get_link().get_label_atom_id().substr(1) + ')';
@@ -1297,19 +1176,19 @@ float branch::weight() const
 // --------------------------------------------------------------------
 //	structure

-structure::structure(file &p, std::size_t modelNr, StructureOpenOptions options)
+structure::structure(file &p, std::size_t modelNr, structure_open_options options)
 	: structure(p.front(), modelNr, options)
 {
 }

-structure::structure(datablock &db, std::size_t modelNr, StructureOpenOptions options)
+structure::structure(datablock &db, std::size_t modelNr, structure_open_options options)
 	: m_db(db)
 	, m_model_nr(modelNr)
 {
 	if (db.get_validator() == nullptr)
 		db.load_dictionary();

-	auto &atomCat = db["atom_site"];
+	auto &atom_site = db["atom_site"];

 	load_atoms_for_model(options);

@@ -1317,7 +1196,7 @@ structure::structure(datablock &db, std::size_t modelNr, StructureOpenOptions op
 	if (m_atoms.empty() and m_model_nr == 1)
 	{
 		std::optional<std::size_t> model_nr;
-		cif::tie(model_nr) = atomCat.front().get("pdbx_PDB_model_num");
+		cif::tie(model_nr) = atom_site.front().get("pdbx_PDB_model_num");
 		if (model_nr and *model_nr != m_model_nr)
 		{
 			if (VERBOSE > 0)
@@ -1336,42 +1215,131 @@ structure::structure(datablock &db, std::size_t modelNr, StructureOpenOptions op
 		load_data();
 }

-void structure::load_atoms_for_model(StructureOpenOptions options)
+void structure::load_atoms_for_model(structure_open_options options)
 {
 	using namespace literals;

-	auto &atomCat = m_db["atom_site"];
+	auto &atom_site = m_db["atom_site"];

 	condition c = "pdbx_PDB_model_num"_key == null or "pdbx_PDB_model_num"_key == m_model_nr;
-	if (options bitand StructureOpenOptions::SkipHydrogen)
-		c = std::move(c) and ("type_symbol"_key != "H" and "type_symbol"_key != "D");

-	for (auto id : atomCat.find<std::string>(std::move(c), "id"))
-		emplace_atom(std::make_shared<atom::atom_impl>(m_db, id));
+	if (options.skip_hydrogen)
+		c = std::move(c) and (cif::key("type_symbol") != "H" and cif::key("type_symbol") != "D");
+
+	if (options.skip_water)
+		c = std::move(c) and (cif::key("auth_comp_id") != "HOH" and cif::key("auth_comp_id") != "H20" and cif::key("auth_comp_id") != "WAT");
+
+	if (options.skip_hetatom)
+	{
+		if (options.skip_water)
+			c = std::move(c) and cif::key("group_PDB") != "HETATM";
+		else
+			c = std::move(c) and (cif::key("group_PDB") != "HETATM" or (cif::key("auth_comp_id") == "HOH" or cif::key("auth_comp_id") == "H20" or cif::key("auth_comp_id") == "WAT"));
+	}
+
+	if (options.min_b_factor.has_value())
+		c = std::move(c) and cif::key("B_iso_or_equiv") >= *options.min_b_factor;
+
+	if (options.max_b_factor.has_value())
+		c = std::move(c) and cif::key("B_iso_or_equiv") <= *options.max_b_factor;
+
+	if (not options.asyms.empty())
+	{
+		condition tmp_c;
+		for (auto asym_id : options.asyms)
+			tmp_c = std::move(tmp_c) or cif::key("label_asym_id") == asym_id;
+		c = std::move(c) and std::move(tmp_c);
+	}
+
+	if (options.occupancy_mode == occupancy_policy::ALL)
+	{
+		for (auto id : atom_site.find<std::string>(std::move(c), "id"))
+			emplace_atom(std::make_shared<atom::atom_impl>(m_db, id));
+	}
+	else if (options.occupancy_mode == occupancy_policy::UNOCCUPIED)
+	{
+		for (auto id : atom_site.find<std::string>(std::move(c), "id"))
+		{
+			auto a = std::make_shared<atom::atom_impl>(m_db, id);
+			if (a->get_property_float("occupancy") > 0)
+				continue;
+			emplace_atom(a);
+		}
+	}
+	else
+	{
+		std::vector<cif::mm::atom> atoms;
+		std::map<std::tuple<std::string, int>, std::map<std::string, float>> alts;
+
+		for (auto id : atom_site.find<std::string>(std::move(c), "id"))
+		{
+			auto a = atoms.emplace_back(std::make_shared<atom::atom_impl>(m_db, id));
+
+			if (a.is_alternate())
+			{
+				auto key = std::make_tuple(a.get_label_asym_id(), a.get_label_seq_id());
+				auto alt_id = a.get_label_alt_id();
+
+				if (auto i = alts.find(key); i != alts.end())
+					i->second[alt_id] += a.get_occupancy();
+				else
+					alts[key][alt_id] = a.get_occupancy();
+			}
+		}
+
+		for (auto &&[key, value] : alts)
+		{
+			// const auto &[asym_id, seq_id] = key;
+
+			// select highest occupancy for this residue's alternates
+			std::string alt_id;
+			float occupancy = options.occupancy_mode == occupancy_policy::MAX ? 0.f : std::numeric_limits<float>::max();
+			for (const auto &[alt_key, alt_value] : value)
+			{
+				if (options.occupancy_mode == occupancy_policy::MAX)
+				{
+					if (occupancy < alt_value)
+					{
+						alt_id = alt_key;
+						occupancy = alt_value;
+					}
+				}
+				else
+				{
+					if (occupancy > alt_value)
+					{
+						alt_id = alt_key;
+						occupancy = alt_value;
+					}
+				}
+			}
+
+			value.clear();
+			value.emplace(alt_id, occupancy);
+		}
+
+		for (auto a : atoms)
+		{
+			if (a.is_alternate())
+			{
+				auto key = std::make_tuple(a.get_label_asym_id(), a.get_label_seq_id());
+
+				if (alts[key].contains(a.get_label_alt_id()))
+					emplace_atom(a);
+			}
+			else
+				emplace_atom(a);
+		}
+	}
 }

-// structure::structure(const structure &s)
-// 	: m_db(s.m_db)
-// 	, m_model_nr(s.m_model_nr)
-// {
-// 	m_atoms.reserve(s.m_atoms.size());
-// 	for (auto &atom : s.m_atoms)
-// 		emplace_atom(atom.clone());
-
-// 	load_data();
-// }
-
-// structure::~structure()
-// {
-// }
-
 void structure::load_data()
 {
 	auto &polySeqScheme = m_db["pdbx_poly_seq_scheme"];

 	for (const auto &[asym_id, auth_asym_id, entityID] : polySeqScheme.rows<std::string, std::string, std::string>("asym_id", "pdb_strand_id", "entity_id"))
 	{
-		if (m_polymers.empty() or m_polymers.back().get_asym_id() != asym_id or m_polymers.back().get_entity_id() != entityID)
+		if (m_polymers.empty() or m_polymers.back().get_asym_id() != asym_id)
 			m_polymers.emplace_back(*this, entityID, asym_id, auth_asym_id);
 	}

@@ -1397,18 +1365,18 @@ void structure::load_data()
 	for (auto &poly : m_polymers)
 	{
 		for (auto &res : poly)
-			resMap[{ res.get_asym_id(), res.get_seq_id(), res.get_auth_seq_id() }] = &res;
+			resMap[{ res.get_asym_id(), res.get_seq_id(), res.get_pdb_seq_num() }] = &res;
 	}

 	for (auto &res : m_non_polymers)
-		resMap[{ res.get_asym_id(), res.get_seq_id(), res.get_auth_seq_id() }] = &res;
+		resMap[{ res.get_asym_id(), res.get_seq_id(), res.get_pdb_seq_num() }] = &res;

 	std::set<std::string> sugars;
 	for (auto &branch : m_branches)
 	{
 		for (auto &sugar : branch)
 		{
-			resMap[{ sugar.get_asym_id(), sugar.get_seq_id(), sugar.get_auth_seq_id() }] = &sugar;
+			resMap[{ sugar.get_asym_id(), sugar.get_seq_id(), sugar.get_pdb_seq_num() }] = &sugar;
 			sugars.insert(sugar.get_compound_id());
 		}
 	}
@@ -1483,30 +1451,6 @@ EntityType structure::get_entity_type_for_asym_id(const std::string asym_id) con
 	return get_entity_type_for_entity_id(entityID);
 }

-// std::vector<atom> structure::waters() const
-// {
-// 	using namespace literals;
-
-// 	std::vector<atom> result;
-
-// 	auto &db = datablock();
-
-// 	// Get the entity id for water. Watch out, structure may not have water at all
-// 	auto &entityCat = db["entity"];
-// 	for (const auto &[waterEntityID] : entityCat.find<std::string>("type"_key == "water", "id"))
-// 	{
-// 		for (auto &a : m_atoms)
-// 		{
-// 			if (a.get_property("label_entity_id") == waterEntityID)
-// 				result.push_back(a);
-// 		}
-
-// 		break;
-// 	}
-
-// 	return result;
-// }
-
 bool structure::has_atom_id(const std::string &id) const
 {
 	assert(m_atoms.size() == m_atom_index.size());
@@ -1655,7 +1599,7 @@ residue &structure::get_residue(const std::string &asym_id, int seqID, const std
 	{
 		for (auto &res : m_non_polymers)
 		{
-			if (res.get_asym_id() == asym_id and (authSeqID.empty() or res.get_auth_seq_id() == authSeqID))
+			if (res.get_asym_id() == asym_id and (authSeqID.empty() or res.get_pdb_seq_num() == authSeqID))
 				return res;
 		}
 	}
@@ -1679,7 +1623,7 @@ residue &structure::get_residue(const std::string &asym_id, int seqID, const std

 		for (auto &sugar : branch)
 		{
-			if (sugar.get_asym_id() == asym_id and sugar.get_auth_seq_id() == authSeqID)
+			if (sugar.get_asym_id() == asym_id and sugar.get_pdb_seq_num() == authSeqID)
 				return sugar;
 		}
 	}
@@ -1701,7 +1645,7 @@ residue &structure::get_residue(const std::string &asym_id, const std::string &c
 	{
 		for (auto &res : m_non_polymers)
 		{
-			if (res.get_asym_id() == asym_id and res.get_auth_seq_id() == authSeqID and res.get_compound_id() == compID)
+			if (res.get_asym_id() == asym_id and res.get_pdb_seq_num() == authSeqID and res.get_compound_id() == compID)
 				return res;
 		}
 	}
@@ -1725,7 +1669,7 @@ residue &structure::get_residue(const std::string &asym_id, const std::string &c

 		for (auto &sugar : branch)
 		{
-			if (sugar.get_asym_id() == asym_id and sugar.get_auth_seq_id() == authSeqID and sugar.get_compound_id() == compID)
+			if (sugar.get_asym_id() == asym_id and sugar.get_pdb_seq_num() == authSeqID and sugar.get_compound_id() == compID)
 				return sugar;
 		}
 	}
@@ -1946,13 +1890,12 @@ void structure::swap_atoms(atom a1, atom a2)
 		auto r1 = atomSites.find1(key("id") == a1.id());
 		auto r2 = atomSites.find1(key("id") == a2.id());

-		auto l1 = r1["label_atom_id"];
-		auto l2 = r2["label_atom_id"];
-		l1.swap(l2);
-
-		auto l3 = r1["auth_atom_id"];
-		auto l4 = r2["auth_atom_id"];
-		l3.swap(l4);
+		for (std::string fld : std::initializer_list<std::string>{ "label_atom_id", "auth_atom_id", "type_symbol" })
+		{
+			auto l1 = r1[fld];
+			auto l2 = r2[fld];
+			l1.swap(l2);
+		}
 	}
 	catch (const std::exception &ex)
 	{
@@ -2075,7 +2018,7 @@ void structure::remove_residue(const std::string &asym_id, int seq_id, const std
 	{
 		for (auto &res : m_non_polymers)
 		{
-			if (res.get_asym_id() == asym_id and (auth_seq_id.empty() or res.get_auth_seq_id() == auth_seq_id))
+			if (res.get_asym_id() == asym_id and (auth_seq_id.empty() or res.get_pdb_seq_num() == auth_seq_id))
 			{
 				remove_residue(res);
 				return;
@@ -2105,7 +2048,7 @@ void structure::remove_residue(const std::string &asym_id, int seq_id, const std

 		for (auto &sugar : branch)
 		{
-			if (sugar.get_asym_id() == asym_id and sugar.get_auth_seq_id() == auth_seq_id)
+			if (sugar.get_asym_id() == asym_id and sugar.get_pdb_seq_num() == auth_seq_id)
 			{
 				remove_residue(sugar);
 				return;
@@ -2238,7 +2181,7 @@ void structure::remove_sugar(sugar &s)
 				// TODO: need fix, collect from nag_atoms?
 				{ "auth_asym_id", asym_id },
 				{ "auth_mon_id", sugar.get_compound_id() },
-				{ "auth_seq_num", sugar.get_auth_seq_id() },
+				{ "auth_seq_num", sugar.get_pdb_seq_num() },

 				{ "hetero", "n" } });
 		}
@@ -2324,8 +2267,8 @@ std::string structure::create_non_poly(const std::string &entity_id, const std::
 		{ "entity_id", entity_id },
 		{ "mon_id", comp_id },
 		{ "ndb_seq_num", ndb_nr },
-		{ "pdb_seq_num", res.get_auth_seq_id() },
-		{ "auth_seq_num", res.get_auth_seq_id() },
+		{ "pdb_seq_num", res.get_pdb_seq_num() },
+		{ "auth_seq_num", res.get_pdb_seq_num() },
 		{ "pdb_mon_id", comp_id },
 		{ "auth_mon_id", comp_id },
 		{ "pdb_strand_id", asym_id },
@@ -2385,8 +2328,8 @@ std::string structure::create_non_poly(const std::string &entity_id, std::vector
 		{ "entity_id", entity_id },
 		{ "mon_id", comp_id },
 		{ "ndb_seq_num", ndb_nr },
-		{ "pdb_seq_num", res.get_auth_seq_id() },
-		{ "auth_seq_num", res.get_auth_seq_id() },
+		{ "pdb_seq_num", res.get_pdb_seq_num() },
+		{ "auth_seq_num", res.get_pdb_seq_num() },
 		{ "pdb_mon_id", comp_id },
 		{ "auth_mon_id", comp_id },
 		{ "pdb_strand_id", asym_id },
@@ -2396,6 +2339,36 @@ std::string structure::create_non_poly(const std::string &entity_id, std::vector
 	return asym_id;
 }

+std::string structure::create_non_poly(const std::string &compound_id, bool skip_hydrogen)
+{
+	auto compound = cif::compound_factory::instance().create(compound_id);
+	if (compound == nullptr)
+		throw std::runtime_error(std::format("{} is not a known compound", compound_id));
+	
+	std::vector<cif::row_initializer> atoms;
+	for (auto a : compound->atoms())
+	{
+		// We skip H-atoms, as fitting without H-atoms works better and we avoid conflicts in protonation states between CCD and MONLIB
+		if (skip_hydrogen and cif::atom_type_traits(a.type_symbol).symbol() == "H")
+			continue;
+
+		auto ax = a.get_location().get_x();
+		auto ay = a.get_location().get_y();
+		auto az = a.get_location().get_z();
+
+		atoms.emplace_back(cif::row_initializer{
+			{ "type_symbol", cif::atom_type_traits(a.type_symbol).symbol() },
+			{ "label_atom_id", a.id },
+			{ "auth_atom_id", a.id },
+			{ "Cartn_x", ax },
+			{ "Cartn_y", ay },
+			{ "Cartn_z", az },
+			{ "B_iso_or_equiv", 30.00 } });
+	}
+
+	return create_non_poly(create_non_poly_entity(compound_id), atoms);
+}
+
 void structure::create_water(row_initializer atom)
 {
 	using namespace literals;
@@ -2460,6 +2433,61 @@ void structure::create_water(row_initializer atom)
 	});
 }

+std::string structure::create_link(atom a1, atom a2, const std::string &link_type, const std::string &role)
+{
+	using namespace literals;
+
+	auto &struct_conn = m_db["struct_conn"];
+	auto &struct_conn_type = m_db["struct_conn_type"];
+
+	// This will validate link_type :-)
+	if (not struct_conn_type.contains("id"_key == link_type))
+		struct_conn_type.emplace({ { "id", link_type } });
+
+	std::string link_id = struct_conn.get_unique_id(link_type + '_');
+
+	item label_seq_id_1("ptnr1_label_seq_id");
+	if (int nr = a1.get_label_seq_id(); nr != 0)
+		label_seq_id_1.value(std::to_string(nr));
+
+	item label_seq_id_2("ptnr2_label_seq_id");
+	if (int nr = a2.get_label_seq_id(); nr != 0)
+		label_seq_id_2.value(std::to_string(nr));
+
+	struct_conn.emplace(
+		{ //
+			{ "id", link_id },
+			{ "conn_type_id", link_type },
+			{ "pdbx_leaving_atom_flag", "one" },
+
+			{ "ptnr1_label_asym_id", a1.get_label_asym_id() },
+			{ "ptnr1_label_comp_id", a1.get_label_comp_id() },
+			label_seq_id_1,
+			{ "ptnr1_label_atom_id", a1.get_label_atom_id() },
+			{ "pdbx_ptnr1_label_alt_id", a1.get_label_alt_id() },
+			{ "pdbx_ptnr1_PDB_ins_code", a1.get_pdb_ins_code() },
+			{ "ptnr1_auth_asym_id", a1.get_auth_asym_id() },
+			{ "ptnr1_auth_comp_id", a1.get_auth_comp_id() },
+			{ "ptnr1_auth_seq_id", a1.get_auth_seq_id() },
+			{ "ptnr1_symmetry", a1.symmetry() },
+
+			{ "ptnr2_label_asym_id", a2.get_label_asym_id() },
+			{ "ptnr2_label_comp_id", a2.get_label_comp_id() },
+			label_seq_id_2,
+			{ "ptnr2_label_atom_id", a2.get_label_atom_id() },
+			{ "pdbx_ptnr2_label_alt_id", a2.get_label_alt_id() },
+			{ "pdbx_ptnr2_PDB_ins_code", a2.get_pdb_ins_code() },
+			{ "ptnr2_auth_asym_id", a2.get_auth_asym_id() },
+			{ "ptnr2_auth_comp_id", a2.get_auth_comp_id() },
+			{ "ptnr2_auth_seq_id", a2.get_auth_seq_id() },
+			{ "ptnr2_symmetry", a2.symmetry() },
+
+			{ "pdbx_dist_value", distance(a1.get_location(), a2.get_location()), 3 },
+			{ "pdbx_role", role } });
+
+	return link_id;
+}
+
 branch &structure::create_branch()
 {
 	auto &entity = m_db["entity"];
@@ -2700,11 +2728,11 @@ std::string structure::create_entity_for_branch(branch &branch)

 			pdbx_entity_branch_link.emplace({ { "link_id", pdbx_entity_branch_link.get_unique_id("") },
 				{ "entity_id", entityID },
-				{ "entity_branch_list_num_1", s1.get_auth_seq_id() },
+				{ "entity_branch_list_num_1", s1.get_pdb_seq_num() },
 				{ "comp_id_1", s1.get_compound_id() },
 				{ "atom_id_1", l1.get_label_atom_id() },
 				{ "leaving_atom_id_1", "O1" },
-				{ "entity_branch_list_num_2", s2.get_auth_seq_id() },
+				{ "entity_branch_list_num_2", s2.get_pdb_seq_num() },
 				{ "comp_id_2", s2.get_compound_id() },
 				{ "atom_id_2", l2.get_label_atom_id() },
 				{ "leaving_atom_id_2", "H" + l2.get_label_atom_id() },
@@ -2717,9 +2745,12 @@ std::string structure::create_entity_for_branch(branch &branch)

 void structure::cleanup_empty_categories()
 {
+
 	using namespace literals;

 	auto &atomSite = m_db["atom_site"];
+	auto &pdbxPolySeqScheme = m_db["pdbx_poly_seq_scheme"];
+	auto &entityPolySeq = m_db["entity_poly_seq"];

 	// Remove chem_comp's for which there are no atoms at all
 	auto &chem_comp = m_db["chem_comp"];
@@ -2728,8 +2759,12 @@ void structure::cleanup_empty_categories()
 	for (auto chemComp : chem_comp)
 	{
 		std::string compID = chemComp["id"].as<std::string>();
-		if (atomSite.contains("label_comp_id"_key == compID or "auth_comp_id"_key == compID))
+		if (atomSite.contains("label_comp_id"_key == compID or "auth_comp_id"_key == compID) or
+			pdbxPolySeqScheme.contains("mon_id"_key == compID or "auth_mon_id"_key == compID or "pdb_mon_id"_key == compID) or
+			entityPolySeq.contains("mon_id"_key == compID))
+		{
 			continue;
+		}

 		obsoleteChemComps.push_back(chemComp);
 	}
@@ -2886,8 +2921,8 @@ static int compare_numbers(std::string_view a, std::string_view b)

 	std::from_chars_result ra, rb;

-	ra = selected_charconv<double>::from_chars(a.data(), a.data() + a.length(), da);
-	rb = selected_charconv<double>::from_chars(b.data(), b.data() + b.length(), db);
+	ra = from_chars(a.data(), a.data() + a.length(), da);
+	rb = from_chars(b.data(), b.data() + b.length(), db);

 	if (not(bool) ra.ec and not(bool) rb.ec)
 	{
@@ -2908,6 +2943,14 @@ static int compare_numbers(std::string_view a, std::string_view b)
 	return result;
 }

+int compare_cif_id(const std::string &a, const std::string &b)
+{
+	int d = static_cast<int>(a.length() - b.length());
+	if (d == 0)
+		d = a.compare(b);
+	return d;
+}
+
 void structure::reorder_atoms()
 {
 	auto &atom_site = m_db["atom_site"];
@@ -2919,7 +2962,7 @@ void structure::reorder_atoms()
 			// First by model number
 			d = a.get<int>("pdbx_PDB_model_num") - b.get<int>("pdbx_PDB_model_num");
 			if (d == 0)
-				d = a.get<std::string>("label_asym_id").compare(b.get<std::string>("label_asym_id"));
+				d = compare_cif_id(a.get<std::string>("label_asym_id"), b.get<std::string>("label_asym_id"));
 			if (d == 0)
 			{
 				auto na = a.get<std::optional<int>>("label_seq_id");
--- a/src/pdb/cif2pdb.cpp
+++ b/src/pdb/cif2pdb.cpp
@@ -33,7 +33,6 @@
 #include <regex>
 #include <set>

-
 namespace cif::pdb
 {

@@ -58,9 +57,9 @@ std::string cif2pdbDate(const std::string &d)
 		int month = std::stoi(m[2].str());

 		if (m[3].matched)
-			result = cif::format("%02.2d-%3.3s-%02.2d", stoi(m[3].str()), kMonths[month - 1], (year % 100)).str();
+			result = cif::format("{:02}-{:3.3}-{:02}", stoi(m[3].str()), kMonths[month - 1], (year % 100));
 		else
-			result = cif::format("%3.3s-%02.2d", kMonths[month - 1], (year % 100)).str();
+			result = cif::format("{:3.3}-{:02}", kMonths[month - 1], (year % 100));
 	}

 	return result;
@@ -258,16 +257,14 @@ std::size_t WriteCitation(std::ostream &pdbFile, const datablock &db, row_handle
 	{
 		to_upper(pubname);

-		const std::string kRefHeader = s1 + "REF %2.2s %-28.28s  %2.2s%4.4s %5.5s %4.4s";
-		pdbFile << cif::format(kRefHeader, "" /* continuation */, pubname, (volume.empty() ? "" : "V."), volume, pageFirst, year)
+		pdbFile << s1 << cif::format("REF {:2.2s} {:<28.28s}  {:2.2s}{:>4.4s} {:>5.5s} {:4.4s}", "" /* continuation */, pubname, (volume.empty() ? "" : "V."), volume, pageFirst, year)
 				<< '\n';
 		++result;
 	}

 	if (not issn.empty())
 	{
-		const std::string kRefHeader = s1 + "REFN                   ISSN %-25.25s";
-		pdbFile << cif::format(kRefHeader, issn) << '\n';
+		pdbFile << s1 << cif::format("REFN                   ISSN {:<25.25s}", issn) << '\n';
 		++result;
 	}

@@ -276,27 +273,25 @@ std::size_t WriteCitation(std::ostream &pdbFile, const datablock &db, row_handle
 	////    0         1         2         3         4         5         6         7         8
 	////    HEADER    xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxDDDDDDDDD   IIII
 	// const char kRefHeader[] =
-	//          "REMARK   1  REFN    %4.4s %-6.6s  %2.2s %-25.25s";
+	//          "REMARK   1  REFN    {:4.4s} {:<6.6s}  {:2.2s} {:<25.25s}";
 	//
 	//			pdbFile << (boost::cif::format(kRefHeader)
 	//						% (astm.empty() ? "" : "ASTN")
 	//						% astm
 	//						% country
-	//						% issn).str()
+	//						% issn)
 	//					<< '\n';
 	//		}

 	if (not pmid.empty())
 	{
-		const std::string kPMID = s1 + "PMID   %-60.60s ";
-		pdbFile << cif::format(kPMID, pmid) << '\n';
+		pdbFile << s1 << cif::format("PMID   {:<60.60s} ", pmid) << '\n';
 		++result;
 	}

 	if (not doi.empty())
 	{
-		const std::string kDOI = s1 + "DOI    %-60.60s ";
-		pdbFile << cif::format(kDOI, doi) << '\n';
+		pdbFile << s1 << cif::format("DOI    {:<60.60s} ", doi) << '\n';
 		++result;
 	}

@@ -307,10 +302,10 @@ void write_header_lines(std::ostream &pdbFile, const datablock &db)
 {
 	//    0         1         2         3         4         5         6         7         8
 	//    HEADER    xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxDDDDDDDDD   IIII
-	const char kHeader[] =
-		"HEADER    %-40.40s"
-		"%-9.9s"
-		"   %-4.4s";
+	// const char kHeader[] =
+	// 	"HEADER    {:<40.40s}"
+	// 	"{:<9.9s}"
+	// 	"   {:<4.4s}";

 	// HEADER

@@ -345,7 +340,12 @@ void write_header_lines(std::ostream &pdbFile, const datablock &db)
 		}
 	}

-	pdbFile << cif::format(kHeader, keywords, date, db.name()) << '\n';
+	pdbFile << cif::format(/* kHeader */
+		"HEADER    {:<40.40s}"
+		"{:<9.9s}"
+		"   {:<4.4s}"
+	
+	, keywords, date, db.name()) << '\n';

 	// TODO: implement
 	// OBSLTE (skip for now)
@@ -535,7 +535,6 @@ void WriteTitle(std::ostream &pdbFile, const datablock &db)
 	write_header_lines(pdbFile, db);

 	// REVDAT
-	const char kRevDatFmt[] = "REVDAT %3d%2.2s %9.9s %4.4s    %1d      ";
 	auto &cat2 = db["database_PDB_rev"];
 	std::vector<row_handle> rev(cat2.begin(), cat2.end());
 	sort(rev.begin(), rev.end(), [](row_handle a, row_handle b) -> bool
@@ -559,9 +558,9 @@ void WriteTitle(std::ostream &pdbFile, const datablock &db)
 		{
 			std::string cs = ++continuation > 1 ? std::to_string(continuation) : std::string();

-			pdbFile << cif::format(kRevDatFmt, revNum, cs, date, db.name(), modType);
+			pdbFile << cif::format("REVDAT {:3}{:2.2s} {:9.9s} {:4.4s}    {:1}      ", revNum, cs, date, db.name(), modType);
 			for (std::size_t i = 0; i < 4; ++i)
-				pdbFile << cif::format(" %-6.6s", (i < types.size() ? types[i] : std::string()));
+				pdbFile << cif::format(" {:<6.6s}", (i < types.size() ? types[i] : std::string()));
 			pdbFile << '\n';

 			if (types.size() > 4)
@@ -614,7 +613,7 @@ void WriteRemark2(std::ostream &pdbFile, const datablock &db)
 		{
 			float resHigh = refine.front()["ls_d_res_high"].as<float>();
 			pdbFile << "REMARK   2\n"
-					<< cif::format("REMARK   2 RESOLUTION. %7.2f ANGSTROMS.", resHigh) << '\n';
+					<< cif::format("REMARK   2 RESOLUTION. {:7.2f} ANGSTROMS.", resHigh) << '\n';
 		}
 		catch (...)
 		{ /* skip it */
@@ -761,10 +760,7 @@ class Fs : public FBase
 		else
 		{
 			os << '\n';
-
-			std::stringstream ss;
-			ss << "REMARK " << std::setw(3) << std::right << mNr << ' ';
-			WriteOneContinuedLine(os, ss.str(), 0, s);
+			WriteOneContinuedLine(os, cif::format("REMARK {:3} ", mNr), 0, s);
 		}
 	}

@@ -1617,7 +1613,7 @@ void WriteRemark3Phenix(std::ostream &pdbFile, const datablock &db)

 		percent_reflns_obs /= 100;

-		pdbFile << RM3("  ") << cif::format("%3d %7.4f - %7.4f    %4.2f %8d %5d  %6.4f %6.4f", bin++, d_res_low, d_res_high, percent_reflns_obs, number_reflns_R_work, number_reflns_R_free, R_factor_R_work, R_factor_R_free) << '\n';
+		pdbFile << RM3("  ") << cif::format("{:3} {:7.4f} - {:7.4f}    {:4.2f} {:8} {:5}  {:6.4f} {:6.4f}", bin++, d_res_low, d_res_high, percent_reflns_obs, number_reflns_R_work, number_reflns_R_free, R_factor_R_work, R_factor_R_free) << '\n';
 	}

 	pdbFile << RM3("") << '\n'
@@ -2585,7 +2581,7 @@ void WriteRemark465(std::ostream &pdbFile, const datablock &db)
 		cif::tie(modelNr, resName, chainID, iCode, seqNr) =
 			r.get("PDB_model_num", "auth_comp_id", "auth_asym_id", "PDB_ins_code", "auth_seq_id");

-		pdbFile << cif::format("REMARK 465 %3.3s %3.3s %1.1s %5d%1.1s", modelNr, resName, chainID, seqNr, iCode) << '\n';
+		pdbFile << cif::format("REMARK 465 {:3.3s} {:3.3s} {:1.1s} {:5}{:1.1s}", modelNr, resName, chainID, seqNr, iCode) << '\n';
 	}
 }

@@ -2632,7 +2628,7 @@ void WriteRemark470(std::ostream &pdbFile, const datablock &db)

 			while (not a.second.empty())
 			{
-				pdbFile << cif::format("REMARK 470 %3.3s %3.3s %1.1s%4d%1.1s  ", modelNr, resName, chainID, seqNr, iCode) << "  ";
+				pdbFile << cif::format("REMARK 470 {:>3.3s} {:3.3s} {:1.1s}{:4}{:1.1s}  ", modelNr, resName, chainID, seqNr, iCode) << "  ";

 				for (std::size_t i = 0; i < 6 and not a.second.empty(); ++i)
 				{
@@ -2730,16 +2726,16 @@ int WritePrimaryStructure(std::ostream &pdbFile, const datablock &db)

 			if (dbAccession.length() > 8 or db_code.length() > 12 or atoi(dbseqEnd.c_str()) >= 100000)
 				pdbFile << cif::format(
-							   "DBREF1 %4.4s %1.1s %4.4s%1.1s %4.4s%1.1s %-6.6s               %-20.20s",
+							   "DBREF1 {:>4.4s} {:1.1s} {:>4.4s}{:1.1s} {:>4.4s}{:1.1s} {:<6.6s}               {:<20.20s}",
 							   idCode, chainID, seqBegin, insertBegin, seqEnd, insertEnd, db_name, db_code)
 						<< '\n'
 						<< cif::format(
-							   "DBREF2 %4.4s %1.1s     %-22.22s     %10.10s  %10.10s",
+							   "DBREF2 {:>4.4s} {:1.1s}     {:<22.22s}     {:10.10s}  {:10.10s}",
 							   idCode, chainID, dbAccession, dbseqBegin, dbseqEnd)
 						<< '\n';
 			else
 				pdbFile << cif::format(
-							   "DBREF  %4.4s %1.1s %4.4s%1.1s %4.4s%1.1s %-6.6s %-8.8s %-12.12s %5.5s%1.1s %5.5s%1.1s",
+							   "DBREF  {:>4.4s} {:1.1s} {:>4.4s}{:1.1s} {:>4.4s}{:1.1s} {:<6.6s} {:<8.8s} {:<12.12s} {:>5.5s}{:1.1s} {:>5.5s}{:1.1s}",
 							   idCode, chainID, seqBegin, insertBegin, seqEnd, insertEnd, db_name, dbAccession, db_code, dbseqBegin, dbinsBeg, dbseqEnd, dbinsEnd)
 						<< '\n';
 		}
@@ -2758,9 +2754,8 @@ int WritePrimaryStructure(std::ostream &pdbFile, const datablock &db)
 		to_upper(conflict);

 		pdbFile << cif::format(
-					   "SEQADV %4.4s %3.3s %1.1s %4.4s%1.1s %-4.4s %-9.9s %3.3s %5.5s %-21.21s",
+					   "SEQADV {:4.4s} {:3.3s} {:1.1s} {:>4.4s}{:1.1s} {:<4.4s} {:<9.9s} {:3.3s} {:>5.5s} {:<21.21s}",
 					   idCode, resName, chainID, seqNum, iCode, database, dbAccession, dbRes, dbSeq, conflict)
-					   .str()
 				<< '\n';
 	}

@@ -2788,7 +2783,7 @@ int WritePrimaryStructure(std::ostream &pdbFile, const datablock &db)
 				t = 13;

 			pdbFile << cif::format(
-						   "SEQRES %3d %1.1s %4d  %-51.51s          ",
+						   "SEQRES {:3} {:1.1s} {:4}  {:<51.51s}          ",
 						   n++, std::string{ chainID }, seqresl[chainID], join(seq.begin(), seq.begin() + t, " "))
 					<< '\n';

@@ -2808,9 +2803,8 @@ int WritePrimaryStructure(std::ostream &pdbFile, const datablock &db)
 			r.get("auth_asym_id", "auth_seq_id", "auth_comp_id", "PDB_ins_code", "parent_comp_id", "details");

 		pdbFile << cif::format(
-					   "MODRES %4.4s %3.3s %1.1s %4.4s%1.1s %3.3s  %-41.41s",
+					   "MODRES {:4.4s} {:3.3s} {:1.1s} {:4.4s}{:1.1s} {:3.3s}  {:<41.41s}",
 					   db.name(), resName, chainID, seqNum, iCode, stdRes, comment)
-					   .str()
 				<< '\n';
 	}

@@ -2925,7 +2919,7 @@ int WriteHeterogen(std::ostream &pdbFile, const datablock &db)
 	{
 		if (h.water)
 			continue;
-		pdbFile << cif::format("HET    %3.3s  %c%4d%c  %5d", h.hetID, h.chainID, h.seqNum, h.iCode, h.numHetAtoms) << '\n';
+		pdbFile << cif::format("HET    {:3.3s}  {:1c}{:4}{:1c}  {:5}", h.hetID, h.chainID, h.seqNum, h.iCode, h.numHetAtoms) << '\n';
 		++numHet;
 	}

@@ -2940,7 +2934,7 @@ int WriteHeterogen(std::ostream &pdbFile, const datablock &db)

 		for (;;)
 		{
-			pdbFile << cif::format("HETNAM  %2.2s %3.3s ", (c > 1 ? std::to_string(c) : std::string()), id);
+			pdbFile << cif::format("HETNAM  {:2.2s} {:3.3s} ", (c > 1 ? std::to_string(c) : std::string()), id);
 			++c;

 			if (name.length() > 55)
@@ -3032,7 +3026,7 @@ int WriteHeterogen(std::ostream &pdbFile, const datablock &db)
 			{
 				std::stringstream fs;

-				fs << cif::format("FORMUL  %2d  %3.3s %2.2s%c", componentNr, hetID, (c > 1 ? std::to_string(c) : std::string()), (hetID == water_comp_id ? '*' : ' '));
+				fs << cif::format("FORMUL  {:2}  {:3.3s} {:2.2s}{:1c}", componentNr, hetID, (c > 1 ? std::to_string(c) : std::string()), (hetID == water_comp_id ? '*' : ' '));
 				++c;

 				if (formula.length() > 51)
@@ -3099,7 +3093,7 @@ std::tuple<int, int> WriteSecondaryStructure(std::ostream &pdbFile, const databl
 				"pdbx_PDB_helix_class", "pdbx_PDB_helix_length", "beg_auth_seq_id", "end_auth_seq_id");

 		++numHelix;
-		pdbFile << cif::format("HELIX  %3d %3.3s %3.3s %1.1s %4d%1.1s %3.3s %1.1s %4d%1.1s%2d%-30.30s %5d",
+		pdbFile << cif::format("HELIX  {:3} {:>3.3s} {:3.3s} {:1.1s} {:4}{:1.1s} {:3.3s} {:1.1s} {:4}{:1.1s}{:2}{:<30.30s} {:5}",
 					   numHelix, pdbx_PDB_helix_id, beg_label_comp_id, beg_auth_asym_id, beg_auth_seq_id, pdbx_beg_PDB_ins_code, end_label_comp_id, end_auth_asym_id, end_auth_seq_id, pdbx_end_PDB_ins_code, pdbx_PDB_helix_class, details, pdbx_PDB_helix_length)
 				<< '\n';
 	}
@@ -3136,7 +3130,7 @@ std::tuple<int, int> WriteSecondaryStructure(std::ostream &pdbFile, const databl
 					"pdbx_end_PDB_ins_code", "beg_auth_comp_id", "beg_auth_asym_id", "beg_auth_seq_id",
 					"end_auth_comp_id", "end_auth_asym_id", "end_auth_seq_id");

-				pdbFile << cif::format("SHEET  %3.3s %3.3s%2d %3.3s %1.1s%4d%1.1s %3.3s %1.1s%4d%1.1s%2d", rangeID1, sheetID, numStrands, initResName, initChainID, initSeqNum, initICode, endResName, endChainID, endSeqNum, endICode, 0) << '\n';
+				pdbFile << cif::format("SHEET  {:>3.3s} {:>3.3s}{:2} {:3.3s} {:1.1s}{:4}{:1.1s} {:3.3s} {:1.1s}{:4}{:1.1s}{:2}", rangeID1, sheetID, numStrands, initResName, initChainID, initSeqNum, initICode, endResName, endChainID, endSeqNum, endICode, 0) << '\n';

 				first = false;
 			}
@@ -3155,7 +3149,7 @@ std::tuple<int, int> WriteSecondaryStructure(std::ostream &pdbFile, const databl

 			if (h.empty())
 			{
-				pdbFile << cif::format("SHEET  %3.3s %3.3s%2d %3.3s %1.1s%4d%1.1s %3.3s %1.1s%4d%1.1s%2d", rangeID2, sheetID, numStrands, initResName, initChainID, initSeqNum, initICode, endResName, endChainID, endSeqNum, endICode, sense) << '\n';
+				pdbFile << cif::format("SHEET  {:>3.3s} {:>3.3s}{:2} {:3.3s} {:1.1s}{:4}{:1.1s} {:3.3s} {:1.1s}{:4}{:1.1s}{:2}", rangeID2, sheetID, numStrands, initResName, initChainID, initSeqNum, initICode, endResName, endChainID, endSeqNum, endICode, sense) << '\n';
 			}
 			else
 			{
@@ -3168,8 +3162,8 @@ std::tuple<int, int> WriteSecondaryStructure(std::ostream &pdbFile, const databl
 				curAtom = cif2pdbAtomName(curAtom, compID[0], db);
 				prevAtom = cif2pdbAtomName(prevAtom, compID[1], db);

-				pdbFile << cif::format("SHEET  %3.3s %3.3s%2d %3.3s %1.1s%4d%1.1s %3.3s %1.1s%4d%1.1s%2d "
-										"%-4.4s%3.3s %1.1s%4d%1.1s %-4.4s%3.3s %1.1s%4d%1.1s",
+				pdbFile << cif::format("SHEET  {:>3.3s} {:>3.3s}{:2} {:3.3s} {:1.1s}{:4}{:1.1s} {:3.3s} {:1.1s}{:4}{:1.1s}{:2} "
+										"{:<4.4s}{:3.3s} {:1.1s}{:4}{:1.1s} {:<4.4s}{:3.3s} {:1.1s}{:4}{:1.1s}",
 							   rangeID2, sheetID, numStrands, initResName, initChainID, initSeqNum, initICode, endResName, endChainID, endSeqNum, endICode, sense, curAtom, curResName, curChainID, curResSeq, curICode, prevAtom, prevResName, prevChainID, prevResSeq, prevICode)
 						<< '\n';
 			}
@@ -3207,7 +3201,7 @@ void WriteConnectivity(std::ostream &pdbFile, const datablock &db)
 		sym1 = cif2pdbSymmetry(sym1);
 		sym2 = cif2pdbSymmetry(sym2);

-		pdbFile << cif::format("SSBOND %3d CYS %1.1s %4d%1.1s   CYS %1.1s %4d%1.1s                       %6.6s %6.6s %5.2f", nr, chainID1, seqNum1, icode1, chainID2, seqNum2, icode2, sym1, sym2, Length) << '\n';
+		pdbFile << cif::format("SSBOND {:3} CYS {:1.1s} {:4}{:1.1s}   CYS {:1.1s} {:4}{:1.1s}                       {:6.6s} {:6.6s} {:5.2f}", nr, chainID1, seqNum1, icode1, chainID2, seqNum2, icode2, sym1, sym2, Length) << '\n';

 		++nr;
 	}
@@ -3234,10 +3228,10 @@ void WriteConnectivity(std::ostream &pdbFile, const datablock &db)
 		sym1 = cif2pdbSymmetry(sym1);
 		sym2 = cif2pdbSymmetry(sym2);

-		pdbFile << cif::format("LINK        %-4.4s%1.1s%3.3s %1.1s%4d%1.1s               %-4.4s%1.1s%3.3s %1.1s%4d%1.1s  %6.6s %6.6s", name1, altLoc1, resName1, chainID1, resSeq1, iCode1, name2, altLoc2, resName2, chainID2, resSeq2, iCode2, sym1, sym2);
+		pdbFile << cif::format("LINK        {:<4.4s}{:1.1s}{:3.3s} {:1.1s}{:4}{:1.1s}               {:<4.4s}{:1.1s}{:3.3s} {:1.1s}{:4}{:1.1s}  {:>6.6s} {:>6.6s}", name1, altLoc1, resName1, chainID1, resSeq1, iCode1, name2, altLoc2, resName2, chainID2, resSeq2, iCode2, sym1, sym2);

 		if (not Length.empty())
-			pdbFile << cif::format(" %5.2f", stod(Length));
+			pdbFile << cif::format(" {:5.2f}", stod(Length));

 		pdbFile << '\n';
 	}
@@ -3255,7 +3249,7 @@ void WriteConnectivity(std::ostream &pdbFile, const datablock &db)
 				"pdbx_label_comp_id_2", "pdbx_auth_asym_id_2", "pdbx_auth_seq_id_2", "pdbx_PDB_ins_code_2",
 				"pdbx_PDB_model_num", "pdbx_omega_angle");

-		pdbFile << cif::format("CISPEP %3.3s %3.3s %1.1s %4d%1.1s   %3.3s %1.1s %4d%1.1s       %3.3s       %6.2f",
+		pdbFile << cif::format("CISPEP {:3.3s} {:3.3s} {:1.1s} {:4}{:1.1s}   {:3.3s} {:1.1s} {:4}{:1.1s}       {:3.3s}       {:6.2f}",
 			serNum, pep1, chainID1, seqNum1, icode1, pep2, chainID2, seqNum2, icode2, modNum, measure) << '\n';
 	}
 }
@@ -3276,7 +3270,7 @@ int WriteMiscellaneousFeatures(std::ostream &pdbFile, const datablock &db)
 		cif::tie(siteID, resName, chainID, seq, iCode) =
 			r.get("site_id", "auth_comp_id", "auth_asym_id", "auth_seq_id", "pdbx_auth_ins_code");

-		sites[siteID].push_back(cif::format("%3.3s %1.1s%4d%1.1s ", resName, chainID, seq, iCode).str());
+		sites[siteID].push_back(cif::format("{:3.3s} {:1.1s}{:4}{:1.1s} ", resName, chainID, seq, iCode));
 	}

 	for (auto s : sites)
@@ -3289,7 +3283,7 @@ int WriteMiscellaneousFeatures(std::ostream &pdbFile, const datablock &db)
 		int nr = 1;
 		while (res.empty() == false)
 		{
-			pdbFile << cif::format("SITE   %3d %3.3s %2d ", nr, siteID, numRes);
+			pdbFile << cif::format("SITE   {:3} {:3.3s} {:2} ", nr, siteID, numRes);

 			for (int i = 0; i < 4; ++i)
 			{
@@ -3318,7 +3312,7 @@ void WriteCrystallographic(std::ostream &pdbFile, const datablock &db)

 	r = db["cell"].find_first(key("entry_id") == db.name());

-	pdbFile << cif::format("CRYST1%9.3f%9.3f%9.3f%7.2f%7.2f%7.2f %-11.11s%4d", r["length_a"].as<double>(), r["length_b"].as<double>(), r["length_c"].as<double>(), r["angle_alpha"].as<double>(), r["angle_beta"].as<double>(), r["angle_gamma"].as<double>(), symmetry, r["Z_PDB"].as<int>()) << '\n';
+	pdbFile << cif::format("CRYST1{:9.3f}{:9.3f}{:9.3f}{:7.2f}{:7.2f}{:7.2f} {:<11.11s}{:4}", r["length_a"].as<double>(), r["length_b"].as<double>(), r["length_c"].as<double>(), r["angle_alpha"].as<double>(), r["angle_beta"].as<double>(), r["angle_gamma"].as<double>(), symmetry, r["Z_PDB"].as<int>()) << '\n';
 }

 int WriteCoordinateTransformation(std::ostream &pdbFile, const datablock &db)
@@ -3327,18 +3321,18 @@ int WriteCoordinateTransformation(std::ostream &pdbFile, const datablock &db)

 	for (auto r : db["database_PDB_matrix"])
 	{
-		pdbFile << cif::format("ORIGX%1d    %10.6f%10.6f%10.6f     %10.5f", 1, r["origx[1][1]"].as<float>(), r["origx[1][2]"].as<float>(), r["origx[1][3]"].as<float>(), r["origx_vector[1]"].as<float>()) << '\n';
-		pdbFile << cif::format("ORIGX%1d    %10.6f%10.6f%10.6f     %10.5f", 2, r["origx[2][1]"].as<float>(), r["origx[2][2]"].as<float>(), r["origx[2][3]"].as<float>(), r["origx_vector[2]"].as<float>()) << '\n';
-		pdbFile << cif::format("ORIGX%1d    %10.6f%10.6f%10.6f     %10.5f", 3, r["origx[3][1]"].as<float>(), r["origx[3][2]"].as<float>(), r["origx[3][3]"].as<float>(), r["origx_vector[3]"].as<float>()) << '\n';
+		pdbFile << cif::format("ORIGX{:1}    {:10.6f}{:10.6f}{:10.6f}     {:10.5f}", 1, r["origx[1][1]"].as<float>(), r["origx[1][2]"].as<float>(), r["origx[1][3]"].as<float>(), r["origx_vector[1]"].as<float>()) << '\n';
+		pdbFile << cif::format("ORIGX{:1}    {:10.6f}{:10.6f}{:10.6f}     {:10.5f}", 2, r["origx[2][1]"].as<float>(), r["origx[2][2]"].as<float>(), r["origx[2][3]"].as<float>(), r["origx_vector[2]"].as<float>()) << '\n';
+		pdbFile << cif::format("ORIGX{:1}    {:10.6f}{:10.6f}{:10.6f}     {:10.5f}", 3, r["origx[3][1]"].as<float>(), r["origx[3][2]"].as<float>(), r["origx[3][3]"].as<float>(), r["origx_vector[3]"].as<float>()) << '\n';
 		result += 3;
 		break;
 	}

 	for (auto r : db["atom_sites"])
 	{
-		pdbFile << cif::format("SCALE%1d    %10.6f%10.6f%10.6f     %10.5f", 1, r["fract_transf_matrix[1][1]"].as<float>(), r["fract_transf_matrix[1][2]"].as<float>(), r["fract_transf_matrix[1][3]"].as<float>(), r["fract_transf_vector[1]"].as<float>()) << '\n';
-		pdbFile << cif::format("SCALE%1d    %10.6f%10.6f%10.6f     %10.5f", 2, r["fract_transf_matrix[2][1]"].as<float>(), r["fract_transf_matrix[2][2]"].as<float>(), r["fract_transf_matrix[2][3]"].as<float>(), r["fract_transf_vector[2]"].as<float>()) << '\n';
-		pdbFile << cif::format("SCALE%1d    %10.6f%10.6f%10.6f     %10.5f", 3, r["fract_transf_matrix[3][1]"].as<float>(), r["fract_transf_matrix[3][2]"].as<float>(), r["fract_transf_matrix[3][3]"].as<float>(), r["fract_transf_vector[3]"].as<float>()) << '\n';
+		pdbFile << cif::format("SCALE{:1}    {:10.6f}{:10.6f}{:10.6f}     {:10.5f}", 1, r["fract_transf_matrix[1][1]"].as<float>(), r["fract_transf_matrix[1][2]"].as<float>(), r["fract_transf_matrix[1][3]"].as<float>(), r["fract_transf_vector[1]"].as<float>()) << '\n';
+		pdbFile << cif::format("SCALE{:1}    {:10.6f}{:10.6f}{:10.6f}     {:10.5f}", 2, r["fract_transf_matrix[2][1]"].as<float>(), r["fract_transf_matrix[2][2]"].as<float>(), r["fract_transf_matrix[2][3]"].as<float>(), r["fract_transf_vector[2]"].as<float>()) << '\n';
+		pdbFile << cif::format("SCALE{:1}    {:10.6f}{:10.6f}{:10.6f}     {:10.5f}", 3, r["fract_transf_matrix[3][1]"].as<float>(), r["fract_transf_matrix[3][2]"].as<float>(), r["fract_transf_matrix[3][3]"].as<float>(), r["fract_transf_vector[3]"].as<float>()) << '\n';
 		result += 3;
 		break;
 	}
@@ -3348,9 +3342,9 @@ int WriteCoordinateTransformation(std::ostream &pdbFile, const datablock &db)
 	{
 		std::string given = r["code"] == "given" ? "1" : "";

-		pdbFile << cif::format("MTRIX%1d %3d%10.6f%10.6f%10.6f     %10.5f    %1.1s", 1, nr, r["matrix[1][1]"].as<float>(), r["matrix[1][2]"].as<float>(), r["matrix[1][3]"].as<float>(), r["vector[1]"].as<float>(), given) << '\n';
-		pdbFile << cif::format("MTRIX%1d %3d%10.6f%10.6f%10.6f     %10.5f    %1.1s", 2, nr, r["matrix[2][1]"].as<float>(), r["matrix[2][2]"].as<float>(), r["matrix[2][3]"].as<float>(), r["vector[2]"].as<float>(), given) << '\n';
-		pdbFile << cif::format("MTRIX%1d %3d%10.6f%10.6f%10.6f     %10.5f    %1.1s", 3, nr, r["matrix[3][1]"].as<float>(), r["matrix[3][2]"].as<float>(), r["matrix[3][3]"].as<float>(), r["vector[3]"].as<float>(), given) << '\n';
+		pdbFile << cif::format("MTRIX{:1} {:3}{:10.6f}{:10.6f}{:10.6f}     {:10.5f}    {:1.1s}", 1, nr, r["matrix[1][1]"].as<float>(), r["matrix[1][2]"].as<float>(), r["matrix[1][3]"].as<float>(), r["vector[1]"].as<float>(), given) << '\n';
+		pdbFile << cif::format("MTRIX{:1} {:3}{:10.6f}{:10.6f}{:10.6f}     {:10.5f}    {:1.1s}", 2, nr, r["matrix[2][1]"].as<float>(), r["matrix[2][2]"].as<float>(), r["matrix[2][3]"].as<float>(), r["vector[2]"].as<float>(), given) << '\n';
+		pdbFile << cif::format("MTRIX{:1} {:3}{:10.6f}{:10.6f}{:10.6f}     {:10.5f}    {:1.1s}", 3, nr, r["matrix[3][1]"].as<float>(), r["matrix[3][2]"].as<float>(), r["matrix[3][3]"].as<float>(), r["vector[3]"].as<float>(), given) << '\n';

 		++nr;
 		result += 3;
@@ -3369,10 +3363,6 @@ std::tuple<int, int> WriteCoordinatesForModel(std::ostream &pdbFile, const datab

 	auto &atom_site = db["atom_site"];
 	auto &atom_site_anisotrop = db["atom_site_anisotrop"];
-	auto &entity = db["entity"];
-	// auto &pdbx_poly_seq_scheme = db["pdbx_poly_seq_scheme"];
-	// auto &pdbx_nonpoly_scheme = db["pdbx_nonpoly_scheme"];
-	auto &pdbx_branch_scheme = db["pdbx_branch_scheme"];

 	int serial = 1;
 	auto ri = atom_site.begin();
@@ -3417,7 +3407,7 @@ std::tuple<int, int> WriteCoordinatesForModel(std::ostream &pdbFile, const datab

 			if (terminate)
 			{
-				pdbFile << cif::format("TER   %5d      %3.3s %1.1s%4d%1.1s",  serial,  resName,  chainID,  resSeq,  iCode) << '\n';
+				pdbFile << cif::format("TER   {:5}      {:3.3s} {:1.1s}{:4}{:1.1s}",  serial,  resName,  chainID,  resSeq,  iCode) << '\n';

 				++serial;
 				terminatedChains.insert(chainID);
@@ -3446,26 +3436,6 @@ std::tuple<int, int> WriteCoordinatesForModel(std::ostream &pdbFile, const datab
 			r.get("id", "group_PDB", "label_atom_id", "label_alt_id", "auth_comp_id", "auth_asym_id", "auth_seq_id",
 				"pdbx_PDB_ins_code", "Cartn_x", "Cartn_y", "Cartn_z", "occupancy", "B_iso_or_equiv", "type_symbol", "pdbx_formal_charge");

-		if (resName != "HOH")
-		{
-			int entity_id = r.get<int>("label_entity_id");
-			try
-			{
-				auto type = entity.find1<std::string>("id"_key == entity_id, "type");
-
-				if (type == "branched")	// find the real auth_seq_num, since sugars have their auth_seq_num reused as sugar number... sigh.
-					resSeq = pdbx_branch_scheme.find1<int>("asym_id"_key == r.get<std::string>("label_asym_id") and "pdb_seq_num"_key == resSeq, "auth_seq_num");
-				// else if (type == "non-polymer")	// same for non-polymers
-				// 	resSeq = pdbx_nonpoly_scheme.find1<int>("asym_id"_key == r.get<std::string>("label_asym_id") and "pdb_seq_num"_key == resSeq, "auth_seq_num");
-				// else if (type == "polymer")
-				// 	resSeq = pdbx_poly_seq_scheme.find1<int>("asym_id"_key == r.get<std::string>("label_asym_id") and "pdb_seq_num"_key == resSeq, "auth_seq_num");
-			}
-			catch (const std::exception &ex)
-			{
-				std::cerr << "Oops, there was not exactly one entity with id " << entity_id << '\n';
-			}
-		}
-		
 		if (chainID.length() > 1)
 			throw std::runtime_error("Chain ID " + chainID + " won't fit into a PDB file");

@@ -3476,7 +3446,8 @@ std::tuple<int, int> WriteCoordinatesForModel(std::ostream &pdbFile, const datab
 		if (charge != 0)
 			sCharge = std::to_string(charge) + (charge > 0 ? '+' : '-');

-		pdbFile << cif::format("%-6.6s%5d %-4.4s%1.1s%3.3s %1.1s%4d%1.1s   %8.3f%8.3f%8.3f%6.2f%6.2f          %2.2s%2.2s", group, serial, name, altLoc, resName, chainID, resSeq, iCode, x, y, z, occupancy, tempFactor, element, sCharge) << '\n';
+		pdbFile << cif::format("{:<6.6s}{:5} {:<4.4s}{:1.1s}{:3.3s} {:1.1s}{:4}{:1.1s}   {:8.3f}{:8.3f}{:8.3f}{:6.2f}{:6.2f}          {:>2.2s}{:2.2s}",
+			group, serial, name, altLoc, resName, chainID, resSeq, iCode, x, y, z, occupancy, tempFactor, element, sCharge) << '\n';

 		++numCoord;

@@ -3491,7 +3462,7 @@ std::tuple<int, int> WriteCoordinatesForModel(std::ostream &pdbFile, const datab
 			tie(u11, u22, u33, u12, u13, u23) =
 				ai.get("U[1][1]", "U[2][2]", "U[3][3]", "U[1][2]", "U[1][3]", "U[2][3]");

-			pdbFile << cif::format("ANISOU%5d %-4.4s%1.1s%3.3s %1.1s%4d%1.1s %7d%7d%7d%7d%7d%7d      %2.2s%2.2s", serial, name, altLoc, resName, chainID, resSeq, iCode, std::lrintf(u11 * 10000), std::lrintf(u22 * 10000), std::lrintf(u33 * 10000), std::lrintf(u12 * 10000), std::lrintf(u13 * 10000), std::lrintf(u23 * 10000), element, sCharge) << '\n';
+			pdbFile << cif::format("ANISOU{:5} {:<4.4s}{:1.1s}{:3.3s} {:1.1s}{:4}{:1.1s} {:7}{:7}{:7}{:7}{:7}{:7}      {:2.2s}{:2.2s}", serial, name, altLoc, resName, chainID, resSeq, iCode, std::lrintf(u11 * 10000), std::lrintf(u22 * 10000), std::lrintf(u33 * 10000), std::lrintf(u12 * 10000), std::lrintf(u13 * 10000), std::lrintf(u23 * 10000), element, sCharge) << '\n';
 		}

 		++serial;
@@ -3543,7 +3514,7 @@ std::tuple<int, int> WriteCoordinate(std::ostream &pdbFile, const datablock &db)
 		for (int model_nr : models)
 		{
 			if (models.size() > 1)
-				pdbFile << cif::format("MODEL     %4d",  model_nr) << '\n';
+				pdbFile << cif::format("MODEL     {:4}",  model_nr) << '\n';

 			std::set<std::string> TERminatedChains;
 			auto n = WriteCoordinatesForModel(pdbFile, db, last_resseq_for_chain_map, TERminatedChains, model_nr);
@@ -3615,7 +3586,7 @@ std::string get_HEADER_line(const datablock &db, std::string::size_type truncate
 		}
 	}

-	return FixStringLength(cif::format("HEADER    %-40.40s%-9.9s   %-4.4s", keywords, date, db.name()).str(), truncate_at);
+	return FixStringLength(cif::format("HEADER    {:<40.40s}{:<9.9s}   {:<4.4s}", keywords, date, db.name()), truncate_at);
 }

 std::string get_COMPND_line(const datablock &db, std::string::size_type truncate_at)
@@ -3788,7 +3759,7 @@ void write(std::ostream &os, const datablock &db)
 	numXform = WriteCoordinateTransformation(os, db);
 	std::tie(numCoord, numTer) = WriteCoordinate(os, db);

-	os << cif::format("MASTER    %5d    0%5d%5d%5d%5d%5d%5d%5d%5d%5d%5d",  numRemark,  numHet,  numHelix,  numSheet,  numTurn,  numSite,  numXform,  numCoord,  numTer,  numConect,  numSeq) << '\n'
+	os << cif::format("MASTER    {:5}    0{:5}{:5}{:5}{:5}{:5}{:5}{:5}{:5}{:5}{:5}",  numRemark,  numHet,  numHelix,  numSheet,  numTurn,  numSite,  numXform,  numCoord,  numTer,  numConect,  numSeq) << '\n'
 			<< "END\n";
 }

--- a/src/pdb/pdb2cif.cpp
+++ b/src/pdb/pdb2cif.cpp
@@ -32,6 +32,7 @@
 #include <map>
 #include <set>
 #include <stack>
+#include <stdexcept>

 using cif::category;
 using cif::datablock;
@@ -895,12 +896,7 @@ class PDBFileParser
 				if (year < 1950)
 					year += 100;

-				std::stringstream ss;
-				ss << std::setw(4) << std::setfill('0') << year << '-'
-				   << std::setw(2) << std::setfill('0') << month << '-'
-				   << std::setw(2) << std::setfill('0') << day;
-
-				s = ss.str();
+				s = cif::format("{:04}-{:02}-{:02}", year, month, day);
 			}
 			else if (regex_match(s, m, rx2))
 			{
@@ -912,7 +908,7 @@ class PDBFileParser
 				if (year < 1950)
 					year += 100;

-				s = cif::format("%04d-%02d", year, month).str();
+				s = cif::format("{:04}-{:02}", year, month);
 			}
 			else
 				ec = error::make_error_code(error::pdbErrors::invalidDate);
@@ -3146,7 +3142,6 @@ void PDBFileParser::ParseRemark350()
 	std::map<std::string, std::string> values;
 	std::vector<std::string> asymIdList;
 	std::smatch m;
-	cif::row_handle genR;

 	std::vector<double> mat, vec;

@@ -3334,38 +3329,27 @@ void PDBFileParser::ParseRemark350()

 						std::string type = mat == std::vector<double>{ 1, 0, 0, 0, 1, 0, 0, 0, 1 } and vec == std::vector<double>{ 0, 0, 0 } ? "identity operation" : "crystal symmetry operation";

-						// if (type == "identity operation")
-						// {
-
-						// }
-						// else
-						try
-						{
-							// clang-format off
-							getCategory("pdbx_struct_oper_list")->emplace({
+						auto pdbx_struct_oper_list = getCategory("pdbx_struct_oper_list");
+						if (not pdbx_struct_oper_list->contains(cif::key("id") == operID))
+							getCategory("pdbx_struct_oper_list")->emplace({ // clang-format off
 								{ "id", operID },
 								{ "type", type },
 								// { "name", "" },
 							    // { "symmetryOperation", "" },
-								{ "matrix[1][1]", cif::format("%12.10f", mat[0]).str() },
-								{ "matrix[1][2]", cif::format("%12.10f", mat[1]).str() },
-								{ "matrix[1][3]", cif::format("%12.10f", mat[2]).str() },
-								{ "vector[1]", cif::format("%12.10f", vec[0]).str() },
-								{ "matrix[2][1]", cif::format("%12.10f", mat[3]).str() },
-								{ "matrix[2][2]", cif::format("%12.10f", mat[4]).str() },
-								{ "matrix[2][3]", cif::format("%12.10f", mat[5]).str() },
-								{ "vector[2]", cif::format("%12.10f", vec[1]).str() },
-								{ "matrix[3][1]", cif::format("%12.10f", mat[6]).str() },
-								{ "matrix[3][2]", cif::format("%12.10f", mat[7]).str() },
-								{ "matrix[3][3]", cif::format("%12.10f", mat[8]).str() },
-								{ "vector[3]", cif::format("%12.10f", vec[2]).str() }
+								{ "matrix[1][1]", cif::format("{:12.10f}", mat[0]) },
+								{ "matrix[1][2]", cif::format("{:12.10f}", mat[1]) },
+								{ "matrix[1][3]", cif::format("{:12.10f}", mat[2]) },
+								{ "vector[1]", cif::format("{:12.10f}", vec[0]) },
+								{ "matrix[2][1]", cif::format("{:12.10f}", mat[3]) },
+								{ "matrix[2][2]", cif::format("{:12.10f}", mat[4]) },
+								{ "matrix[2][3]", cif::format("{:12.10f}", mat[5]) },
+								{ "vector[2]", cif::format("{:12.10f}", vec[1]) },
+								{ "matrix[3][1]", cif::format("{:12.10f}", mat[6]) },
+								{ "matrix[3][2]", cif::format("{:12.10f}", mat[7]) },
+								{ "matrix[3][3]", cif::format("{:12.10f}", mat[8]) },
+								{ "vector[3]", cif::format("{:12.10f}", vec[2]) }
 							});
-							// clang-format on
-						}
-						catch (duplicate_key_error &ex)
-						{
-							// so what?
-						}
+																			// clang-format on

 						mat.clear();
 						vec.clear();
@@ -4300,6 +4284,8 @@ void PDBFileParser::ConstructEntities()
 			type = "polypeptide(L)";
 		else if (mightBeDNA and not mightBePolyPeptide)
 			type = "polyribonucleotide";
+		else
+			type = "other";

 		// clang-format off
 		getCategory("entity_poly")->emplace({
@@ -4327,7 +4313,7 @@ void PDBFileParser::ConstructEntities()
 	}

 	// build sugar trees first
-	ConstructSugarTrees(asymNr);
+	// ConstructSugarTrees(asymNr);

 	// done with the sugar, resume operation as before

@@ -4505,7 +4491,7 @@ void PDBFileParser::ConstructEntities()
 	int modResID = 1;
 	std::set<std::string> modResSet;
 	for (auto rec = FindRecord("MODRES"); rec != nullptr and rec->is("MODRES");
-		 rec = rec->mNext)                     //	 1 -  6        Record name   "MODRES"
+		rec = rec->mNext)                      //	 1 -  6        Record name   "MODRES"
 	{                                          //	 8 - 11        IDcode        idCode      ID code of this datablock.
 		std::string resName = rec->vS(13, 15); //	13 - 15        Residue name  resName     Residue name used in this datablock.
 		char chainID = rec->vC(17);            //	17             Character     chainID     Chain identifier.
@@ -5627,7 +5613,7 @@ void PDBFileParser::ParseCoordinateTransformation()
 			igiven = vC(60) == '1';   //	60             Integer       iGiven        1 if coordinates for the  representations
 			                          //	                                           which  are approximately related by the
 			GetNextRecord();          //	                                           transformations  of the molecule are
-		}                             //	                                           contained in the datablock. Otherwise, blank.
+		} //	                                           contained in the datablock. Otherwise, blank.

 		// clang-format off
 		getCategory("struct_ncs_oper")->emplace({
@@ -5781,6 +5767,9 @@ void PDBFileParser::ParseCoordinate(int modelNr)
 		std::string element = vS(77, 78);    //	77 - 78        LString(2)    element      Element symbol, right-justified.
 		std::string charge = vS(79, 80);     //	79 - 80        LString(2)    charge       Charge  on the atom.

+		if (element.empty())
+			throw std::runtime_error("Empty element column in PDB file at line " + std::to_string(mRec->mLineNr));
+
 		std::string entityID = mAsymID2EntityID[asymID];

 		charge = pdb2cifCharge(charge);
@@ -5859,7 +5848,7 @@ void PDBFileParser::ParseCoordinate(int modelNr)

 			auto f = [](float f) -> std::string
 			{
-				return cif::format("%6.4f", f).str();
+				return cif::format("{:6.4f}", f);
 			};

 			// clang-format off
@@ -6413,7 +6402,10 @@ file read(std::istream &is)
 		// apart from the letter 'd', the test has changed into the following:

 		if (std::isalpha(ch) and std::toupper(ch) != 'D')
+		{
 			read_pdb_file(is, result);
+			fixup_pdbx(result);
+		}
 		else
 		{
 			try
@@ -6424,16 +6416,38 @@ file read(std::istream &is)
 			{
 				std::throw_with_nested(std::runtime_error("Since the file did not start with a valid PDB HEADER line mmCIF was assumed, but that failed."));
 			}
-		}

-		// Since we're using the cif::pdb way of reading the file, the data may need
-		// reconstruction
-		reconstruct_pdbx(result);
+			if (not(result.empty() or result.front().empty()))
+			{
+				if (auto &db = result.front(); db.get("audit_conform") == nullptr)
+					reconstruct_pdbx(result);
+				else
+				{
+					try
+					{
+						// Try to see if we can create an mm::structure out of this data.
+						// If that fails, we need to reconstruct a PDBx file out of it.
+
+						cif::mm::structure s(result);
+					}
+					catch (const std::exception &e)
+					{
+						reconstruct_pdbx(result);
+					}
+				}
+			}
+		}
 	}

 	// Must be a PDB like file, right?
-	if (not result.empty() and result.front().get_validator() == nullptr)
-		result.front().set_validator(&validator_factory::instance().get("mmcif_pdbx.dic"));
+	if (not result.empty())
+	{
+		auto &db = result.front();
+		if (db.get_validator() == nullptr)
+			db.set_validator(&validator_factory::instance().get("mmcif_pdbx.dic"));
+		if (db.is_valid())
+			db.get_validator()->fill_audit_conform(db["audit_conform"]);
+	}

 	return result;
 }
--- a/src/pdb/pdb2cif_remark_3.cpp
+++ b/src/pdb/pdb2cif_remark_3.cpp
@@ -1478,6 +1478,8 @@ bool Remark3Parser::parse(const std::string &expMethod, PDBRecord *r, cif::datab

 		best.parser->fixup();

+		auto &validator = cif::validator_factory::instance().get("mmcif_pdbx.dic");
+
 		for (auto &cat1 : best.parser->mDb)
 		{
 			if (cat1.empty())
@@ -1496,8 +1498,15 @@ bool Remark3Parser::parse(const std::string &expMethod, PDBRecord *r, cif::datab
 					auto r1 = cat1.front();
 					auto r2 = cat2.front();

-					for (auto item : cat1.key_items())
-						r2[item] = r1[item].text();
+					auto cv = cat1.get_cat_validator();
+					if (cv == nullptr)
+						cv = validator.get_validator_for_category(cat1.name());
+					
+					if (cv == nullptr)
+						continue;
+
+					for (auto &iv : cv->m_item_validators)
+						r2[iv.m_item_name] = r1[iv.m_item_name].text();
 				}
 			}
 			else
--- a/src/pdb/reconstruct.cpp
+++ b/src/pdb/reconstruct.cpp
@@ -25,6 +25,11 @@
 */

 #include "cif++.hpp"
+#include "cif++/compound.hpp"
+#include "cif++/row.hpp"
+
+#include <stdexcept>
+#include <string>

 // --------------------------------------------------------------------

@@ -104,7 +109,7 @@ void checkEntities(datablock &db)

 		float formula_weight = 0;

-		if (type.empty())	// yes, that happens
+		if (type.empty()) // yes, that happens
 		{
 			const auto comp_id = db["atom_site"].find_first<std::string>("label_entity_id"_key == entity_id, "label_comp_id");
 			auto compound = cf.create(comp_id);
@@ -125,10 +130,10 @@ void checkEntities(datablock &db)

 			if (type.empty())
 				throw std::runtime_error("Entity without type and cannot determine what it should be");
-			
+
 			entity["type"] = type;
 		}
-		
+
 		if (type == "polymer")
 		{
 			int n = 0;
@@ -136,17 +141,17 @@ void checkEntities(datablock &db)
 			for (std::string comp_id : db["pdbx_poly_seq_scheme"].find<std::string>("entity_id"_key == entity_id, "mon_id"))
 			{
 				auto compound = cf.create(comp_id);
-				assert(compound);
-				if (not compound)
-					throw std::runtime_error("missing information for compound " + comp_id);
-				formula_weight += compound->formula_weight();
+				if (compound)
+					formula_weight += compound->formula_weight();
+				// else if (cif::VERBOSE > 0)
+				// 	std::clog << "missing information for compound " + comp_id << '\n';
 				++n;
 			}

-			formula_weight -= (n - 1) * 18.015;
+			formula_weight -= (n - 1) * 18.015f;
 		}
 		else if (type == "water")
-			formula_weight = 18.015;
+			formula_weight = 18.015f;
 		else if (type == "branched")
 		{
 			int n = 0;
@@ -154,14 +159,14 @@ void checkEntities(datablock &db)
 			for (std::string comp_id : db["pdbx_entity_branch_list"].find<std::string>("entity_id"_key == entity_id, "comp_id"))
 			{
 				auto compound = cf.create(comp_id);
-				assert(compound);
-				if (not compound)
-					throw std::runtime_error("missing information for compound " + comp_id);
-				formula_weight += compound->formula_weight();
+				if (compound)
+					formula_weight += compound->formula_weight();
+				// else if (cif::VERBOSE > 0)
+				// 	std::clog << "missing information for compound " + comp_id << '\n';
 				++n;
 			}

-			formula_weight -= (n - 1) * 18.015;
+			formula_weight -= (n - 1) * 18.015f;
 		}
 		else if (type == "non-polymer")
 		{
@@ -171,7 +176,7 @@ void checkEntities(datablock &db)
 				auto compound = cf.create(*comp_id);
 				if (not compound)
 				{
-					std::cerr << "missing information for compound " << *comp_id << "\n";
+					// std::cerr << "missing information for compound " << *comp_id << "\n";
 					continue;
 				}
 				formula_weight = compound->formula_weight();
@@ -185,6 +190,8 @@ void checkEntities(datablock &db)

 void createEntityIDs(datablock &db)
 {
+	using namespace literals;
+
 	// Suppose the file does not have entity ID's. We have to make up some

 	// walk the atoms. For each auth_asym_id we have a new struct_asym.
@@ -196,28 +203,44 @@ void createEntityIDs(datablock &db)
 	// that should cover it

 	auto &atom_site = db["atom_site"];
+	auto &entity = db["entity"];
 	auto &cf = compound_factory::instance();

-	std::vector<std::vector<residue_key_type>> entities;
+	std::vector<std::vector<row_handle>> entities;

 	std::string lastAsymID;
 	int lastSeqID = -1;
-	std::vector<residue_key_type> waters;
+	std::vector<row_handle> waters;

-	for (residue_key_type k : atom_site.rows<std::optional<std::string>,
-							  std::optional<int>,
-							  std::optional<std::string>,
-							  std::optional<std::string>,
-							  std::optional<int>,
-							  std::optional<std::string>>(
-			 "auth_asym_id", "auth_seq_id", "auth_comp_id",
-			 "label_asym_id", "label_seq_id", "label_comp_id"))
+	int nextEntityID;
+	try
 	{
+		if (entity.empty())
+			nextEntityID = 1;
+		else
+			nextEntityID = entity.find_max<int>("id") + 1;
+	}
+	catch (...)
+	{
+		nextEntityID = 1;
+	}
+
+	for (auto rh : atom_site.find("label_entity_id"_key == cif::null))
+	{
+		residue_key_type k = rh.get<std::optional<std::string>,
+			std::optional<int>,
+			std::optional<std::string>,
+			std::optional<std::string>,
+			std::optional<int>,
+			std::optional<std::string>>(
+			"auth_asym_id", "auth_seq_id", "auth_comp_id",
+			"label_asym_id", "label_seq_id", "label_comp_id");
+
 		std::string comp_id = get_comp_id(k);

 		if (cf.is_water(comp_id))
 		{
-			waters.emplace_back(k);
+			waters.emplace_back(rh);
 			continue;
 		}

@@ -226,19 +249,20 @@ void createEntityIDs(datablock &db)

 		bool is_monomer = cf.is_monomer(comp_id);

-		if (lastAsymID == asym_id and lastSeqID == seq_id and not is_monomer)
-			continue;
+		// if (lastAsymID == asym_id and lastSeqID == seq_id and not is_monomer)
+		// 	continue;

 		if (asym_id != lastAsymID or (not is_monomer and lastSeqID != seq_id))
 			entities.push_back({});

-		entities.back().emplace_back(k);
+		entities.back().emplace_back(rh);

 		lastAsymID = asym_id;
 		lastSeqID = seq_id;
 	}

 	std::map<std::size_t, std::string> entity_ids;
+	std::map<std::string, std::string> newEntitiesForCompound;

 	atom_site.add_item("label_entity_id");

@@ -247,7 +271,39 @@ void createEntityIDs(datablock &db)
 		if (entity_ids.contains(i))
 			continue;

-		auto entity_id = std::to_string(i + 1);
+		residue_key_type k = entities[i].front().get<std::optional<std::string>, std::optional<int>, std::optional<std::string>, std::optional<std::string>, std::optional<int>, std::optional<std::string>>(
+			"auth_asym_id", "auth_seq_id", "auth_comp_id",
+			"label_asym_id", "label_seq_id", "label_comp_id");
+
+		std::string comp_id = get_comp_id(k);
+
+		std::string entity_id;
+		if (auto v = db["pdbx_entity_nonpoly"].find_first("comp_id"_key == comp_id); v)
+			entity_id = v.get<std::string>("entity_id");
+		else if (auto i2 = newEntitiesForCompound.find(comp_id); i2 != newEntitiesForCompound.end())
+			entity_id = i2->second;
+		else
+		{
+			entity_id = std::to_string(nextEntityID++);
+
+			if (cf.is_monomer(comp_id))
+				entity.emplace({ //
+					{ "id", entity_id },
+					{ "type", "polymer" } });
+			else if (cf.is_water(comp_id))
+				entity.emplace({ //
+					{ "id", entity_id },
+					{ "type", "water" } });
+			else
+			{
+				entity.emplace({ //
+					{ "id", entity_id },
+					{ "type", "non-polymer" } });
+
+				newEntitiesForCompound[comp_id] = entity_id;
+			}
+		}
+
 		entity_ids[i] = entity_id;

 		for (std::size_t j = i + 1; j < entities.size(); ++j)
@@ -259,20 +315,17 @@ void createEntityIDs(datablock &db)

 	for (std::size_t ix = 0; auto &e : entities)
 	{
-		auto k = e.front();
 		const auto &entity_id = entity_ids[ix++];

-		std::string comp_id = get_comp_id(k);
-
-		for (auto &k : e)
-			atom_site.update_value(get_condition(k), "label_entity_id", entity_id);
+		for (auto rh : e)
+			rh["label_entity_id"] = entity_id;
 	}

 	if (not waters.empty())
 	{
 		std::string waterEntityID = std::to_string(entities.size() + 1);
-		for (auto &k : waters)
-			atom_site.update_value(get_condition(k), "label_entity_id", waterEntityID);
+		for (auto rh : waters)
+			rh["label_entity_id"] = waterEntityID;
 	}
 }

@@ -319,7 +372,7 @@ void fillLabelAsymID(category &atom_site)
 		{
 			if (not mapAuthAsymIDAndEntityToLabelAsymID.contains(key))
 			{
-				std::string asym_id = cif_id_for_number(mapAuthAsymIDAndEntityToLabelAsymID.size());
+				std::string asym_id = cif_id_for_number(static_cast<int>(mapAuthAsymIDAndEntityToLabelAsymID.size()));
 				mapAuthAsymIDAndEntityToLabelAsymID[key] = asym_id;
 			}
 		}
@@ -439,9 +492,38 @@ void checkAtomRecords(datablock &db)
 	if (atom_site.contains(key("label_seq_id") < 0))
 		fixNegativeSeqID(atom_site);

-	std::set<int> polymer_entities;
-	for (int id : db["entity"].find<int>("type"_key == "polymer", "id"))
-		polymer_entities.insert(id);
+	std::set<std::string> polymer_entities;
+	if (db["entity"].empty())
+	{
+		// No entity, so we have to guess the types based on the content of atom_site
+
+		std::string last_entity_id;
+		std::optional<int> last_label_seq_id, last_auth_seq_id;
+
+		std::set<std::string> entityIDs;
+		for (auto &[entity_id, label_comp_id, label_seq_id, auth_comp_id, auth_seq_id] :
+			atom_site.rows<std::string, std::string, std::optional<int>, std::string, std::optional<int>>(
+				"label_entity_id", "label_comp_id", "label_seq_id", "auth_comp_id", "auth_seq_id"))
+		{
+			if (cf.is_water(label_comp_id) or cf.is_water(auth_comp_id))
+				continue;
+
+			if (polymer_entities.contains(entity_id))
+				continue;
+
+			if (last_entity_id == entity_id and (last_label_seq_id != label_seq_id or last_auth_seq_id != auth_seq_id))
+				polymer_entities.emplace(entity_id);
+
+			last_entity_id = entity_id;
+			last_label_seq_id = label_seq_id;
+			last_auth_seq_id = auth_seq_id;
+		}
+	}
+	else
+	{
+		for (std::string id : db["entity"].find<std::string>("type"_key == "polymer", "id"))
+			polymer_entities.insert(id);
+	}

 	std::set<std::string> missingCompounds;

@@ -478,20 +560,21 @@ void checkAtomRecords(datablock &db)
 		if (missingCompounds.contains(comp_id))
 			continue;

-		bool is_polymer = polymer_entities.contains(row["label_entity_id"].as<int>());
+		bool is_polymer = polymer_entities.contains(row["label_entity_id"].as<std::string>());
 		auto compound = cf.create(comp_id);

 		if (not compound)
 		{
 			missingCompounds.insert(comp_id);
-			std::cerr << "Missing compound information for " << comp_id << "\n";
+			// if (cif::VERBOSE > 0)
+			// 	std::cerr << "Missing compound information for " << comp_id << "\n";
 			continue;
 		}

 		auto chem_comp_entry = chem_comp.find_first("id"_key == comp_id);

 		std::optional<bool> non_std;
-		if  (cf.is_monomer(comp_id))
+		if (cf.is_monomer(comp_id))
 			non_std = cf.is_std_monomer(comp_id);

 		if (not chem_comp_entry)
@@ -531,18 +614,24 @@ void checkAtomRecords(datablock &db)
 		if (is_polymer and row["label_seq_id"].empty() and cf.is_monomer(comp_id))
 			row["label_seq_id"] = std::to_string(seq_id);

-		if (row["label_atom_id"].empty())
-			row["label_atom_id"] = row["auth_atom_id"].text();
 		if (row["label_asym_id"].empty())
 			row["label_asym_id"] = row["auth_asym_id"].text();
+		else if (row["auth_asym_id"].empty())
+			row["auth_asym_id"] = row["label_asym_id"].text();
+
 		if (row["label_comp_id"].empty())
 			row["label_comp_id"] = row["auth_comp_id"].text();
+		else if (row["auth_comp_id"].empty())
+			row["auth_comp_id"] = row["label_comp_id"].text();
+
 		if (row["label_atom_id"].empty())
 			row["label_atom_id"] = row["auth_atom_id"].text();
+		else if (row["auth_atom_id"].empty())
+			row["auth_atom_id"] = row["label_atom_id"].text();

 		// Rewrite the coordinates and other items that look better in a fixed format
 		// Be careful not to nuke invalidly formatted data here
-		for (auto [item_name, prec] : std::vector<std::tuple<std::string_view, std::string::size_type>>{
+		for (auto [item_name, prec] : std::vector<std::tuple<std::string_view, int>>{
 				 { "cartn_x", 3 },
 				 { "cartn_y", 3 },
 				 { "cartn_z", 3 },
@@ -557,11 +646,11 @@ void checkAtomRecords(datablock &db)
 			if (auto [ptr, ec] = cif::from_chars(s.data(), s.data() + s.length(), v); (bool)ec)
 				continue;

-			if (s.length() < prec + 1 or s[s.length() - prec - 1] != '.')
+			if (s.length() < prec + 1UL or s[s.length() - prec - 1] != '.')
 			{
 				char b[12];

-				if (auto [ptr, ec] = cif::to_chars(b, b + sizeof(b), v, cif::chars_format::fixed, prec); (bool)ec)
+				if (auto [ptr, ec] = std::to_chars(b, b + sizeof(b), v, std::chars_format::fixed, prec); ec == std::errc{})
 					row.assign(item_name, { b, static_cast<std::string::size_type>(ptr - b) }, false, false);
 			}
 		}
@@ -603,19 +692,24 @@ void checkAtomAnisotropRecords(datablock &db)

 	std::vector<row_handle> to_be_deleted;

+	std::map<int, row_handle> atoms;
+	for (auto rh : atom_site)
+		atoms[rh.get<int>("id")] = rh;
+
 	bool warnReplaceTypeSymbol = true;
 	for (auto row : atom_site_anisotrop)
 	{
-		auto parents = atom_site_anisotrop.get_parents(row, atom_site);
-		if (parents.size() != 1)
+		auto ai = atoms.find(row.get<int>("id"));
+
+		if (ai == atoms.end())
 		{
 			to_be_deleted.emplace_back(row);
 			continue;
 		}

-		// this happens sometimes (Phenix):
+		auto parent = ai->second;

-		auto parent = parents.front();
+		// this happens sometimes (Phenix):

 		if (row["type_symbol"].empty())
 			row["type_symbol"] = parent["type_symbol"].text();
@@ -628,16 +722,14 @@ void checkAtomAnisotropRecords(datablock &db)

 		if (row["pdbx_auth_alt_id"].empty() and not parent["pdbx_auth_alt_id"].empty())
 			row["pdbx_auth_alt_id"] = parent["pdbx_auth_alt_id"].text();
-		if (row["pdbx_label_seq_id"].empty() and not parent["pdbx_label_seq_id"].empty())
+		if (row["pdbx_label_seq_id"].empty() and not parent["label_seq_id"].empty())
 			row["pdbx_label_seq_id"] = parent["label_seq_id"].text();
-		if (row["pdbx_label_asym_id"].empty() and not parent["pdbx_label_asym_id"].empty())
+		if (row["pdbx_label_asym_id"].empty() and not parent["label_asym_id"].empty())
 			row["pdbx_label_asym_id"] = parent["label_asym_id"].text();
-		if (row["pdbx_label_atom_id"].empty() and not parent["pdbx_label_atom_id"].empty())
+		if (row["pdbx_label_atom_id"].empty() and not parent["label_atom_id"].empty())
 			row["pdbx_label_atom_id"] = parent["label_atom_id"].text();
-		if (row["pdbx_label_comp_id"].empty() and not parent["pdbx_label_comp_id"].empty())
+		if (row["pdbx_label_comp_id"].empty() and not parent["label_comp_id"].empty())
 			row["pdbx_label_comp_id"] = parent["label_comp_id"].text();
-		// if (row["pdbx_PDB_model_num"].empty() and not parent["pdbx_PDB_model_num"].empty())
-		// 	row["pdbx_PDB_model_num"] = parent["pdbx_PDB_model_num"].text();
 	}

 	if (not to_be_deleted.empty())
@@ -650,23 +742,53 @@ void checkAtomAnisotropRecords(datablock &db)
 	}
 }

-void createStructAsym(datablock &db)
+void checkStructAsym(datablock &db)
 {
 	auto &atom_site = db["atom_site"];
 	auto &struct_asym = db["struct_asym"];

-	for (const auto &[label_asym_id, entity_id] : atom_site.rows<std::string, std::string>("label_asym_id", "label_entity_id"))
+	if (struct_asym.empty())
 	{
-		if (label_asym_id.empty())
-			throw std::runtime_error("File contains atom_site records without a label_asym_id");
-		if (struct_asym.count(key("id") == label_asym_id) == 0)
+		for (const auto &[label_asym_id, entity_id] : atom_site.rows<std::string, std::string>("label_asym_id", "label_entity_id"))
 		{
-			struct_asym.emplace({
-				// clang-format off
-				{ "id", label_asym_id },
-				{ "entity_id", entity_id }
-				//clang-format on
-			});
+			if (label_asym_id.empty())
+				throw std::runtime_error("File contains atom_site records without a label_asym_id");
+			if (struct_asym.count(key("id") == label_asym_id) == 0)
+			{
+				struct_asym.emplace({
+					// clang-format off
+					{ "id", label_asym_id },
+					{ "entity_id", entity_id }
+					//clang-format on
+				});
+			}
+		}
+	}
+	else
+	{
+		for (const auto &[label_asym_id, entity_id] :
+			atom_site.rows<std::string, std::string>("label_asym_id", "label_entity_id"))
+		{
+			if (label_asym_id.empty())
+				throw std::runtime_error("File contains atom_site records without a label_asym_id");
+			
+			auto sa = struct_asym.find_first(key("id") == label_asym_id);
+			if (sa)
+			{
+				if (sa["entity_id"].empty())
+					sa.assign("entity_id", entity_id, false, true);
+				else if (sa.get<std::string>("entity_id") != entity_id)
+					throw std::runtime_error("Inconsistent entity ID's in struct_asym");
+			}
+			else
+			{
+				struct_asym.emplace({
+					// clang-format off
+					{ "id", label_asym_id },
+					{ "entity_id", entity_id }
+					//clang-format on
+				});
+			}
 		}
 	}
 }
@@ -722,7 +844,7 @@ void createEntity(datablock &db)

 		std::string type, desc;
 		float weight = 0;
-		int count = 0;
+		size_t count = 0;

 		auto first_comp_id = std::get<0>(content.front());

@@ -737,8 +859,11 @@ void createEntity(datablock &db)
 			auto c = cf.create(first_comp_id);

 			type = "non-polymer";
-			desc = c->name();
-			weight = c->formula_weight();
+			if (c)
+			{
+				desc = c->name();
+				weight = c->formula_weight();
+			}
 		}
 		else
 		{
@@ -776,6 +901,7 @@ void createEntity(datablock &db)
 void createEntityPoly(datablock &db)
 {
 	using namespace literals;
+	using namespace std::literals;

 	auto &cf = compound_factory::instance();

@@ -802,34 +928,34 @@ void createEntityPoly(datablock &db)
 			auto c = cf.create(comp_id);

 			std::string letter;
-			char letter_can;
+			char letter_can{};

 			// TODO: Perhaps we should improve this...
 			if (type != "other")
 			{
 				std::string c_type;
-				if (cf.is_base(comp_id))
+				if (auto i = compound_factory::kBaseMap.find(comp_id); i != compound_factory::kBaseMap.end())
 				{
 					c_type = "polydeoxyribonucleotide";
-					letter_can = compound_factory::kBaseMap.at(comp_id);
+
+					letter_can = i->second;
+
 					if (comp_id.length() == 1)
 						letter = letter_can;
 					else
-						letter = '(' + letter_can + ')';
+						letter = '(' + comp_id + ')';
 				}
-				else if (cf.is_peptide(comp_id))
+				else if (auto i2 = compound_factory::kAAMap.find(comp_id); i2 != compound_factory::kAAMap.end())
 				{
 					c_type = "polypeptide(L)";
-					letter = letter_can = compound_factory::kAAMap.at(comp_id);
+
+					letter = letter_can = i2->second;
 				}
 				else if (iequals(c->type(), "D-PEPTIDE LINKING"))
 				{
 					c_type = "polypeptide(D)";

 					letter_can = c->one_letter_code();
-					if (letter_can == 0)
-						letter_can = 'X';
-
 					letter = '(' + comp_id + ')';

 					non_std_linkage = true;
@@ -840,9 +966,6 @@ void createEntityPoly(datablock &db)
 					c_type = "polypeptide(L)";

 					letter_can = c->one_letter_code();
-					if (letter_can == 0)
-						letter_can = 'X';
-
 					letter = '(' + comp_id + ')';

 					non_std_monomer = true;
@@ -852,9 +975,6 @@ void createEntityPoly(datablock &db)
 					// c_type = "other";

 					letter_can = c->one_letter_code();
-					if (letter_can == 0)
-						letter_can = 'X';
-
 					letter = '(' + comp_id + ')';

 					non_std_monomer = true;
@@ -867,7 +987,7 @@ void createEntityPoly(datablock &db)
 			}

 			seq[auth_asym_id] += letter;
-			seq_can[auth_asym_id] += letter_can;
+			seq_can[auth_asym_id] += letter_can ? letter_can : 'X';

 			if (find(pdb_strand_ids.begin(), pdb_strand_ids.end(), auth_asym_id) == pdb_strand_ids.end())
 				pdb_strand_ids.emplace_back(auth_asym_id);
@@ -914,7 +1034,7 @@ void createEntityPoly(datablock &db)

 		entity_poly.emplace({ //
 			{ "entity_id", entity_id },
-			{ "type", type },
+			{ "type", type.empty() ? "other"s : type },
 			{ "nstd_linkage", non_std_linkage },
 			{ "nstd_monomer", non_std_monomer },
 			{ "pdbx_seq_one_letter_code", entity_seq },
@@ -1166,7 +1286,7 @@ void createPdbxNonpolyScheme(datablock &db)
 		for (int ndb_nr = 1; auto row : atom_site.find("label_entity_id"_key == entity_id and "label_comp_id"_key == comp_id))
 		{
 			// Skip existing records
-			auto linked = atom_site.get_linked(row, pdbx_nonpoly_scheme);
+			auto linked = atom_site.get_children(row, pdbx_nonpoly_scheme);
 			if (not linked.empty())
 				continue;

@@ -1190,6 +1310,152 @@ void createPdbxNonpolyScheme(datablock &db)
 	}
 }

+void createPdbxBranchScheme(datablock &db)
+{
+	using namespace literals;
+
+	createPdbxEntityNonpoly(db);
+
+	auto &entity = db["entity"];
+	auto &pdbx_branch_scheme = db["pdbx_branch_scheme"];
+	auto &pdbx_entity_branch_list = db["pdbx_entity_branch_list"];
+	auto &atom_site = db["atom_site"];
+
+	for (const auto entity_id : entity.find<std::string>("type"_key == "branched", "id"))
+	{
+		for (const auto &[comp_id, asym_id, auth_seq_id] : atom_site.find<std::string, std::string, std::optional<int>>("label_entity_id"_key == entity_id, "label_comp_id", "label_asym_id", "auth_seq_id"))
+		{
+			if (not auth_seq_id.has_value())
+				throw std::runtime_error("Missing auth_seq_id on sugar atom");
+
+			int num = *auth_seq_id;
+
+			if (not pdbx_entity_branch_list.contains("entity_id"_key == entity_id and "num"_key == num))
+			{
+				pdbx_entity_branch_list.emplace({
+					// clang-format off
+
+					{ "entity_id", entity_id },
+					{ "comp_id", comp_id },
+					{ "num", num },
+
+					// clang-format on
+				});
+			}
+
+			if (not pdbx_branch_scheme.contains("entity_id"_key == entity_id and "asym_id"_key == asym_id and "num"_key == num))
+			{
+				pdbx_branch_scheme.emplace({
+					// clang-format off
+					{ "asym_id", asym_id },
+					{ "entity_id", entity_id },
+					{ "mon_id", comp_id },
+					{ "num", num },
+					{ "pdb_asym_id", asym_id },
+					{ "pdb_mon_id", comp_id },
+					{ "pdb_seq_num", num }
+					// clang-format on
+				});
+			}
+		}
+	}
+}
+
+void reconstruct_index_for_category(const validator &validator, category &cat, datablock &db)
+{
+	auto cv = validator.get_validator_for_category(cat.name());
+
+	enum class State
+	{
+		Start,
+		MissingKeys,
+		DuplicateKeys
+	} state = State::Start;
+
+	for (;;)
+	{
+		// See if we can build an index
+		try
+		{
+			cat.set_validator(&validator, db);
+		}
+		catch (const missing_key_error &ex)
+		{
+			if (state == State::MissingKeys)
+			{
+				if (cif::VERBOSE > 0)
+					std::clog << "Repairing failed for category " << cat.name() << ", missing keys remain: " << ex.what() << '\n';
+
+				throw;
+			}
+
+			state = State::MissingKeys;
+
+			auto key = ex.get_key();
+
+			if (cif::VERBOSE > 1)
+				std::clog << "Need to add key " << key << " to category " << cat.name() << '\n';
+
+			for (auto row : cat)
+			{
+				auto ord = row.get<std::string>(key.c_str());
+				if (ord.empty())
+					row.assign({ //
+						{ key, cat.get_unique_value(key) } });
+			}
+
+			continue;
+		}
+		catch (const duplicate_key_error &ex)
+		{
+			if (state == State::DuplicateKeys)
+			{
+				if (cif::VERBOSE > 0)
+					std::clog << "Repairing failed for category " << cat.name() << ", duplicate keys remain: " << ex.what() << '\n';
+
+				throw;
+			}
+
+			state = State::DuplicateKeys;
+
+			if (cif::VERBOSE > 0)
+				std::clog << "Attempt to fix " << cat.name() << " failed: " << ex.what() << '\n';
+
+			// replace items that do not define a relation to a parent
+
+			std::set<std::string> replaceableKeys;
+			for (auto key : cv->m_keys)
+			{
+				bool replaceable = true;
+				for (auto lv : validator.get_links_for_child(cat.name()))
+				{
+					if (find(lv->m_child_keys.begin(), lv->m_child_keys.end(), key) != lv->m_child_keys.end())
+					{
+						replaceable = false;
+						break;
+					}
+				}
+
+				if (replaceable)
+					replaceableKeys.insert(key);
+			}
+
+			if (replaceableKeys.empty())
+				throw std::runtime_error("Cannot repair category " + cat.name() + " since it contains duplicate keys that cannot be replaced");
+
+			for (auto key : replaceableKeys)
+			{
+				for (auto row : cat)
+					row.assign(key, cat.get_unique_value(key), false, false);
+			}
+
+			continue;
+		}
+
+		break;
+	}
+}
+
 bool reconstruct_pdbx(file &file)
 {
 	if (file.empty())
@@ -1241,7 +1507,7 @@ bool reconstruct_pdbx(file &file, const validator &validator)
 	checkChemCompRecords(db);

 	// If the data is really horrible, it might not contain entities
-	if (not db["atom_site"].find_first(key("label_entity_id") != null))
+	if (db["atom_site"].find_first(key("label_entity_id") == null))
 		createEntityIDs(db);

 	// Now see if atom records make sense at all
@@ -1285,7 +1551,7 @@ bool reconstruct_pdbx(file &file, const validator &validator)
 				              iv->m_type != nullptr and
 				              iv->m_type->m_primitive_type == cif::DDL_PrimitiveType::Numb;

-				for (std::size_t ix = 0; auto row : cat)
+				for (int ix = 0; auto row : cat)
 				{
 					if (number)
 						row.assign(key, std::to_string(++ix), false, false);
@@ -1356,95 +1622,7 @@ bool reconstruct_pdbx(file &file, const validator &validator)
 				}
 			}

-			enum class State
-			{
-				Start,
-				MissingKeys,
-				DuplicateKeys
-			} state = State::Start;
-
-			for (;;)
-			{
-				// See if we can build an index
-				try
-				{
-					cat.set_validator(&validator, db);
-				}
-				catch (const missing_key_error &ex)
-				{
-					if (state == State::MissingKeys)
-					{
-						if (cif::VERBOSE > 0)
-							std::clog << "Repairing failed for category " << cat.name() << ", missing keys remain: " << ex.what() << '\n';
-
-						throw;
-					}
-
-					state = State::MissingKeys;
-
-					auto key = ex.get_key();
-
-					if (cif::VERBOSE > 0)
-						std::clog << "Need to add key " << key << " to category " << cat.name() << '\n';
-
-					for (auto row : cat)
-					{
-						auto ord = row.get<std::string>(key.c_str());
-						if (ord.empty())
-							row.assign({ //
-								{ key, cat.get_unique_value(key) } });
-					}
-
-					continue;
-				}
-				catch (const duplicate_key_error &ex)
-				{
-					if (state == State::DuplicateKeys)
-					{
-						if (cif::VERBOSE > 0)
-							std::clog << "Repairing failed for category " << cat.name() << ", duplicate keys remain: " << ex.what() << '\n';
-
-						throw;
-					}
-
-					state = State::DuplicateKeys;
-
-					if (cif::VERBOSE > 0)
-						std::clog << "Attempt to fix " << cat.name() << " failed: " << ex.what() << '\n';
-
-					// replace items that do not define a relation to a parent
-
-					std::set<std::string> replaceableKeys;
-					for (auto key : cv->m_keys)
-					{
-						bool replaceable = true;
-						for (auto lv : validator.get_links_for_child(cat.name()))
-						{
-							if (find(lv->m_child_keys.begin(), lv->m_child_keys.end(), key) != lv->m_child_keys.end())
-							{
-								replaceable = false;
-								break;
-							}
-						}
-
-						if (replaceable)
-							replaceableKeys.insert(key);
-					}
-
-					if (replaceableKeys.empty())
-						throw std::runtime_error("Cannot repair category " + cat.name() + " since it contains duplicate keys that cannot be replaced");
-
-					for (auto key : replaceableKeys)
-					{
-						for (auto row : cat)
-							row.assign(key, cat.get_unique_value(key), false, false);
-					}
-
-					continue;
-				}
-
-				break;
-			}
+			reconstruct_index_for_category(validator, cat, db);
 		}
 		catch (const std::exception &ex)
 		{
@@ -1473,24 +1651,26 @@ bool reconstruct_pdbx(file &file, const validator &validator)
 		checkAtomAnisotropRecords(db);

 	// Now create any missing categories
-	// Next make sure we have struct_asym records
-	if (auto cat = db.get("struct_asym"); cat == nullptr or cat->empty())
-		createStructAsym(db);
+	// Next make sure we have good struct_asym records
+	checkStructAsym(db);

 	if (auto cat = db.get("entity"); cat == nullptr or cat->empty())
 		createEntity(db);

-	// fill in missing formula_weight, e.g.
-	checkEntities(db);
-
 	if (auto cat = db.get("pdbx_poly_seq_scheme"); cat == nullptr or cat->empty())
 		createPdbxPolySeqScheme(db);

 	if (auto cat = db.get("ndb_poly_seq_scheme"); cat == nullptr or cat->empty())
 		comparePolySeqSchemes(db);
-	
+
 	createPdbxNonpolyScheme(db);

+	// Create a minimal set of branch records
+	createPdbxBranchScheme(db);
+
+	// fill in missing formula_weight, e.g.
+	checkEntities(db);
+
 	// skip unknown categories for now
 	bool valid = true;
 	for (auto &cat : db)
@@ -1499,4 +1679,110 @@ bool reconstruct_pdbx(file &file, const validator &validator)
 	return valid and is_valid_pdbx_file(file, validator);
 }

+// --------------------------------------------------------------------
+
+void fixup_pdbx(file &file)
+{
+	if (file.empty())
+		throw std::runtime_error("Cannot reconstruct PDBx, file seems to be empty");
+
+	auto &db = file.front();
+
+	if (auto ac = db.get("audit_conform"); ac != nullptr)
+		fixup_pdbx(file, validator_factory::instance().get(*ac));
+	else
+		fixup_pdbx(file, validator_factory::instance().get("mmcif_pdbx.dic"));
+}
+
+void fixup_pdbx(file &file, const validator &validator)
+{
+	if (file.empty())
+		throw std::runtime_error("Cannot reconstruct PDBx, file seems to be empty");
+
+	// assuming the first datablock contains the entry ...
+	auto &db = file.front();
+
+	if (auto cat = db.get("atom_site"); cat == nullptr or cat->empty())
+		throw std::runtime_error("Cannot reconstruct PDBx file, atom data missing");
+
+	// ... and any additional datablock will contain compound information
+	cif::compound_source cs(file);
+
+	// Be silent about missing compound info in fixup
+	auto &cf = compound_factory::instance();
+	bool save_report = cf.get_report_missing();
+	cf.set_report_missing(cif::VERBOSE > 1);
+
+	std::string entry_id;
+
+	// Phenix files do not have an entry record
+	if (auto cat = db.get("entry"); cat == nullptr or cat->empty())
+	{
+		entry_id = db.name();
+		category entry("entry");
+		entry.emplace({ { "id", entry_id } });
+		db.emplace_back(std::move(entry));
+	}
+	else
+	{
+		auto &entry = db["entry"];
+		if (entry.size() != 1)
+			throw std::runtime_error("Unexpected size of entry category");
+
+		entry_id = entry.front().get<std::string>("id");
+	}
+
+	// Start with chem_comp, it is often missing many fields
+	// that can easily be filled in.
+	checkChemCompRecords(db);
+
+	// If the data is really horrible, it might not contain entities
+	if (not db["atom_site"].find_first(key("label_entity_id") != null))
+		createEntityIDs(db);
+
+	// Now see if atom records make sense at all, but in a silent way, this time
+	checkAtomRecords(db);
+
+	db["chem_comp"].reorder_by_index();
+
+	// See if we can easily reconstruct missing data fields in order to create an index
+	for (auto &cat : db)
+	{
+		try
+		{
+			cat.set_validator(&validator, db);
+		}
+		catch (const missing_key_error &)
+		{
+			reconstruct_index_for_category(validator, cat, db);
+		}
+	}
+
+	db.set_validator(&validator);
+
+	// Now create any missing categories
+	// Next make sure we have good struct_asym records
+	checkStructAsym(db);
+
+	if (auto cat = db.get("entity"); cat == nullptr or cat->empty())
+		createEntity(db);
+
+	if (auto cat = db.get("pdbx_poly_seq_scheme"); cat == nullptr or cat->empty())
+		createPdbxPolySeqScheme(db);
+
+	if (auto cat = db.get("ndb_poly_seq_scheme"); cat == nullptr or cat->empty())
+		comparePolySeqSchemes(db);
+
+	createPdbxNonpolyScheme(db);
+
+	// Create a minimal set of branch records
+	createPdbxBranchScheme(db);
+
+	// fill in missing formula_weight, e.g.
+	checkEntities(db);
+
+	// That's it
+	cf.set_report_missing(save_report);
+}
+
 } // namespace cif::pdb
--- a/src/pdb/validate-pdbx.cpp
+++ b/src/pdb/validate-pdbx.cpp
@@ -61,8 +61,6 @@ condition get_parents_condition(const validator &validator, row_handle rh, const
 			result = std::move(result) or std::move(cond);
 		}
 	}
-	else if (cif::VERBOSE > 0)
-		std::cerr << "warning: no child to parent links were found for child " << childName << " and parent " << parentName << '\n';

 	return result;
 }
@@ -71,7 +69,7 @@ bool is_valid_pdbx_file(const file &file, const validator &v)
 {
 	std::error_code ec;
 	bool result = is_valid_pdbx_file(file, v, ec);
-	return result and not (bool)ec;
+	return result and not(bool) ec;
 }

 bool is_valid_pdbx_file(const file &file, std::error_code &ec)
@@ -84,7 +82,7 @@ bool is_valid_pdbx_file(const file &file, std::error_code &ec)
 		result = is_valid_pdbx_file(file, validator_factory::instance().get(*ac), ec);
 	else
 		result = is_valid_pdbx_file(file, validator_factory::instance().get("mmcif_pdbx.dic"), ec);
-	
+
 	return result;
 }

@@ -92,7 +90,7 @@ bool is_valid_pdbx_file(const file &file, const validator &validator, std::error
 {
 	using namespace cif::literals;

-	bool result = true;
+	bool result = true, warned_missing_parents = false;

 	try
 	{
@@ -129,10 +127,18 @@ bool is_valid_pdbx_file(const file &file, const validator &validator, std::error
 			if (not cf.is_monomer(comp_id))
 				continue;

-			auto p = pdbx_poly_seq_scheme.find(get_parents_condition(validator, r, pdbx_poly_seq_scheme));
+			auto cond = get_parents_condition(validator, r, pdbx_poly_seq_scheme);
+			if (not cond)
+			{
+				if (VERBOSE > 0 and std::exchange(warned_missing_parents, true) == false)
+					std::cerr << "warning: missing links for atom_site/pdbx_poly_seq_scheme\n";
+				continue;
+			}
+
+			auto p = pdbx_poly_seq_scheme.find(std::move(cond));
 			if (p.size() != 1)
 			{
-				if (cif::VERBOSE > 0)
+				if (VERBOSE > 0)
 					std::clog << "In atom_site record: " << r["id"].text() << '\n';
 				throw std::runtime_error("For each monomer in atom_site there should be exactly one pdbx_poly_seq_scheme record");
 			}
@@ -161,7 +167,7 @@ bool is_valid_pdbx_file(const file &file, const validator &validator, std::error

 			const auto entity_poly_type = entity_poly.find1<std::string>("entity_id"_key == entity_id, "type");

-			std::map<int,std::set<std::string>> mon_per_seq_id;
+			std::map<int, std::set<std::string>> mon_per_seq_id;

 			for (const auto &[num, mon_id, hetero] : entity_poly_seq.find<int, std::string, bool>("entity_id"_key == entity_id, "num", "mon_id", "hetero"))
 			{
@@ -196,28 +202,37 @@ bool is_valid_pdbx_file(const file &file, const validator &validator, std::error
 					throw std::runtime_error("Mismatch between the hetero flag in the poly seq schemes and the number residues per seq_id");
 			}

-			for (const auto &[seq_id, mon_ids] : mon_per_seq_id)
-			{
-				for (auto asym_id : struct_asym.find<std::string>("entity_id"_key == entity_id, "id"))
-				{
-					condition cond;
-					
-					for (auto mon_id : mon_ids)
-						cond = std::move(cond) or "label_comp_id"_key == mon_id;
+			// This code proved to take too much time ...

-					cond = "label_entity_id"_key == entity_id and
-						"label_asym_id"_key == asym_id and
-						"label_seq_id"_key == seq_id and not std::move(cond);
-					
-					if (atom_site.contains(std::move(cond)))
-						throw std::runtime_error("An atom_site record exists that has no parent in the poly seq scheme categories");
-				}
+			// for (const auto &[seq_id, mon_ids] : mon_per_seq_id)
+			// {
+			// 	for (auto asym_id : struct_asym.find<std::string>("entity_id"_key == entity_id, "id"))
+			// 	{
+			// 		condition cond;
+
+			// 		for (auto mon_id : mon_ids)
+			// 			cond = std::move(cond) or "label_comp_id"_key == mon_id;
+
+			// 		cond = "label_entity_id"_key == entity_id and
+			// 			"label_asym_id"_key == asym_id and
+			// 			"label_seq_id"_key == seq_id and not std::move(cond);
+
+			// 		if (atom_site.contains(std::move(cond)))
+			// 			throw std::runtime_error("An atom_site record exists that has no parent in the poly seq scheme categories");
+			// 	}
+			// }
+
+			// ... so we're using this instead, should be almost the same...
+
+			for (const auto &[comp_id, seq_id] :
+				atom_site.find<std::string, int>("label_entity_id"_key == entity_id, "label_comp_id", "label_seq_id"))
+			{
+				if (not mon_per_seq_id[seq_id].contains(comp_id))
+					throw std::runtime_error("An atom_site record exists that has no parent in the poly seq scheme categories");
 			}

 			auto &&[seq, seq_can] = entity_poly.find1<std::optional<std::string>, std::optional<std::string>>("entity_id"_key == entity_id,
 				"pdbx_seq_one_letter_code", "pdbx_seq_one_letter_code_can");
-			
-			std::string::const_iterator si, sci, se, sce;

 			auto seq_match = [&](bool can, std::string::const_iterator si, std::string::const_iterator se)
 			{
@@ -254,8 +269,8 @@ bool is_valid_pdbx_file(const file &file, const validator &validator, std::error
 							else
 								letter = '(' + comp_id + ')';
 						}
-						
-						if (iequals(std::string{si, si + letter.length()}, letter))
+
+						if (iequals(std::string{ si, si + letter.length() }, letter))
 						{
 							match = true;
 							si += letter.length();
@@ -274,12 +289,14 @@ bool is_valid_pdbx_file(const file &file, const validator &validator, std::error

 			if (not seq.has_value())
 			{
-				if (cif::VERBOSE > 0)
+				if (VERBOSE > 0)
 					std::clog << "Warning: entity_poly has no sequence for entity_id " << entity_id << '\n';
 			}
 			else
 			{
-				seq->erase(std::remove_if(seq->begin(), seq->end(), [](char ch) { return std::isspace(ch); }), seq->end());
+				seq->erase(std::remove_if(seq->begin(), seq->end(), [](char ch)
+							   { return std::isspace(ch); }),
+					seq->end());

 				if (not seq_match(false, seq->begin(), seq->end()))
 					throw std::runtime_error("Sequences do not match for entity " + entity_id);
@@ -287,12 +304,14 @@ bool is_valid_pdbx_file(const file &file, const validator &validator, std::error

 			if (not seq_can.has_value())
 			{
-				if (cif::VERBOSE > 1)
+				if (VERBOSE > 1)
 					std::clog << "Warning: entity_poly has no canonical sequence for entity_id " << entity_id << '\n';
 			}
 			else
 			{
-				seq_can->erase(std::remove_if(seq_can->begin(), seq_can->end(), [](char ch) { return std::isspace(ch); }), seq_can->end());
+				seq_can->erase(std::remove_if(seq_can->begin(), seq_can->end(), [](char ch)
+								   { return std::isspace(ch); }),
+					seq_can->end());

 				if (not seq_match(true, seq_can->begin(), seq_can->end()))
 					throw std::runtime_error("Canonical sequences do not match for entity " + entity_id);
@@ -304,16 +323,15 @@ bool is_valid_pdbx_file(const file &file, const validator &validator, std::error
 	catch (const std::exception &ex)
 	{
 		result = false;
-		if (cif::VERBOSE > 0)
+		if (VERBOSE > 0)
 			std::clog << ex.what() << '\n';
 		ec = make_error_code(validation_error::not_valid_pdbx);
 	}

-	if (not result and (bool)ec)
+	if (not result and (bool) ec)
 		ec = make_error_code(validation_error::not_valid_pdbx);

 	return result;
 }

 } // namespace cif::pdb
-  
--- a/src/point.cpp
+++ b/src/point.cpp
@@ -25,17 +25,19 @@
 */

 #include "cif++/point.hpp"
-#include "cif++/matrix.hpp"

-#include <cassert>
-#include <random>
+#include "cif++/matrix.hpp" // for matrix_subtraction, matrix_cofactors
+
+#include <initializer_list>
+#include <random> // for uniform_real_distribution, normal_distri...
+#include <stdexcept>

 namespace cif
 {

 // --------------------------------------------------------------------

-template<typename T>
+template <typename T>
 quaternion_type<T> normalize(quaternion_type<T> q)
 {
 	std::valarray<double> t(4);
@@ -126,10 +128,9 @@ quaternion construct_for_dihedral_angle(point p1, point p2, point p3, point p4,
 	p4 -= p3;
 	p3 -= p3;

-	quaternion q;
 	auto axis = -p2;
-
 	float dh = dihedral_angle(p1, p2, p3, p4);
+
 	return construct_from_angle_axis(angle - dh, axis);
 }

@@ -293,9 +294,9 @@ quaternion align_points(const std::vector<point> &pa, const std::vector<point> &
 	}

 	quaternion q(
-		static_cast<float>(cf(maxR, 0)), 
-		static_cast<float>(cf(maxR, 1)), 
-		static_cast<float>(cf(maxR, 2)), 
+		static_cast<float>(cf(maxR, 0)),
+		static_cast<float>(cf(maxR, 1)),
+		static_cast<float>(cf(maxR, 2)),
 		static_cast<float>(cf(maxR, 3)));
 	q = normalize(q);

@@ -327,4 +328,251 @@ point nudge(point p, float offset)
 	return p + r;
 }

+// --------------------------------------------------------------------
+
+std::tuple<point, float> smallest_sphere_around_2_points(std::array<cif::point, 2> pts)
+{
+	return { (pts[0] + pts[1]) / 2, distance(pts[0], pts[1]) / 2 };
+}
+
+std::tuple<point, float> smallest_sphere_around_3_points(std::array<cif::point, 3> pts)
+{
+	// Find two bisectors
+	auto vz = cross_product(pts[1] - pts[0], pts[2] - pts[0]);
+
+	auto bs1 = cross_product(vz, pts[1] - pts[0]);
+	bs1.normalize();
+
+	auto v1 = (pts[1] - pts[0]);
+	v1.normalize();
+
+	auto s1 = pts[0] + (distance(pts[1], pts[0]) / 2) * v1;
+
+	auto bs2 = cross_product(vz, pts[2] - pts[0]);
+	bs2.normalize();
+
+	auto v2 = (pts[2] - pts[0]);
+	v2.normalize();
+
+	auto s2 = pts[0] + (distance(pts[2], pts[0]) / 2) * v2;
+
+	auto c = line_line_intersection(s1, s1 + bs1, s2, s2 + bs2);
+	if (c)
+		return { *c, distance(*c, pts[0]) };
+
+	// Colinear points I guess, try something else
+	auto l1 = distance_squared(pts[0], pts[1]);
+	auto l2 = distance_squared(pts[0], pts[2]);
+	auto l3 = distance_squared(pts[1], pts[2]);
+
+	if (l1 > l2 and l1 > l3)
+		return smallest_sphere_around_2_points({ pts[0], pts[1] });
+	else if (l2 > l1 and l2 > l3)
+		return smallest_sphere_around_2_points({ pts[0], pts[2] });
+	else
+		return smallest_sphere_around_2_points({ pts[1], pts[2] });
+}
+
+std::tuple<point, float> smallest_sphere_around_4_points(std::array<cif::point, 4> pts)
+{
+	auto t0 = -norm_squared(pts[0]);
+	auto t1 = -norm_squared(pts[1]);
+	auto t2 = -norm_squared(pts[2]);
+	auto t3 = -norm_squared(pts[3]);
+
+	// clang-format off
+	matrix4x4<float> Tm({
+		pts[0].m_x, pts[0].m_y, pts[0].m_z, 1,
+		pts[1].m_x, pts[1].m_y, pts[1].m_z, 1,
+		pts[2].m_x, pts[2].m_y, pts[2].m_z, 1,
+		pts[3].m_x, pts[3].m_y, pts[3].m_z, 1
+	});
+	auto T = determinant(Tm);
+
+	if (T != 0)
+	{
+		matrix4x4<float> Dm({
+			t0, pts[0].m_y, pts[0].m_z, 1,
+			t1, pts[1].m_y, pts[1].m_z, 1,
+			t2, pts[2].m_y, pts[2].m_z, 1,
+			t3, pts[3].m_y, pts[3].m_z, 1
+		});
+		auto D = determinant(Dm) / T;
+		
+		matrix4x4<float> Em({
+			pts[0].m_x, t0, pts[0].m_z, 1,
+			pts[1].m_x, t1, pts[1].m_z, 1,
+			pts[2].m_x, t2, pts[2].m_z, 1,
+			pts[3].m_x, t3, pts[3].m_z, 1
+		});
+		auto E = determinant(Em) / T;
+		
+		matrix4x4<float> Fm({
+			pts[0].m_x, pts[0].m_y, t0, 1,
+			pts[1].m_x, pts[1].m_y, t1, 1,
+			pts[2].m_x, pts[2].m_y, t2, 1,
+			pts[3].m_x, pts[3].m_y, t3, 1
+		});
+		
+		auto F = determinant(Fm) / T;
+		
+		matrix4x4<float> Gm({
+			pts[0].m_x, pts[0].m_y, pts[0].m_z, t0,
+			pts[1].m_x, pts[1].m_y, pts[1].m_z, t1,
+			pts[2].m_x, pts[2].m_y, pts[2].m_z, t2,
+			pts[3].m_x, pts[3].m_y, pts[3].m_z, t3
+		});
+		auto G = determinant(Gm) / T;
+		
+		point center{ -D / 2, -E / 2, -F / 2 };
+		float radius = std::sqrt(D * D + E * E + F * F - 4 * G) / 2;
+
+		// clang-format on
+
+		return { center, radius };
+	}
+
+	// Perhaps some colinear points, try something else:
+
+	for (auto ix : std::initializer_list<std::array<size_t, 4>>{
+			 { 1, 2, 3, 0 },
+			 { 0, 2, 3, 1 },
+			 { 0, 1, 3, 2 },
+			 { 0, 1, 2, 3 },
+		 })
+	{
+		auto [center, radius] =
+			smallest_sphere_around_3_points({ pts[ix[0]], pts[ix[1]], pts[ix[2]] });
+
+		if (distance(pts[ix[3]], center) <= radius)
+			return { center, radius };
+	}
+
+	assert(false);
+	exit(1);
+}
+
+std::tuple<point, float> smallest_sphere_around_all_points(std::vector<point> P, std::vector<point> R)
+{
+	if (P.empty() or R.size() == 4)
+	{
+		switch (R.size())
+		{
+			case 1:
+				return { R[0], 0 };
+
+			case 2:
+				return smallest_sphere_around_2_points({ R[0], R[1] });
+
+			case 3:
+				return smallest_sphere_around_3_points({ R[0], R[1], R[2] });
+
+			case 4:
+				return smallest_sphere_around_4_points({ R[0], R[1], R[2], R[3] });
+
+			default:
+				assert(false);
+		}
+	}
+
+	auto p = P.back();
+	P.pop_back();
+
+	auto [c, r] = smallest_sphere_around_all_points(P, R);
+	assert(not std::isnan(r));
+	if (distance(c, p) <= r)
+		return { c, r };
+
+	R.emplace_back(p);
+	return smallest_sphere_around_all_points(P, R);
+}
+
+bool point_in_circle(point p, std::vector<point> c)
+{
+	switch (c.size())
+	{
+		case 0:
+			return false;
+
+		case 1:
+			return p == c.front();
+
+		case 2:
+		{
+			auto [center, radius] = smallest_sphere_around_2_points({ c[0], c[1] });
+			return cif::distance_squared(p, center) <= radius * radius;
+		}
+
+		case 3:
+		{
+			auto [center, radius] = smallest_sphere_around_3_points({ c[0], c[1], c[2] });
+			return cif::distance_squared(p, center) <= radius * radius;
+		}
+
+		case 4:
+		{
+			auto [center, radius] = smallest_sphere_around_4_points({ c[0], c[1], c[2], c[3] });
+			return cif::distance_squared(p, center) <= radius * radius;
+		}
+
+		default:
+			assert(false);
+			throw std::runtime_error("Error finding smallest sphere");
+	}
+}
+
+std::tuple<point, float> smallest_sphere_around_points(std::vector<point> pts)
+{
+	std::random_device rd;
+	std::mt19937 g(rd());
+
+	std::shuffle(pts.begin(), pts.end(), g);
+
+	std::vector<size_t> cix;
+
+	auto cirle_points = [&]()
+	{
+		std::vector<point> result;
+		for (auto ix : cix)
+			result.emplace_back(pts[ix]);
+		return result;
+	};
+
+	size_t i = 0;
+	while (i < pts.size())
+	{
+		if (std::find(cix.begin(), cix.end(), i) != cix.end() or
+			point_in_circle(pts[i], cirle_points()))
+		{
+			++i;
+		}
+		else
+		{
+			cix.erase(std::remove_if(cix.begin(), cix.end(), [i](size_t j)
+						  { return j < i; }),
+				cix.end());
+			cix.push_back(i);
+			if (cix.size() < 4)
+				i = 0;
+			else
+				++i;
+		}
+	}
+
+	switch (cix.size())
+	{
+		case 1:
+			return { pts[cix[0]], 0 };
+		case 2:
+			return smallest_sphere_around_2_points({ pts[cix[0]], pts[cix[1]] });
+		case 3:
+			return smallest_sphere_around_3_points({ pts[cix[0]], pts[cix[1]], pts[cix[2]] });
+		case 4:
+			return smallest_sphere_around_4_points({ pts[cix[0]], pts[cix[1]], pts[cix[2]], pts[cix[3]] });
+		default:
+			assert(false);
+			throw std::runtime_error("Error finding smallest sphere");
+	}
+}
+
 } // namespace cif
--- a/src/symmetry.cpp
+++ b/src/symmetry.cpp
@@ -32,6 +32,11 @@

 #include "symop_table_data.hpp"

+#if defined(_MSC_VER)
+#pragma warning (disable : 5054)	// warning C5054: operator '&': deprecated between enumerations of different types
+#pragma warning (disable : 4127)	// conditional expression is constant
+#endif
+
 #include <Eigen/Eigen>

 namespace cif
@@ -90,10 +95,10 @@ float cell::get_volume() const
 	auto cos_beta = std::cos(beta);
 	auto cos_gamma = std::cos(gamma);

-	auto vol = m_a * m_b * m_c;
+	double vol = m_a * m_b * m_c;
 	vol *= std::sqrt(1.0f - cos_alpha * cos_alpha - cos_beta * cos_beta - cos_gamma * cos_gamma + 2.0f * cos_alpha * cos_beta * cos_gamma);

-	return vol;
+	return static_cast<float>(vol);
 }

 // --------------------------------------------------------------------
--- a/src/text.cpp
+++ b/src/text.cpp
@@ -28,6 +28,11 @@

 #include <algorithm>
 #include <cassert>
+#include <charconv>
+
+#if __has_include("fast_float/fast_float.h")
+#include "fast_float/fast_float.h"
+#endif

 namespace cif
 {
@@ -512,4 +517,32 @@ std::vector<std::string> word_wrap(const std::string &text, std::size_t width)
 	return result;
 }

+#if __has_include("fast_float/fast_float.h")
+
+template <typename T>
+std::from_chars_result ff_charconv<T, typename std::enable_if_t<std::is_floating_point_v<T>>>::from_chars(const char *a, const char *b, T &v)
+{
+	auto r = fast_float::from_chars(a, b, v);
+	return { r.ptr, r.ec };
+}
+
+template struct ff_charconv<float>;
+template struct ff_charconv<double>;
+// template struct ff_charconv<long double>;
+
+#ifdef __STDCPP_FLOAT64_T__
+template struct ff_charconv<std::float64_t>;
+#endif
+#ifdef __STDCPP_FLOAT32_T__
+template struct ff_charconv<std::float32_t>;
+#endif
+#ifdef __STDCPP_FLOAT16_T__
+template struct ff_charconv<std::float16_t>;
+#endif
+#ifdef __STDCPP_BFLOAT16_T__
+template struct ff_charconv<std::bfloat16_t>;
+#endif
+
+#endif
+
 } // namespace cif
--- a/src/utilities.cpp
+++ b/src/utilities.cpp
@@ -34,14 +34,20 @@
 #include <condition_variable>
 #include <cstring>
 #include <deque>
+#include <format>
 #include <fstream>
-#include <functional>
 #include <iomanip>
 #include <iostream>
 #include <map>
 #include <mutex>
 #include <sstream>
+#include <string>
 #include <thread>
+#include <utility>
+
+#if __cpp_lib_jthread >= 201911L
+#include <stop_token>
+#endif

 namespace fs = std::filesystem;

@@ -65,27 +71,50 @@ std::string get_version_nr()

 #if defined(_WIN32) or defined(__MINGW32__)
 }
-#include <windows.h>
-#include <libloaderapi.h>
-#include <wincon.h>
-
-#include <codecvt>
+// clang-format off
+# include <windows.h>
+# include <libloaderapi.h>
+# include <wincon.h>
+// clang-format on

 namespace cif
 {

 uint32_t get_terminal_width()
 {
-    CONSOLE_SCREEN_BUFFER_INFO csbi;
-    ::GetConsoleScreenBufferInfo(::GetStdHandle(STD_OUTPUT_HANDLE), &csbi);
-    return csbi.srWindow.Right - csbi.srWindow.Left + 1;
+	CONSOLE_SCREEN_BUFFER_INFO csbi;
+	return ::GetConsoleScreenBufferInfo(::GetStdHandle(STD_OUTPUT_HANDLE), &csbi)
+	           ? csbi.srWindow.Right - csbi.srWindow.Left + 1
+	           : 80;
+}
+
+void write_to_console(const std::string &s)
+{
+	auto h = ::GetStdHandle(STD_OUTPUT_HANDLE);
+	CONSOLE_SCREEN_BUFFER_INFO csbi;
+
+	if (auto l = ::MultiByteToWideChar(CP_UTF8, 0, s.data(), s.length(), nullptr, 0);
+		l > 0 and ::GetConsoleScreenBufferInfo(::GetStdHandle(STD_OUTPUT_HANDLE), &csbi))
+	{
+		std::u16string ws(l, 0);
+
+		::MultiByteToWideChar(CP_UTF8, 0, s.data(), s.length(), (LPWSTR)ws.data(), l);
+
+		DWORD w;
+		::WriteConsoleW(h, ws.data(), ws.length(), &w, nullptr);
+	}
+	else
+	{
+		std::cout.write(s.data(), s.length());
+		std::cout.flush();
+	}
 }

 #else

-#include <sys/ioctl.h>
-#include <termios.h>
-#include <limits.h>
+# include <limits.h>
+# include <sys/ioctl.h>
+# include <termios.h>

 uint32_t get_terminal_width()
 {
@@ -100,59 +129,220 @@ uint32_t get_terminal_width()
 	return result;
 }

+inline void write_to_console(const std::string &s)
+{
+	std::cout << s << std::flush;
+}
+
 #endif

 // --------------------------------------------------------------------

 struct progress_bar_impl
 {
-	progress_bar_impl(int64_t inMax, const std::string &inAction)
-		: m_max_value(inMax)
+	progress_bar_impl(uint64_t max_value, const std::string &message)
+		: m_max_value(max_value)
 		, m_consumed(0)
-		, m_action(inAction)
-		, m_message(inAction)
-		, m_thread(std::bind(&progress_bar_impl::run, this))
+		, m_action(message)
+		, m_message(message)
 	{
 	}

-	progress_bar_impl(const progress_bar_impl&) = delete;
-	progress_bar_impl &operator=(const progress_bar_impl &) = delete;
+	virtual ~progress_bar_impl() {}

-	~progress_bar_impl();
-
-	void run();
-
-	void consumed(int64_t n);
-	void progress(int64_t p);
-	void message(const std::string &msg);
-
-	void print_progress();
-	void print_done();
+	virtual void consumed(uint64_t n);
+	virtual void progress(uint64_t p);
+	virtual void message(const std::string &msg);
+	virtual void print_done();

 	using time_point = std::chrono::time_point<std::chrono::system_clock>;

-	int64_t m_max_value;
-	std::atomic<int64_t> m_consumed;
-	int64_t m_last_consumed = 0;
-	int m_spinner_index = 0;
+	uint64_t m_max_value;
+	std::atomic<uint64_t> m_consumed;
 	std::string m_action, m_message;
-	std::mutex m_mutex;
-	std::thread m_thread;
 	time_point m_start = std::chrono::system_clock::now();
-	time_point m_last = std::chrono::system_clock::now();
-	bool m_stop = false;
 };

-progress_bar_impl::~progress_bar_impl()
+void progress_bar_impl::consumed(uint64_t n)
+{
+	m_consumed += n;
+}
+
+void progress_bar_impl::progress(uint64_t p)
+{
+	m_consumed = p;
+}
+
+void progress_bar_impl::message(const std::string &msg)
+{
+	m_message = msg;
+}
+
+void progress_bar_impl::print_done()
+{
+	std::chrono::duration<double> elapsed = std::chrono::system_clock::now() - m_start;
+	std::string days, hours, minutes, seconds;
+
+	uint64_t s = static_cast<uint64_t>(std::trunc(elapsed.count()));
+	if (s > 24 * 60 * 60)
+	{
+		days = std::format("{:d}d ", s / (24 * 60 * 60));
+		s %= 24 * 60 * 60;
+	}
+
+	if (s > 60 * 60)
+	{
+		hours = std::format("{:2d}h ", s / (60 * 60));
+		s %= 60 * 60;
+	}
+
+	if (s > 60)
+	{
+		minutes = std::format("{:2d}m ", s / 60);
+		s %= 60;
+	}
+
+	std::string msg = std::format("{} done in {}{}{}{:.1f}s", m_action, days, hours, minutes, s + 1e-6 * (elapsed.count() - s));
+
+	uint32_t width = get_terminal_width();
+
+	if (msg.length() < width)
+		msg += std::string(width - msg.length(), ' ');
+
+	write_to_console(msg += '\n');
+}
+
+// --------------------------------------------------------------------
+
+struct simple_progress_bar_impl : public progress_bar_impl
+{
+	simple_progress_bar_impl(uint64_t max_value, const std::string &message)
+		: progress_bar_impl(max_value, message)
+	{
+	}
+
+	~simple_progress_bar_impl()
+	{
+		if (m_printed_any)
+			print_done();
+	}
+
+	void consumed(uint64_t n) override
+	{
+		using namespace std::literals;
+
+		progress_bar_impl::consumed(n);
+
+		// print at most 10 steps, but only if it took long enough
+
+		int percentile = static_cast<int>(std::floor(10.f * m_consumed / m_max_value));
+		if (percentile > m_last_percentile and (m_printed_any or std::chrono::system_clock::now() - m_start >= 1s))
+		{
+			if (not std::exchange(m_printed_any, true))
+				write_to_console(m_action + ": ");
+
+			write_to_console(std::format("...{:d}0%", percentile));
+			m_last_percentile = percentile;
+		}
+	}
+
+	void progress(uint64_t p) override
+	{
+		consumed(p - m_consumed);
+	}
+
+	void print_done() override
+	{
+		if (m_printed_any)
+		{
+			write_to_console("\n");
+			progress_bar_impl::print_done();
+		}
+	}
+
+	bool m_printed_any = false;
+	int m_last_percentile = 0;
+};
+
+// --------------------------------------------------------------------
+
+struct fancy_progress_bar_impl : public progress_bar_impl
+{
+	fancy_progress_bar_impl(uint64_t max_value, const std::string &message)
+		: progress_bar_impl(max_value, message)
+		, m_thread(
+#if __cpp_lib_jthread >= 201911L
+			  [this](std::stop_token stoken)
+			  { this->run(stoken); }
+#else
+			  [this]()
+			  { this->run(); }
+#endif
+		  )
+	{
+	}
+
+	~fancy_progress_bar_impl();
+
+#if __cpp_lib_jthread >= 201911L
+	void run(std::stop_token stoken);
+#else
+	void run();
+#endif
+
+	void consumed(uint64_t n) override;
+	void progress(uint64_t p) override;
+	void message(const std::string &msg) override;
+
+	void print_progress();
+
+	std::mutex m_mutex;
+	std::condition_variable m_cv;
+
+	float m_progress;
+	uint32_t m_width, m_bar_width;
+	uint32_t m_steps, m_last_steps = 0;
+	uint64_t m_last_consumed = 0;
+#if __cpp_lib_jthread >= 201911L
+	std::jthread m_thread;
+#else
+	std::thread m_thread;
+	bool m_stop = false;
+#endif
+};
+
+const char *kBlocks[] = {
+	" ",
+	"▏",
+	"▎",
+	"▍",
+	"▌",
+	"▋",
+	"▊",
+	"▉",
+	"█",
+};
+
+const size_t kBlockCount = sizeof(kBlocks) / sizeof(void *) - 1;
+
+fancy_progress_bar_impl::~fancy_progress_bar_impl()
 {
 	using namespace std::literals;
 	assert(m_thread.joinable());

+#if __cpp_lib_jthread >= 201911L
+	m_thread.request_stop();
+#else
 	m_stop = true;
+#endif
 	m_thread.join();
 }

-void progress_bar_impl::run()
+#if __cpp_lib_jthread >= 201911L
+void fancy_progress_bar_impl::run(std::stop_token stoken)
+#else
+void fancy_progress_bar_impl::run()
+#endif
 {
 	using namespace std::literals;

@@ -160,25 +350,44 @@ void progress_bar_impl::run()

 	try
 	{
-		while (not m_stop)
+		for (;;)
 		{
+			std::unique_lock lock(m_mutex);
+
+			m_cv.wait_for(lock, 25ms);
+
+#if __cpp_lib_jthread >= 201911L
+			if (stoken.stop_requested())
+				break;
+#else
+			if (m_stop)
+				break;
+#endif
+
 			auto now = std::chrono::system_clock::now();

-			if (now - m_start < 2s or now - m_last < 100ms)
-			{
-				std::this_thread::sleep_for(10ms);
+			if (m_consumed == m_last_consumed or now - m_start < 1s)
 				continue;
-			}

-			std::lock_guard lock(m_mutex);
+			m_last_consumed = m_consumed;

-			if (not printedAny and isatty(STDOUT_FILENO))
-				std::cout << "\x1b[?25l";
+			// See if we need to do work
+			m_width = get_terminal_width();
+			m_progress = static_cast<float>(m_consumed) / m_max_value;
+			m_bar_width = 7 * m_width / 10; // 70% of the width of the terminal
+			m_steps = static_cast<uint32_t>(std::ceil(m_progress * m_bar_width * kBlockCount));
+
+			if (m_steps == m_last_steps)
+				continue;
+
+			m_last_steps = m_steps;
+
+			if (not printedAny)
+				write_to_console("\x1b[?25l");

 			print_progress();

 			printedAny = true;
-			m_last = std::chrono::system_clock::now();
 		}
 	}
 	catch (...)
@@ -187,161 +396,94 @@ void progress_bar_impl::run()

 	if (printedAny)
 	{
+		write_to_console("\r\x1b[?25h");
 		print_done();
-		if (isatty(STDOUT_FILENO))
-			std::cout << "\x1b[?25h";
 	}
 }

-void progress_bar_impl::consumed(int64_t n)
+void fancy_progress_bar_impl::consumed(uint64_t n)
 {
-	m_consumed += n;
+	progress_bar_impl::consumed(n);
+	// m_cv.notify_one();
 }

-void progress_bar_impl::progress(int64_t p)
+void fancy_progress_bar_impl::progress(uint64_t p)
 {
-	m_consumed = p;
+	progress_bar_impl::progress(p);
+	// m_cv.notify_one();
 }

-void progress_bar_impl::message(const std::string &msg)
+void fancy_progress_bar_impl::message(const std::string &msg)
 {
 	std::unique_lock lock(m_mutex);
-	m_message = msg;
+	progress_bar_impl::message(msg);
+	// m_cv.notify_one();
 }

-const char* kSpinner[] = {
-	// ".", "o", "O", "0", "O", "o", ".", " "
-	// "⢄", "⢂", "⢁", "⡁", "⡈", "⡐", "⡠"
-	 ".", "o", "O", "0", "@", "*", " "
-};
-
-const std::size_t kSpinnerCount = sizeof(kSpinner) / sizeof(char*);
-
-const int kSpinnerTimeInterval = 100;
-
 const uint32_t kMinBarWidth = 40, kMinMsgWidth = 12;

-void progress_bar_impl::print_progress()
+void fancy_progress_bar_impl::print_progress()
 {
-	const char *kBlocks[] = {
-		// "▯", // 0
-		// "▮", // 1
-		"=",
-		"-"
-	};
+	const uint32_t pct_width = 5;
+	uint32_t msg_width = m_width - m_bar_width - pct_width - 1;

-	uint32_t width = get_terminal_width();
-
-	float progress = static_cast<float>(m_consumed) / m_max_value;
-	
-	if (width < kMinBarWidth)
-		std::cout << (100 * progress) << "%\n";
-	else
+	if (msg_width < kMinMsgWidth)
 	{
-		uint32_t bar_width = 7 * width / 10;
-		uint32_t pct_width = 7;
-		uint32_t msg_width = width - bar_width - pct_width - 1;
+		m_bar_width += kMinMsgWidth - msg_width;
+		msg_width = kMinMsgWidth;
+	}

-		if (msg_width < kMinMsgWidth)
-		{
-			bar_width += kMinMsgWidth - msg_width;
-			msg_width = kMinMsgWidth;
-		}
+	std::string bar;
+	bar.reserve(m_bar_width * 4);

-		std::ostringstream msg;
-
-		if (m_message.length() <= msg_width)
-		{
-			msg << m_message;
-			if (m_message.length() < msg_width)
-				msg << std::string(msg_width - m_message.length(), ' ');
-		}
+	for (uint32_t i = 0; i < m_bar_width; ++i)
+	{
+		if (i * kBlockCount <= m_steps)
+			bar += kBlocks[kBlockCount];
+		else if (i * kBlockCount > m_steps + kBlockCount)
+			bar += kBlocks[0];
 		else
-			msg << m_message.substr(0, msg_width - 3) << "...";
-
-		msg << ' ';
-
-		uint32_t pi = static_cast<uint32_t>(std::ceil(progress * bar_width));
-
-		for (uint32_t i = 0; i < bar_width; ++i)
-			msg << kBlocks[i > pi ? 1 : 0];
-
-		msg << ' ';
-
-		msg << std::setw(3) << static_cast<int>(std::ceil(progress * 100)) << "% ";
-
-		auto now = std::chrono::system_clock::now();
-		m_spinner_index = (std::chrono::duration_cast<std::chrono::milliseconds>(now - m_start).count() / kSpinnerTimeInterval) % kSpinnerCount;
-
-		msg << kSpinner[m_spinner_index];
-
-		std::cout << '\r' << msg.str();
-		std::cout.flush();
+			bar += kBlocks[1 + m_steps % kBlockCount];
 	}
-}

-namespace
-{
-
-	std::ostream &operator<<(std::ostream &os, const std::chrono::duration<double> &t)
+	// make the bar more colorfull
+	struct color_type
 	{
-		uint64_t s = static_cast<uint64_t>(std::trunc(t.count()));
-		if (s > 24 * 60 * 60)
-		{
-			auto days = s / (24 * 60 * 60);
-			os << days << "d ";
-			s %= 24 * 60 * 60;
-		}
+		uint8_t r, g, b;
+	} fg{ 0, 3, 5 }, bg{ 0, 1, 2 };

-		if (s > 60 * 60)
-		{
-			auto hours = s / (60 * 60);
-			os << hours << "h ";
-			s %= 60 * 60;
-		}
+	auto esc_1 = std::format("\x1b[38;5;{}m\x1b[48;5;{}m",
+		16 + (fg.r * 36) + (fg.g * 6) + fg.b,
+		16 + (bg.r * 36) + (bg.g * 6) + bg.b);
+	std::string esc_2("\x1b[0m");

-		if (s > 60)
-		{
-			auto minutes = s / 60;
-			os << minutes << "m ";
-			s %= 60;
-		}
+	bar = esc_1 + bar + esc_2;

-		double ss = s + 1e-6 * (t.count() - s);
+	std::string msg = m_message.length() <= msg_width
+	                      ? m_message
+	                      : m_message.substr(0, msg_width - 3) + "...";

-		os << std::fixed << std::setprecision(1) << ss << 's';
-
-		return os;
-	}
-
-} // namespace
-
-void progress_bar_impl::print_done()
-{
-	std::chrono::duration<double> elapsed = std::chrono::system_clock::now() - m_start;
-
-	std::ostringstream msgstr;
-	msgstr << m_action << " done in " << elapsed << " seconds";
-	auto msg = msgstr.str();
-
-	uint32_t width = get_terminal_width();
-
-	if (msg.length() < width)
-		msg += std::string(width - msg.length(), ' ');
-
-	std::cout << '\r' << msg << '\n';
+	write_to_console(std::format("{:{}} {} {:3d}%\r", msg, msg_width, bar,
+		static_cast<int>(std::ceil(m_progress * 100))));
 }

-progress_bar::progress_bar(int64_t inMax, const std::string &inAction)
+// --------------------------------------------------------------------
+
+progress_bar::progress_bar(int64_t max_value, const std::string &message)
 	: m_impl(nullptr)
 {
-	if (isatty(STDOUT_FILENO) and VERBOSE >= 0)
-		m_impl = new progress_bar_impl(inMax, inAction);
+	if (VERBOSE >= 0)
+	{
+		if (isatty(STDOUT_FILENO) and get_terminal_width() > kMinBarWidth)
+			m_impl = new fancy_progress_bar_impl(max_value, message);
+		else
+			m_impl = new simple_progress_bar_impl(max_value, message);
+	}
 }

 progress_bar::~progress_bar()
 {
-	delete m_impl;
+	flush();
 }

 void progress_bar::consumed(int64_t inConsumed)
@@ -350,16 +492,25 @@ void progress_bar::consumed(int64_t inConsumed)
 		m_impl->consumed(inConsumed);
 }

-void progress_bar::progress(int64_t inProgress)
+void progress_bar::progress(int64_t value)
 {
 	if (m_impl != nullptr)
-		m_impl->progress(inProgress);
+		m_impl->progress(value);
 }

-void progress_bar::message(const std::string &inMessage)
+void progress_bar::message(const std::string &message)
 {
 	if (m_impl != nullptr)
-		m_impl->message(inMessage);
+		m_impl->message(message);
+}
+
+void progress_bar::flush()
+{
+	if (m_impl)
+	{
+		delete m_impl;
+		m_impl = nullptr;
+	}
 }

 } // namespace cif
@@ -387,13 +538,13 @@ struct rsrc_imp

 #if _WIN32

-#if __MINGW32__
+# if __MINGW32__

 extern "C" __attribute__((weak, alias("gResourceIndexDefault"))) const mrsrc::rsrc_imp gResourceIndex[];
 extern "C" __attribute__((weak, alias("gResourceDataDefault"))) const char gResourceData[];
 extern "C" __attribute__((weak, alias("gResourceNameDefault"))) const char gResourceName[];

-#else
+# else

 extern "C" const mrsrc::rsrc_imp *gResourceIndexDefault[1] = {};
 extern "C" const char *gResourceDataDefault[1] = {};
@@ -403,11 +554,11 @@ extern "C" const mrsrc::rsrc_imp gResourceIndex[];
 extern "C" const char gResourceData[];
 extern "C" const char gResourceName[];

-#pragma comment(linker, "/alternatename:gResourceIndex=gResourceIndexDefault")
-#pragma comment(linker, "/alternatename:gResourceData=gResourceDataDefault")
-#pragma comment(linker, "/alternatename:gResourceName=gResourceNameDefault")
+#  pragma comment(linker, "/alternatename:gResourceIndex=gResourceIndexDefault")
+#  pragma comment(linker, "/alternatename:gResourceData=gResourceDataDefault")
+#  pragma comment(linker, "/alternatename:gResourceName=gResourceNameDefault")

-#endif
+# endif

 #else
 extern const __attribute__((weak)) mrsrc::rsrc_imp gResourceIndex[];
--- a/src/validate.cpp
+++ b/src/validate.cpp
@@ -36,16 +36,11 @@

 // The validator depends on regular expressions. Unfortunately,
 // the implementation of std::regex in g++ is buggy and crashes
-// on reading the pdbx dictionary. Therefore, in case g++ is used
-// the code will use boost::regex instead.
+// on reading the pdbx dictionary. We used to use boost regex
+// instead but using pcre2 is even easier and faster.

-#if USE_BOOST_REGEX
-# include <boost/regex.hpp>
-using boost::regex;
-#else
-# include <regex>
-using std::regex;
-#endif
+#define PCRE2_CODE_UNIT_WIDTH 8
+#include <pcre2.h>

 namespace cif
 {
@@ -67,14 +62,58 @@ validation_exception::validation_exception(std::error_code ec, std::string_view

 // --------------------------------------------------------------------

-struct regex_impl : public regex
+struct regex_impl
 {
-	regex_impl(std::string_view rx)
-		: regex(rx.begin(), rx.end(), regex::extended | regex::optimize)
-	{
-	}
+	regex_impl(std::string_view rx);
+	~regex_impl();
+
+	regex_impl(const regex_impl &) = delete;
+	regex_impl &operator=(const regex_impl &) = delete;
+
+	bool match(std::string_view v) const;
+
+  private:
+	pcre2_code *m_rx = nullptr;
+	pcre2_match_data *m_data = nullptr;
 };

+regex_impl::regex_impl(std::string_view rx)
+{
+	int err_code;
+	size_t err_offset;
+	m_rx = pcre2_compile((PCRE2_SPTR)rx.data(), rx.length(), 0, &err_code, &err_offset, nullptr);
+	if (m_rx == nullptr)
+	{
+		PCRE2_UCHAR buffer[256];
+		int n = pcre2_get_error_message(err_code, buffer, sizeof(buffer));
+
+		throw std::runtime_error(std::string("PCRE2 compilation failed: ") + std::string{ (char *)buffer, (char *)buffer + n });
+	}
+
+	m_data = pcre2_match_data_create_from_pattern(m_rx, nullptr);
+}
+
+regex_impl::~regex_impl()
+{
+	if (m_data)
+		pcre2_match_data_free(m_data);
+
+	if (m_rx)
+		pcre2_code_free(m_rx);
+}
+
+bool regex_impl::match(std::string_view v) const
+{
+	bool result = false;
+
+	if (int rc = pcre2_match(m_rx, (PCRE2_SPTR)v.data(), v.length(), 0, 0, m_data, nullptr); rc >= 0)
+		result = true;
+	else if (rc != PCRE2_ERROR_NOMATCH)
+		std::cerr << "Error matching with pcre\n";
+
+	return result;
+}
+
 // --------------------------------------------------------------------

 DDL_PrimitiveType map_to_primitive_type(std::string_view s, std::error_code &ec) noexcept
@@ -142,8 +181,8 @@ int type_validator::compare(std::string_view a, std::string_view b) const

 				std::from_chars_result ra, rb;

-				ra = selected_charconv<double>::from_chars(a.data(), a.data() + a.length(), da);
-				rb = selected_charconv<double>::from_chars(b.data(), b.data() + b.length(), db);
+				ra = from_chars(a.data(), a.data() + a.length(), da);
+				rb = from_chars(b.data(), b.data() + b.length(), db);

 				if (not(bool) ra.ec and not(bool) rb.ec)
 				{
@@ -233,7 +272,7 @@ bool item_validator::validate_value(std::string_view value, std::error_code &ec)

 	if (not value.empty() and value != "?" and value != ".")
 	{
-		if (m_type != nullptr and not regex_match(value.begin(), value.end(), *m_type->m_rx))
+		if (m_type != nullptr and not m_type->m_rx->match(value))
 			ec = make_error_code(validation_error::value_does_not_match_rx);
 		else if (not m_enums.empty() and m_enums.count(std::string{ value }) == 0)
 			ec = make_error_code(validation_error::value_is_not_in_enumeration_list);
@@ -522,8 +561,10 @@ const validator &validator_factory::get(const category &audit_conform)
 	// If the audit conform contains only one record, this is easy
 	if (audit_conform.size() == 1)
 	{
-		const auto &[name, version] = audit_conform.front().get<std::string, std::optional<std::string>>("dict_name", "dict_version");
-		return m_validators.emplace_back(construct_validator(name, version));
+		const auto &[name, version] =
+			audit_conform.front().get<std::string, std::optional<std::string>>("dict_name", "dict_version");
+		if (not name.empty())
+			return m_validators.emplace_back(construct_validator(name, version));
 	}

 	// A new, merged dictionary
@@ -531,6 +572,9 @@ const validator &validator_factory::get(const category &audit_conform)
 	std::optional<validator> v;
 	for (const auto &[name, version] : audit_conform.rows<std::string, std::optional<std::string>>("dict_name", "dict_version"))
 	{
+		if (name.empty())
+			continue;
+
 		if (not v)
 			v = construct_validator(name, version);
 		else
@@ -552,6 +596,9 @@ const validator &validator_factory::get(const category &audit_conform)
 validator validator_factory::construct_validator(std::string_view name, std::optional<std::string> version)
 {
 	auto data = load_resource(name);
+	if (not data and name == "mmcif_pdbx_v50")
+		data = load_resource("mmcif_pdbx.dic");
+
 	if (not data)
 		throw std::runtime_error("Could not load dictionary " + std::string{ name });

@@ -562,7 +609,7 @@ validator validator_factory::construct_validator(std::string_view name, std::opt
 		not v.matches_audit_conform(category{ "audit_conform", //
 			{ { "dict_name", name }, { "dict_version", version } } }))
 	{
-		std::clog << "Invalid dictionary?\n";
+		std::clog << "Loaded dictionary does not match name=" << name << " and version=" << version.value_or("''") << "\n";
 	}

 	return v;
--- a/test/1cbs-dssp.cif
+++ b/test/1cbs-dssp.cif
--- a/test/CMakeLists.txt
+++ b/test/CMakeLists.txt
@@ -1,19 +1,39 @@
-# We're using the older version 2 of Catch2
+# SPDX-License-Identifier: BSD-2-Clause
+# 
+# Copyright (c) 2025  NKI/AVL, Netherlands Cancer Institute
+# 
+# Redistribution and use in source and binary forms, with or without
+# modification, are permitted provided that the following conditions are met:
+# 
+# 1. Redistributions of source code must retain the above copyright notice, this
+#    list of conditions and the following disclaimer
+# 2. Redistributions in binary form must reproduce the above copyright notice,
+#    this list of conditions and the following disclaimer in the documentation
+#    and/or other materials provided with the distribution.
+# 
+# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
+# ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
+# WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
+# DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
+# ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+# (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
+# LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
+# ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
+# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
+# SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.

-if(NOT(Catch2_FOUND OR TARGET Catch2))
-	find_package(Catch2 QUIET)
+if(NOT (Catch2_FOUND OR TARGET Catch2))
+	find_package(Catch2 3 QUIET)

 	if(NOT Catch2_FOUND)
-		include(FetchContent)
-
 		FetchContent_Declare(
 			Catch2
 			GIT_REPOSITORY https://github.com/catchorg/Catch2.git
-			GIT_TAG v2.13.9)
+			GIT_TAG v3.4.0)

 		FetchContent_MakeAvailable(Catch2)

-		set(Catch2_VERSION "2.13.9")
+		target_compile_features(Catch2 PRIVATE cxx_std_20)
 	endif()
 endif()

@@ -22,24 +42,21 @@ list(
 	CIFPP_tests
 	unit-v2
 	unit-3d
-	format
 	model
+	query
 	rename-compound
 	sugar
 	spinner
-	# reconstruction
-	validate-pdbx)
+	reconstruction
+	validate-pdbx
+	cql
+	matrix
+)

 add_library(test-main OBJECT "${CMAKE_CURRENT_SOURCE_DIR}/test-main.cpp")

 target_link_libraries(test-main cifpp::cifpp Catch2::Catch2)

-if("${Catch2_VERSION}" VERSION_LESS 3.0.0)
-	target_compile_definitions(test-main PUBLIC CATCH22=1)
-else()
-	target_compile_definitions(test-main PUBLIC CATCH22=0)
-endif()
-
 foreach(CIFPP_TEST IN LISTS CIFPP_tests)
 	set(CIFPP_TEST "${CIFPP_TEST}-test")
 	set(CIFPP_TEST_SOURCE "${CMAKE_CURRENT_SOURCE_DIR}/${CIFPP_TEST}.cpp")
@@ -47,12 +64,6 @@ foreach(CIFPP_TEST IN LISTS CIFPP_tests)
 	add_executable(
 		${CIFPP_TEST} ${CIFPP_TEST_SOURCE} $<TARGET_OBJECTS:test-main>)

-	if(${Catch2_VERSION} VERSION_GREATER_EQUAL 3.0.0)
-		target_compile_definitions(${CIFPP_TEST} PUBLIC CATCH22=0)
-	else()
-		target_compile_definitions(${CIFPP_TEST} PUBLIC CATCH22=1)
-	endif()
-
 	target_link_libraries(${CIFPP_TEST} PRIVATE cifpp::cifpp Catch2::Catch2)
 	target_include_directories(${CIFPP_TEST} PRIVATE "${EIGEN_INCLUDE_DIR}")

@@ -61,15 +72,8 @@ foreach(CIFPP_TEST IN LISTS CIFPP_tests)
 		target_compile_options(${CIFPP_TEST} PRIVATE /EHsc)
 	endif()

-	add_custom_target(
-		"run-${CIFPP_TEST}"
-		DEPENDS ${CMAKE_CURRENT_BINARY_DIR}/Run${CIFPP_TEST}.touch ${CIFPP_TEST})
-
-	add_custom_command(
-		OUTPUT ${CMAKE_CURRENT_BINARY_DIR}/Run${CIFPP_TEST}.touch
-		COMMAND $<TARGET_FILE:${CIFPP_TEST}> --data-dir
-		${CMAKE_CURRENT_SOURCE_DIR})
-
-	add_test(NAME ${CIFPP_TEST} COMMAND $<TARGET_FILE:${CIFPP_TEST}> --data-dir
-		${CMAKE_CURRENT_SOURCE_DIR})
-endforeach()
+	if(NOT (CIFPP_TEST STREQUAL "spinner-test"))
+		add_test(NAME ${CIFPP_TEST}
+			COMMAND $<TARGET_FILE:${CIFPP_TEST}> --data-dir ${CMAKE_CURRENT_SOURCE_DIR})
+	endif()
+endforeach()
--- a/test/cql-test.cpp
+++ b/test/cql-test.cpp
@@ -0,0 +1,157 @@
+/*-
+ * SPDX-License-Identifier: BSD-2-Clause
+ *
+ * Copyright (c) 2025 NKI/AVL, Netherlands Cancer Institute
+ *
+ * Redistribution and use in source and binary forms, with or without
+ * modification, are permitted provided that the following conditions are met:
+ *
+ * 1. Redistributions of source code must retain the above copyright notice, this
+ *    list of conditions and the following disclaimer
+ * 2. Redistributions in binary form must reproduce the above copyright notice,
+ *    this list of conditions and the following disclaimer in the documentation
+ *    and/or other materials provided with the distribution.
+ *
+ * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
+ * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
+ * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
+ * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+ * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
+ * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
+ * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
+ * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
+ * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
+ */
+
+#include "test-main.hpp"
+
+#include <catch2/catch_test_macros.hpp>
+#include <cif++.hpp>
+#include <cif++/cql/transaction.hpp>
+#include <stdexcept>
+
+// --------------------------------------------------------------------
+
+cif::file operator""_cf(const char *text, std::size_t length)
+{
+	struct membuf : public std::streambuf
+	{
+		membuf(char *text, std::size_t length)
+		{
+			this->setg(text, text, text + length);
+		}
+	} buffer(const_cast<char *>(text), length);
+
+	std::istream is(&buffer);
+	return cif::file(is);
+}
+
+// --------------------------------------------------------------------
+
+TEST_CASE("cql-1")
+{
+	cif::file f(gTestDir / ".." / "examples" / "1cbs.cif.gz");
+	auto &db = f.front();
+
+	cif::cql::transaction tx(db);
+
+	// CHECK(tx.exec("SELECT COUNT(*) FROM entry").one_field().as<int>() == 1);
+	// CHECK(tx.exec("SELECT COUNT(*) FROM entry WHERE id = '1CBS'").one_field().as<int>() == 1);
+	// CHECK(tx.exec("SELECT COUNT(*) FROM entry WHERE id = 'XXXX'").one_field().as<int>() == 0);
+
+	// CHECK(tx.exec("SELECT COUNT(*) FROM citation").one_field().as<int>() == 4);
+	// CHECK(tx.exec("SELECT COUNT(page_last) FROM citation").one_field().as<int>() == 1);
+
+	const char *kPrimaryAuthors[] = {
+		"Kleywegt, G.J.",
+		"Bergfors, T.",
+		"Senn, H.",
+		"Le Motte, P.",
+		"Gsell, B.",
+		"Shudo, K.",
+		"Jones, T.A."
+	};
+
+	auto r = tx.exec("SELECT name, ordinal FROM citation_author WHERE citation_id = 'primary';");
+	CHECK(r.size() == 7);
+
+	for (size_t ix = 0; auto row : r)
+	{
+		REQUIRE(ix < (sizeof(kPrimaryAuthors) / sizeof(char *)));
+
+		CHECK(row[0].as<std::string>() == kPrimaryAuthors[ix++]);
+		CHECK(row[1].as<size_t>() == ix);
+
+		// CHECK(row["name"].as<std::string>() == kPrimaryAuthors[ix++]);
+		// CHECK(row["ordinal"].as<int>() == ix);
+	}
+
+	r = tx.exec("SELECT ordinal, name FROM citation_author WHERE citation_id = 'primary';");
+	CHECK(r.size() == 7);
+
+	for (size_t ix = 0; auto row : r)
+	{
+		REQUIRE(ix < (sizeof(kPrimaryAuthors) / sizeof(char *)));
+
+		CHECK(row[1].as<std::string>() == kPrimaryAuthors[ix++]);
+		CHECK(row[0].as<size_t>() == ix);
+
+		// CHECK(row["name"].as<std::string>() == kPrimaryAuthors[ix++]);
+		// CHECK(row["ordinal"].as<int>() == ix);
+	}
+
+	r = tx.exec("SELECT * FROM citation_author WHERE citation_id = 'primary';");
+	CHECK(r.size() == 7);
+
+	for (size_t ix = 0; auto row : r)
+	{
+		REQUIRE(ix < (sizeof(kPrimaryAuthors) / sizeof(char *)));
+
+		for (auto fld : row)
+		{
+			switch (fld.num())
+			{
+				case 0:
+					CHECK(fld.name() == "citation_id");
+					CHECK(fld.as<std::string>() == "primary");
+					break;
+				case 1:
+					CHECK(fld.name() == "name");
+					CHECK(fld.as<std::string>() == kPrimaryAuthors[ix]);
+					break;
+				case 2:
+					CHECK(fld.name() == "ordinal");
+					CHECK(fld.as<int>() == ix);
+					break;
+				default:
+					REQUIRE(false);
+					break;
+			}
+		}
+
+		++ix;
+
+		// CHECK(row[0].as<std::string>() == kPrimaryAuthors[ix++]);
+		// CHECK(row[1].as<int>() == ix);
+
+		CHECK(row["name"].as<std::string>() == kPrimaryAuthors[ix++]);
+		CHECK(row["ordinal"].as<size_t>() == ix);
+	}
+
+	// CHECK(tx.query_value<int>("SELECT COUNT(*) FROM citation_author WHERE citation_id = 'primary';") == 7);
+
+	// for (size_t ix = 0; auto row : r)
+	// {
+	// 	REQUIRE(ix < (sizeof(kPrimaryAuthors) / sizeof(char*)));
+	// 	// CHECK(row["name"].as<std::string>() == kPrimaryAuthors[ix++]);
+	// 	// CHECK(row["ordinal"].as<int>() == ix);
+	// }
+
+	// for (size_t ix = 0; const auto &[name, ordinal] : tx.stream<std::string, int>("SELECT name FROM citation_author WHERE citation_id = 'primary'"))
+	// {
+	// 	REQUIRE(ix < (sizeof(kPrimaryAuthors) / sizeof(char*)));
+	// 	CHECK(name == kPrimaryAuthors[ix++]);
+	// 	CHECK(ordinal == ix);
+	// }
+}
--- a/test/matrix-test.cpp
+++ b/test/matrix-test.cpp
@@ -0,0 +1,103 @@
+/*-
+ * SPDX-License-Identifier: BSD-2-Clause
+ *
+ * Copyright (c) 2025 NKI/AVL, Netherlands Cancer Institute
+ *
+ * Redistribution and use in source and binary forms, with or without
+ * modification, are permitted provided that the following conditions are met:
+ *
+ * 1. Redistributions of source code must retain the above copyright notice, this
+ *    list of conditions and the following disclaimer
+ * 2. Redistributions in binary form must reproduce the above copyright notice,
+ *    this list of conditions and the following disclaimer in the documentation
+ *    and/or other materials provided with the distribution.
+ *
+ * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
+ * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
+ * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
+ * DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR
+ * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES
+ * (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES;
+ * LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
+ * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
+ * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS
+ * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
+ */
+
+#include "cif++/matrix.hpp"
+#include "test-main.hpp"
+
+#include <catch2/catch_test_macros.hpp>
+#include <cif++.hpp>
+
+TEST_CASE("m1")
+{
+	cif::matrix3x3<int> m = cif::identity_matrix<int>(3);
+
+	CHECK(cif::determinant(m) == 1);
+}
+
+TEST_CASE("m2")
+{
+	cif::matrix4x4<int> m = cif::identity_matrix<int>(4);
+
+	cif::sub_matrix<cif::matrix4x4<int>> ms(m, 1, 1);
+	CHECK(ms == cif::identity_matrix<int>(3));
+}
+
+TEST_CASE("m3")
+{
+	cif::matrix4x4<int> m{
+		{ 1, 2, 3, 4,      //
+			5, 6, 7, 8,    //
+			9, 10, 11, 12, //
+			13, 14, 15, 16 }
+	};
+	cif::sub_matrix<cif::matrix4x4<int>> ms(m, 1, 1);
+
+	cif::matrix3x3<int> t{
+		{ 1, 3, 4, 9, 11, 12, 13, 15, 16 }
+	};
+
+	CHECK(ms == t);
+}
+
+TEST_CASE("m4")
+{
+	cif::matrix4x4<int> m{
+		{
+			-2,
+			3,
+			1,
+			0,
+			4,
+			1,
+			-3,
+			2,
+			0,
+			-1,
+			2,
+			5,
+			3,
+			2,
+			0,
+			-4,
+		}
+	};
+
+	// std::cout << m << "\n\n";
+
+	// std::cout << cif::matrix3x3<int>(cif::sub_matrix<decltype(m)>(m, 0, 0)) << "\n\n";
+	// std::cout << cif::matrix3x3<int>(cif::sub_matrix<decltype(m)>(m, 0, 1)) << "\n\n";
+	// std::cout << cif::matrix3x3<int>(cif::sub_matrix<decltype(m)>(m, 0, 2)) << "\n\n";
+	// std::cout << cif::matrix3x3<int>(cif::sub_matrix<decltype(m)>(m, 0, 3)) << "\n\n";
+
+	// std::cout << cif::determinant(cif::matrix3x3<int>(cif::sub_matrix<decltype(m)>(m, 0, 0))) << "\n\n";
+	// std::cout << cif::determinant(cif::matrix3x3<int>(cif::sub_matrix<decltype(m)>(m, 0, 1))) << "\n\n";
+	// std::cout << cif::determinant(cif::matrix3x3<int>(cif::sub_matrix<decltype(m)>(m, 0, 2))) << "\n\n";
+	// std::cout << cif::determinant(cif::matrix3x3<int>(cif::sub_matrix<decltype(m)>(m, 0, 3))) << "\n\n";
+
+	CHECK(cif::determinant(m) == 332);
+}
+
+
--- a/test/model-test.cpp
+++ b/test/model-test.cpp
@@ -431,3 +431,169 @@ TEST_CASE("remove_residue_1")

 	REQUIRE_NOTHROW(s.validate_atoms());
 }
+
+// --------------------------------------------------------------------
+// Tests for structure_open_options
+
+TEST_CASE("options_1")
+{
+	using namespace cif::literals;
+
+	const std::filesystem::path example(gTestDir / ".." / "examples" / "1cbs.cif.gz");
+	cif::file file(example.string());
+
+	auto &cf = cif::compound_factory::instance();
+
+	SECTION("skip_water")
+	{
+		cif::mm::structure s(file, 1, { .skip_water = true });
+
+		REQUIRE_NOTHROW(s.validate_atoms());
+
+		for (auto a : s.atoms())
+			CHECK_FALSE(a.is_water());
+	}
+
+	SECTION("skip_hetatom")
+	{
+		cif::mm::structure s(file, 1, { .skip_hetatom = true });
+
+		REQUIRE_NOTHROW(s.validate_atoms());
+
+		for (auto a : s.atoms())
+			CHECK((a.is_water() or cf.is_peptide(a.get_label_comp_id()) or cf.is_base(a.get_label_comp_id())));
+	}
+
+	SECTION("selected_asyms")
+	{
+		cif::mm::structure s(file, 1, { .asyms = { "A" } });
+
+		REQUIRE_NOTHROW(s.validate_atoms());
+
+		for (auto a : s.atoms())
+			CHECK(a.get_label_asym_id() == "A");
+	}
+
+	SECTION("min-b-factor")
+	{
+		cif::mm::structure s(file, 1, { .min_b_factor = 20.f });
+
+		REQUIRE_NOTHROW(s.validate_atoms());
+
+		for (auto a : s.atoms())
+			CHECK(a.get_property_float("B_iso_or_equiv") >= 20.f);
+	}
+
+	SECTION("max-b-factor")
+	{
+		cif::mm::structure s(file, 1, { .max_b_factor = 20.f });
+
+		REQUIRE_NOTHROW(s.validate_atoms());
+
+		for (auto a : s.atoms())
+			CHECK(a.get_property_float("B_iso_or_equiv") <= 20.f);
+	}
+}
+
+TEST_CASE("options_2")
+{
+
+	auto data = R"(
+data_TEST
+# 
+_pdbx_nonpoly_scheme.asym_id         A 
+_pdbx_nonpoly_scheme.ndb_seq_num     1 
+_pdbx_nonpoly_scheme.entity_id       1 
+_pdbx_nonpoly_scheme.mon_id          HEM 
+_pdbx_nonpoly_scheme.pdb_seq_num     1 
+_pdbx_nonpoly_scheme.auth_seq_num    1 
+_pdbx_nonpoly_scheme.pdb_mon_id      HEM 
+_pdbx_nonpoly_scheme.auth_mon_id     HEM 
+_pdbx_nonpoly_scheme.pdb_strand_id   A 
+_pdbx_nonpoly_scheme.pdb_ins_code    . 
+#
+loop_
+_atom_site.id
+_atom_site.auth_asym_id
+_atom_site.label_alt_id
+_atom_site.label_asym_id
+_atom_site.label_atom_id
+_atom_site.label_comp_id
+_atom_site.label_entity_id
+_atom_site.label_seq_id
+_atom_site.type_symbol
+_atom_site.group_PDB
+_atom_site.pdbx_PDB_ins_code
+_atom_site.Cartn_x
+_atom_site.Cartn_y
+_atom_site.Cartn_z
+_atom_site.occupancy
+_atom_site.B_iso_or_equiv
+_atom_site.pdbx_formal_charge
+_atom_site.auth_seq_id
+_atom_site.auth_comp_id
+_atom_site.auth_atom_id
+_atom_site.pdbx_PDB_model_num
+1 A A A CHA HEM 1 . C HETATM ? -5.248 39.769 -0.250 0.75 7.67 ? 1 HEM CHA 1
+3 A A A CHB HEM 1 . C HETATM ? -3.774 36.790 3.280  0.75 7.05 ? 1 HEM CHB 1
+2 A A A CHC HEM 1 . C HETATM ? -2.879 33.328 0.013  0.75 7.69 ? 1 HEM CHC 1
+4 A A A CHD HEM 1 . C HETATM ? -4.342 36.262 -3.536 0.75 8.00 ? 1 HEM CHD 1
+5 A B A CHA HEM 1 . C HETATM ? -5.248 39.769 -0.250 0.25 7.67 ? 1 HEM CHA 1
+6 A B A CHB HEM 1 . C HETATM ? -3.774 36.790 3.280  0.25 7.05 ? 1 HEM CHB 1
+7 A B A CHC HEM 1 . C HETATM ? -2.879 33.328 0.013  0.25 7.69 ? 1 HEM CHC 1
+8 A B A CHD HEM 1 . C HETATM ? -4.342 36.262 -3.536 0.25 8.00 ? 1 HEM CHD 1
+#
+_chem_comp.id               HEM
+_chem_comp.type             NON-POLYMER
+_chem_comp.name             'PROTOPORPHYRIN IX CONTAINING FE'
+_chem_comp.formula          'C34 H32 Fe N4 O4'
+_chem_comp.formula_weight   616.487000
+#
+_pdbx_entity_nonpoly.entity_id   1
+_pdbx_entity_nonpoly.name        'PROTOPORPHYRIN IX CONTAINING FE'
+_pdbx_entity_nonpoly.comp_id     HEM
+#
+_entity.id                 1
+_entity.type               non-polymer
+_entity.pdbx_description   'PROTOPORPHYRIN IX CONTAINING FE'
+_entity.formula_weight     616.487000
+#
+_struct_asym.id                            A
+_struct_asym.entity_id                     1
+_struct_asym.pdbx_blank_PDB_chainid_flag   N
+_struct_asym.pdbx_modified                 N
+_struct_asym.details                       ?
+#
+)"_cf;
+
+	data.front().set_validator(&cif::validator_factory::instance().get("mmcif_pdbx.dic"));
+
+	SECTION("max")
+	{
+		cif::mm::structure s(data, 1, {
+			.occupancy_mode = cif::mm::occupancy_policy::MAX
+		});
+
+		REQUIRE(s.atoms().size() == 4);
+		CHECK(s.atoms().front().get_label_alt_id() == "A");
+	}
+
+	SECTION("min")
+	{
+		cif::mm::structure s(data, 1, {
+			.occupancy_mode = cif::mm::occupancy_policy::MIN
+		});
+
+		REQUIRE(s.atoms().size() == 4);
+		CHECK(s.atoms().front().get_label_alt_id() == "B");
+	}
+
+	SECTION("unoccupied")
+	{
+		cif::mm::structure s(data, 1, {
+			.occupancy_mode = cif::mm::occupancy_policy::UNOCCUPIED
+		});
+
+		CHECK(s.atoms().empty());
+	}
+}
--- a/test/format-test.cpp
+++ b/test/format-test.cpp
@@ -1,17 +1,17 @@
 /*-
 * SPDX-License-Identifier: BSD-2-Clause
- *
- * Copyright (c) 2020 NKI/AVL, Netherlands Cancer Institute
- *
+ * 
+ * Copyright (c) 2025 NKI/AVL, Netherlands Cancer Institute
+ * 
 * Redistribution and use in source and binary forms, with or without
 * modification, are permitted provided that the following conditions are met:
- *
+ * 
 * 1. Redistributions of source code must retain the above copyright notice, this
 *    list of conditions and the following disclaimer
 * 2. Redistributions in binary form must reproduce the above copyright notice,
 *    this list of conditions and the following disclaimer in the documentation
 *    and/or other materials provided with the distribution.
- *
+ * 
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
@@ -24,38 +24,29 @@
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
 */

-
 #include "test-main.hpp"

-#include <stdexcept>
-
 #include <cif++.hpp>

-// --------------------------------------------------------------------
+#include <iostream>
+#include <fstream>

-TEST_CASE("fmt_1")
+TEST_CASE("q-1")
 {
-	std::ostringstream os;
+	using namespace cif::literals;

-	std::string world("world");
-	os << cif::format("Hello, %-10.10s, the magic number is %d and pi is %g", world, 42, cif::kPI);
-	REQUIRE(os.str() == "Hello, world     , the magic number is 42 and pi is 3.14159");
+	cif::compound_factory::instance().push_dictionary(gTestDir / "REA.cif");

-	REQUIRE(cif::format("Hello, %-10.10s, the magic number is %d and pi is %g", world, 42, cif::kPI).str() ==
-		"Hello, world     , the magic number is 42 and pi is 3.14159");
+	cif::file a = cif::pdb::read(gTestDir / "pdb1cbs.ent.gz");
+	auto &pdbx_poly_seq_scheme = a.front()["pdbx_poly_seq_scheme"];
+	REQUIRE_FALSE(pdbx_poly_seq_scheme.empty());
+
+	SECTION("s-11")
+	{
+		CHECK(pdbx_poly_seq_scheme.count("asym_id"_key == "A") == 137);
+
+		CHECK(pdbx_poly_seq_scheme.count("asym_id"_key == "A" and "entity_id"_key == 1 and "seq_id"_key == 1 and "mon_id"_key == "PRO") == 1);
+
+		CHECK(pdbx_poly_seq_scheme.count("asym_id"_key == "A" and "entity_id"_key == 1 and "seq_id"_key == 1 and "mon_id"_key == "PRO" and "hetero"_key == false) == 1);
+	}
 }
-
-// --------------------------------------------------------------------
-
-TEST_CASE("clr_1")
-{
-	using namespace cif::colour;
-
-	std::cout << "Hello, " << cif::coloured("world!", white, red, cif::colour::regular) << '\n'
-			  << "Hello, " << cif::coloured("world!", white, red, bold) << '\n'
-			  << "Hello, " << cif::coloured("world!", black, red) << '\n'
-			  << "Hello, " << cif::coloured("world!", white, green) << '\n'
-			  << "Hello, " << cif::coloured("world!", white, blue) << '\n'
-			  << "Hello, " << cif::coloured("world!", blue, white) << '\n'
-			  << "Hello, " << cif::coloured("world!", red, white, bold) << '\n';
-}
--- a/test/reconstruction-test.cpp
+++ b/test/reconstruction-test.cpp
@@ -28,6 +28,7 @@

 #include <cif++.hpp>

+#include <filesystem>
 #include <iostream>
 #include <fstream>

--- a/test/test-main.cpp
+++ b/test/test-main.cpp
@@ -11,12 +11,7 @@ int main(int argc, char *argv[])
 	Catch::Session session; // There must be exactly one instance

 	// Build a new parser on top of Catch2's
-#if CATCH22
-	using namespace Catch::clara;
-#else
-	// Build a new parser on top of Catch2's
 	using namespace Catch::Clara;
-#endif

 	auto cli = session.cli()                               // Get Catch2's command line parser
 	           | Opt(gTestDir, "data-dir")                 // bind variable to a new option, with a hint string
--- a/test/test-main.hpp
+++ b/test/test-main.hpp
@@ -1,17 +1,17 @@
 /*-
 * SPDX-License-Identifier: BSD-2-Clause
- * 
+ *
 * Copyright (c) 2024 NKI/AVL, Netherlands Cancer Institute
- * 
+ *
 * Redistribution and use in source and binary forms, with or without
 * modification, are permitted provided that the following conditions are met:
- * 
+ *
 * 1. Redistributions of source code must retain the above copyright notice, this
 *    list of conditions and the following disclaimer
 * 2. Redistributions in binary form must reproduce the above copyright notice,
 *    this list of conditions and the following disclaimer in the documentation
 *    and/or other materials provided with the distribution.
- * 
+ *
 * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND
 * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
 * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
@@ -26,11 +26,7 @@

 #pragma once

-#if CATCH22
-#include <catch2/catch.hpp>
-#else
 #include <catch2/catch_all.hpp>
-#endif

 #include <filesystem>

--- a/test/unit-3d-test.cpp
+++ b/test/unit-3d-test.cpp
@@ -24,11 +24,18 @@
 * SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
 */

+#include "cif++/point.hpp"
 #include "test-main.hpp"

+#include <catch2/catch_test_macros.hpp>
+#include <catch2/matchers/catch_matchers_floating_point.hpp>
+#include <cif++.hpp>
 #include <stdexcept>

-#include <cif++.hpp>
+#if defined(_MSC_VER)
+# pragma warning(disable : 5054) // warning C5054: operator '&': deprecated between enumerations of different types
+# pragma warning(disable : 4127) // conditional expression is constant
+#endif

 #include <Eigen/Eigenvalues>

@@ -296,7 +303,7 @@ TEST_CASE("m2q_0a")
 		auto d = cif::kSymopNrTable[i].symop().data();

 		Eigen::Matrix3f rot;
-		rot << d[0], d[1], d[2], d[3], d[4], d[5], d[6], d[7], d[8];
+		rot << static_cast<float>(d[0]), static_cast<float>(d[1]), static_cast<float>(d[2]), static_cast<float>(d[3]), static_cast<float>(d[4]), static_cast<float>(d[5]), static_cast<float>(d[6]), static_cast<float>(d[7]), static_cast<float>(d[8]);

 		// check to see if this matrix contains a true rotation
 		if (rot * rot.transpose() != Eigen::Matrix3f::Identity() or rot.determinant() != 1)
@@ -310,8 +317,7 @@ TEST_CASE("m2q_0a")
 		cif::point p2 = p1;
 		p2.rotate(q);

-		cif::matrix3x3<float> rot_c({
-			static_cast<float>(d[0]),
+		cif::matrix3x3<float> rot_c({ static_cast<float>(d[0]),
 			static_cast<float>(d[1]),
 			static_cast<float>(d[2]),
 			static_cast<float>(d[3]),
@@ -319,8 +325,7 @@ TEST_CASE("m2q_0a")
 			static_cast<float>(d[5]),
 			static_cast<float>(d[6]),
 			static_cast<float>(d[7]),
-			static_cast<float>(d[8])
-		});
+			static_cast<float>(d[8]) });

 		cif::point p3 = rot_c * p1;

@@ -435,11 +440,11 @@ TEST_CASE("symm_4")

 	// based on 2b8h
 	auto sg = cif::spacegroup(154); // p 32 2 1
-	auto c = cif::cell(107.516, 107.516, 338.487, 90.00, 90.00, 120.00);
+	auto c = cif::cell(107.516f, 107.516f, 338.487f, 90.00f, 90.00f, 120.00f);

-	cif::point a{ -8.688, 79.351, 10.439 };  // O6 NAG A 500
-	cif::point b{ -35.356, 33.693, -3.236 }; // CG2 THR D 400
-	cif::point sb(-6.916, 79.34, 3.236);     // 4_565 copy of b
+	cif::point a{ -8.688f, 79.351f, 10.439f };  // O6 NAG A 500
+	cif::point b{ -35.356f, 33.693f, -3.236f }; // CG2 THR D 400
+	cif::point sb(-6.916f, 79.34f, 3.236f);     // 4_565 copy of b

 	CHECK_THAT(distance(a, sg(a, c, "1_455"_symop)), Catch::Matchers::WithinRel(static_cast<float>(c.get_a()), 0.01f));
 	CHECK_THAT(distance(a, sg(a, c, "1_545"_symop)), Catch::Matchers::WithinRel(static_cast<float>(c.get_b()), 0.01f));
@@ -466,7 +471,7 @@ TEST_CASE("symm_4wvp_1")

 	cif::crystal c(db);

-	cif::point p{ -78.722, 98.528, 11.994 };
+	cif::point p{ -78.722f, 98.528f, 11.994f };
 	auto a = s.get_residue("A", 10, "").get_atom_by_atom_id("O");

 	auto sp1 = c.symmetry_copy(a.get_location(), "2_565"_symop);
@@ -605,3 +610,40 @@ TEST_CASE("volume_3bwh_1")

 	CHECK_THAT(c.get_cell().get_volume(), Catch::Matchers::WithinRel(741009.625f, 0.01f));
 }
+
+// --------------------------------------------------------------------
+
+TEST_CASE("smallest_sphere-1")
+{
+	std::vector<cif::point> pts{
+		{ 0.9295, 4.9006, 46.9706 },
+		{ -0.1215, 5.5936, 46.0726 },
+		{ -0.7975, 4.7046, 45.0796 },
+		{ -1.4875, 3.5486, 45.7196 },
+		{ -0.6535, 2.8816, 46.8186 },
+		{ 0.3825, 3.5156, 47.4496 },
+		{ 1.1995, 2.9206, 48.5286 },
+		{ 0.8255, 2.0466, 49.4716 },
+		{ 1.6625, 1.5036, 50.5176 },
+		{ 1.1165, 0.6056, 51.3626 },
+		{ 1.8325, -0.0064, 52.4656 },
+		{ 1.1945, -0.9044, 53.2216 },
+		{ 1.8135, -1.5534, 54.3566 },
+		{ 1.0925, -2.4574, 55.0656 },
+		{ 1.5205, -3.2204, 56.2476 },
+		{ 1.1955, 5.8066, 48.1796 },
+		{ 2.2495, 4.6896, 46.1796 },
+		{ -1.2515, 1.5186, 47.1786 },
+		{ 3.1385, 1.9106, 50.6166 },
+		{ 3.2605, -1.1834, 54.7206 },
+		{ 2.5975, -3.8554, 56.2096 },
+		{ 0.7975, -3.2184, 57.2686 }
+	};
+
+	for (int i = 0; i < 1000; ++i)
+	{
+		auto [c, r] = cif::smallest_sphere_around_points(pts);
+		CHECK_THAT(cif::distance(c, cif::point{ 0, 0.743099928, 51.1741028 }), Catch::Matchers::WithinAbs(0.f, 0.01f));
+		CHECK_THAT(r, Catch::Matchers::WithinAbs(7.31248331f, 0.01f));
+	}
+}
--- a/test/unit-v2-test.cpp
+++ b/test/unit-v2-test.cpp
--- a/tools/depends.cmd
+++ b/tools/depends.cmd
@@ -5,6 +5,8 @@ IF NOT EXIST build_ci\libs (
  MKDIR build_ci\libs
 )
 CD build_ci\libs
+
+@REM Install ZLib
 IF NOT EXIST zlib-%ZLIB_VERSION%.zip (
  ECHO Downloading https://github.com/libarchive/zlib/archive/v%ZLIB_VERSION%.zip
  curl -L -o zlib-%ZLIB_VERSION%.zip https://github.com/libarchive/zlib/archive/v%ZLIB_VERSION%.zip || EXIT /b 1
@@ -14,9 +16,9 @@ IF NOT EXIST zlib-%ZLIB_VERSION% (
  C:\windows\system32\tar.exe -x -f zlib-%ZLIB_VERSION%.zip || EXIT /b 1
 )
 CD zlib-%ZLIB_VERSION%
-cmake -G "Visual Studio 17 2022" . || EXIT /b 1
-cmake --build . --target ALL_BUILD --config Release || EXIT /b 1
-cmake --build . --target RUN_TESTS --config Release || EXIT /b 1
-cmake --build . --target INSTALL --config Release || EXIT /b 1
+cmake -B build || EXIT /b 1
+cmake --build build --target ALL_BUILD --config Release || EXIT /b 1
+cmake --build build --target RUN_TESTS --config Release || EXIT /b 1
+cmake --build build --target INSTALL --config Release || EXIT /b 1

@EXIT /b 0
--- a/tools/update-libcifpp-data.in
+++ b/tools/update-libcifpp-data.in
@@ -63,7 +63,7 @@ update_dictionary() {

 update_dictionary "@CIFPP_CACHE_DIR@/components.cif" "https://files.wwpdb.org/pub/pdb/data/monomers/components.cif.gz"
 update_dictionary "@CIFPP_CACHE_DIR@/mmcif_pdbx.dic" "https://mmcif.wwpdb.org/dictionaries/ascii/mmcif_pdbx_v50.dic.gz"
-update_dictionary "@CIFPP_CACHE_DIR@/mmcif_ma.dic" "https://github.com/ihmwg/ModelCIF/raw/master/dist/mmcif_ma.dic"
+update_dictionary "@CIFPP_CACHE_DIR@/mmcif_ma.dic" "https://mmcif.wwpdb.org/dictionaries/ascii/mmcif_ma.dic"

 # notify subscribers, using find instead of run-parts to make it work on FreeBSD as well
Author	SHA1	Message	Date
Maarten L. Hekkelman	00b0473438	backup	2025-12-03 16:22:44 +01:00
Maarten L. Hekkelman	f15a76e29b	exclude from all for fast_float	2025-11-27 09:05:51 +01:00
Maarten L. Hekkelman	915a147449	backup	2025-11-26 16:35:30 +01:00
Maarten L. Hekkelman	edf24ca9ff	work	2025-11-26 13:46:55 +01:00
Maarten L. Hekkelman	46a9318aa5	revert version string generator	2025-11-26 13:11:54 +01:00
Maarten L. Hekkelman	4a7f48eed8	some initial work	2025-11-19 16:36:44 +01:00
Maarten L. Hekkelman	42e66afd92	Merge branch 'develop' into cql	2025-11-19 13:29:35 +01:00
Maarten L. Hekkelman	b550e9b027	re-enable tests	2025-11-19 13:28:47 +01:00
Maarten L. Hekkelman	452bb83ce7	Remove revision.hpp file when making clean	2025-11-19 11:39:27 +01:00
Maarten L. Hekkelman	6eda9aaf36	better center_and_radius for residue	2025-11-18 16:44:53 +01:00
Maarten L. Hekkelman	251fb55d6a	fixing smallest sphere	2025-11-05 13:18:58 +01:00
Maarten L. Hekkelman	f94e9aece9	create_non_poly, another	2025-11-05 11:03:56 +01:00
Maarten L. Hekkelman	c565bb96be	Do not run the spinner test	2025-10-30 09:19:50 +01:00
Maarten L. Hekkelman	e51f31dc4c	Remove libfmt, fix instantiating templates for fast_float usage	2025-10-30 09:08:44 +01:00
Maarten L. Hekkelman	4e128885d6	Added missing include	2025-10-29 18:21:47 +01:00
Maarten L. Hekkelman	b37054228d	Added smalles sphere function	2025-10-29 17:09:13 +01:00
Maarten L. Hekkelman	815b33fee0	Matrix determinant for 4x4	2025-10-28 15:55:05 +01:00
Maarten L. Hekkelman	97f55c1639	Version bump	2025-10-22 10:05:55 +02:00
Maarten L. Hekkelman	89de73eb6f	Added exists to compound_factory	2025-10-21 13:06:09 +02:00
Maarten L. Hekkelman	75f2ec3792	Remove warning	2025-10-13 14:22:57 +02:00
Maarten L. Hekkelman	f4d29e8da9	re-enable test to see if fast_float is required	2025-10-01 17:09:06 +02:00
Maarten L. Hekkelman	b97b2638b8	More supported float types	2025-10-01 17:08:33 +02:00
Maarten L. Hekkelman	ea8dea8cbd	Merge branch 'develop' into cql	2025-10-01 16:46:27 +02:00
Maarten L. Hekkelman	bc0222dc0e	attempt two	2025-10-01 16:42:05 +02:00
Maarten L. Hekkelman	10a6b5649b	Using fast float instead of home baked version	2025-10-01 16:14:07 +02:00
Maarten L. Hekkelman	ff2a233156	stap 1, een test	2025-10-01 15:48:16 +02:00
Maarten L. Hekkelman	743e2800f8	update changelog	2025-09-30 11:21:03 +02:00
Maarten L. Hekkelman	32ac884127	Do not stop on empty audit_conform fields	2025-09-29 10:34:29 +02:00
Maarten L. Hekkelman	bec69f7d07	Fix reconstruction when entity ID's are missing	2025-09-29 09:59:04 +02:00
Maarten L. Hekkelman	a99215ad6a	version bump	2025-09-24 16:45:05 +02:00
Maarten L. Hekkelman	e3d2cbd044	Lower required catch2 version	2025-09-24 16:42:45 +02:00
Maarten L. Hekkelman	5fc965789d	messages updated	2025-09-24 15:11:33 +02:00
Maarten L. Hekkelman	b4596902aa	Add compile features for Catch2, required on Windows	2025-09-24 14:15:19 +02:00
Maarten L. Hekkelman	cbf8b52f62	Update catch2 usage	2025-09-24 13:25:56 +02:00
Maarten L. Hekkelman	4e0fa1c916	No complete jthread on macOS/CLang	2025-09-24 13:00:50 +02:00
Maarten L. Hekkelman	95b007d38f	Merge branch 'trunk' into develop	2025-09-24 11:38:01 +02:00
Maarten L. Hekkelman	b66f7a30ce	Progress bar using WriteConsole on Windows	2025-09-24 11:36:37 +02:00
Maarten L. Hekkelman	ec7287c503	remove warning	2025-09-24 11:32:12 +02:00
Maarten L. Hekkelman	a41c591f0c	Restore order of imports, avoid reordering by clang-format	2025-09-24 10:51:53 +02:00
Maarten L. Hekkelman	3a6527cdd5	yet another update on progress bar	2025-09-24 10:23:58 +02:00
Maarten L. Hekkelman	5f21a094c0	added flush to progress bar	2025-09-24 10:14:03 +02:00
Maarten L. Hekkelman	2203a1855d	improved progress bar	2025-09-24 09:49:28 +02:00
Maarten L. Hekkelman	7edd2ecc18	new progress bar	2025-09-23 15:53:55 +02:00
Maarten L. Hekkelman	1d2953c850	Fix reconstruction, version bump	2025-09-22 13:51:18 +02:00
Maarten L. Hekkelman	dbf59ce622	reconstruct better when entity ID's are missing	2025-09-22 12:59:16 +02:00
Maarten L. Hekkelman	1596db8499	Merge branch 'develop' of github.com:pdb-redo/libcifpp into develop	2025-09-16 13:29:57 +02:00
Maarten L. Hekkelman	bd1fb5c5cd	Added model::create_link	2025-09-16 13:29:51 +02:00
Maarten L. Hekkelman	da500025c3	swap atoms should swap type_symbol as well	2025-09-10 17:16:22 +02:00
Maarten L. Hekkelman	60eeea9a93	more resilient loading of dictionary data	2025-09-10 15:06:06 +02:00
Maarten L. Hekkelman	1220f01f1e	change location of mmcif_ma.dic	2025-09-10 14:22:52 +02:00
Maarten L. Hekkelman	ad0a34fe98	Update changelog	2025-09-10 12:49:43 +02:00
Maarten L. Hekkelman	a7425ff1a0	Optimise validation code	2025-09-10 12:40:01 +02:00
Maarten L. Hekkelman	1ce25f86ae	better check anisotrop atoms	2025-09-10 12:19:56 +02:00
Maarten L. Hekkelman	cd93f72b96	Merge branch 'develop' into better-create-entity-ids	2025-09-10 09:28:50 +02:00
Maarten L. Hekkelman	23500bd303	Fix reconstruction of really bare files	2025-09-10 09:25:49 +02:00
Maarten L. Hekkelman	14b4753b4f	test for null	2025-09-09 19:55:13 +02:00
Maarten L. Hekkelman	4c37d5db5f	use rowhandles	2025-09-09 19:52:39 +02:00
Maarten L. Hekkelman	fc2c4b4172	fix map::at in reconstruct sequences	2025-09-09 10:51:31 +02:00
Maarten L. Hekkelman	3ac64de16b	Version bump, update changelog	2025-09-03 14:55:20 +02:00
Maarten L. Hekkelman	45eecd72b0	using pkg-config, when available	2025-09-03 14:24:29 +02:00
Maarten L. Hekkelman	d1dd558cda	as object lib	2025-09-03 14:00:22 +02:00
Maarten L. Hekkelman	d19e2c2196	Update pcre2s	2025-09-03 13:11:45 +02:00
Maarten L. Hekkelman	72c7aca074	revert to catch2 version 2, due to linker errors on Windows?	2025-09-03 13:04:13 +02:00
Maarten L. Hekkelman	683a1087d0	Install catch2 before testing	2025-09-03 12:28:06 +02:00
Maarten L. Hekkelman	35bc139deb	Update pcre2s	2025-09-03 11:39:00 +02:00
Maarten L. Hekkelman	45ece2fa0d	pcre2 is no longer a depends on Windows	2025-09-03 11:18:04 +02:00
Maarten L. Hekkelman	11c98f553f	Update Catch2 to version 3 Updated pcre2s	2025-09-03 11:15:31 +02:00
Maarten L. Hekkelman	28aa9b1036	Using pcre2s	2025-09-03 11:07:02 +02:00
Maarten L. Hekkelman	d7b5c0a748	remove message	2025-09-02 15:21:04 +02:00
Maarten L. Hekkelman	065e7f5f18	added clang format file	2025-09-02 14:59:28 +02:00
Maarten L. Hekkelman	4b1623cfdc	pcre2 again	2025-09-02 14:57:07 +02:00
Maarten L. Hekkelman	1de973ddcb	Fix findpcre2 cmake file	2025-09-02 13:12:20 +02:00
Maarten L. Hekkelman	eecc801203	last remaining warning	2025-09-02 12:57:08 +02:00
Maarten L. Hekkelman	5c50154ea4	Merge branch 'develop' of github.com:PDB-REDO/libcifpp into develop	2025-09-02 12:54:03 +02:00
Maarten L. Hekkelman	0fa3d6aa94	Removing warning using MSVC	2025-09-02 12:54:07 +02:00
Maarten L. Hekkelman	01f5242bfb	Revert "Add formatting file" This reverts commit `af6d8d4f71`.	2025-09-02 11:57:21 +02:00
Maarten L. Hekkelman	af6d8d4f71	Add formatting file	2025-09-02 11:47:50 +02:00
Maarten L. Hekkelman	fa8285fc0f	use std::format anyway, even if __cpp_lib_format is not defined.	2025-09-02 10:23:00 +02:00
Maarten L. Hekkelman	2e7f6b8337	cross platform check for lib format	2025-09-02 10:03:36 +02:00
Maarten L. Hekkelman	a6a55020eb	Merge branch 'develop' of github.com:PDB-REDO/libcifpp into develop	2025-09-01 10:54:09 +02:00
Maarten L. Hekkelman	0e84ea454d	fmt fix	2025-09-01 10:53:29 +02:00
Maarten L. Hekkelman	f3bf211d45	pdb formatting	2025-09-01 09:34:07 +02:00
Maarten L. Hekkelman	f5ef44836c	pdb formatting	2025-09-01 09:31:22 +02:00
Maarten L. Hekkelman	070124b6e1	fix cif2pdb	2025-08-29 14:00:27 +02:00
Maarten L. Hekkelman	c8a46fcdd9	Stop when element is missing in reading PDB input	2025-08-29 13:17:17 +02:00
Maarten L. Hekkelman	5306b59fd8	Do not write zero seq ID's in PDB files	2025-08-29 10:08:56 +02:00
Maarten L. Hekkelman	90c5df832a	Check only first datablock	2025-08-29 09:49:27 +02:00
Maarten L. Hekkelman	2aa439d51f	test for fmt	2025-08-28 08:48:20 +02:00
Maarten L. Hekkelman	ac2b68517c	conditional fmt	2025-08-27 16:08:51 +02:00
Maarten L. Hekkelman	e56b568c42	use cif::format... sigh	2025-08-27 15:46:41 +02:00
Maarten L. Hekkelman	63c49b2e04	Fix writing pdb files	2025-08-27 15:20:20 +02:00
Maarten L. Hekkelman	559fd18a20	Fix std::format usage	2025-08-27 08:57:40 +02:00
Maarten L. Hekkelman	beb7585261	fix std::format usage	2025-08-27 08:24:07 +02:00
Maarten L. Hekkelman	8b0f92aa9a	Merge remote-tracking branch 'github/using-fmt' into develop	2025-08-27 07:43:22 +02:00
Maarten L. Hekkelman	0d8beeae5b	No longer needed	2025-08-26 16:04:31 +02:00
Maarten L. Hekkelman	e3da654e67	Speed up build when eigen3 is not available	2025-08-26 15:54:55 +02:00
Maarten L. Hekkelman	dc9e151d89	remove warning	2025-08-26 15:37:22 +02:00
Maarten L. Hekkelman	7cfaf051ba	should now work on windows	2025-08-26 15:16:01 +02:00
Maarten L. Hekkelman	7920491309	hope to find pcre2.h	2025-08-26 14:07:47 +02:00
Maarten L. Hekkelman	0ee493a3fb	implib?	2025-08-26 13:55:25 +02:00
Maarten L. Hekkelman	7e23bc0c0b	finding pcre2 on windows	2025-08-26 13:46:03 +02:00
Maarten L. Hekkelman	579f859562	no find_packge for pcre2	2025-08-26 13:42:14 +02:00
Maarten L. Hekkelman	752938ca44	using find_package?	2025-08-26 13:38:37 +02:00
Maarten L. Hekkelman	fce58c02fe	no pcre2grep please, no tests either	2025-08-26 13:34:43 +02:00
Maarten L. Hekkelman	924f7c1505	finding pcre2, yet another attempt	2025-08-26 13:31:12 +02:00
Maarten L. Hekkelman	8944906fd2	fix warning, pcre2	2025-08-26 12:41:08 +02:00
Maarten L. Hekkelman	cb02969604	Using std::format	2025-08-25 16:31:00 +02:00
Maarten L. Hekkelman	31090c6ec5	attempt 2	2025-08-25 11:25:10 +02:00
Maarten L. Hekkelman	9e30d2bc1a	finding pcre2	2025-08-25 10:39:48 +02:00
Maarten L. Hekkelman	93d703f7a1	Do not buld pcre tests	2025-08-20 20:40:08 +02:00
Maarten L. Hekkelman	3c241048a5	do not install pcre	2025-08-20 17:02:40 +02:00
Maarten L. Hekkelman	2788536799	this should work	2025-08-20 16:58:52 +02:00
Maarten L. Hekkelman	314d435a18	Another way of importing pcre	2025-08-20 16:49:12 +02:00
Maarten L. Hekkelman	37edcd8666	Finding and optionally building pcre	2025-08-20 15:47:49 +02:00
Maarten L. Hekkelman	10e290fbdf	pcre2 in github actions?	2025-08-20 14:40:50 +02:00
Maarten L. Hekkelman	58cda1241e	cleanup	2025-08-20 13:41:45 +02:00
Maarten L. Hekkelman	3659aaabff	remove unneeded allocations	2025-08-20 13:35:15 +02:00
Maarten L. Hekkelman	727a39cc54	Finishing up replacing boost with pcre	2025-08-20 13:28:24 +02:00
Maarten L. Hekkelman	fd9ccdfff9	Using pcre instead of boost::regex	2025-08-19 16:16:43 +02:00
Maarten L. Hekkelman	aabee270b3	update .gitignore	2025-08-19 14:28:51 +02:00
Maarten L. Hekkelman	647c58f8ec	allow code to be built with older compilers...	2025-08-19 12:44:41 +02:00
Maarten L. Hekkelman	0b8024d19c	Optimise query processing	2025-08-19 12:24:33 +02:00
Maarten L. Hekkelman	d59b0bf27f	Remove wrong warnings	2025-08-13 11:30:53 +02:00
Maarten L. Hekkelman	398c16eac2	Merge branch 'develop' of github.com:PDB-REDO/libcifpp into develop	2025-08-13 09:02:10 +02:00
Maarten L. Hekkelman	fa869bdc7d	lightweight fixup	2025-08-13 09:01:50 +02:00
Maarten L. Hekkelman	c20d0d2a30	lightweight fixup	2025-08-12 16:45:19 +02:00
Maarten L. Hekkelman	000f2736c2	Merge branch 'develop' into clebreto-feature/enrich_structure_options	2025-08-11 10:13:20 +02:00
Maarten L. Hekkelman	cfcc81bb62	verbose messages	2025-08-11 10:13:07 +02:00
Maarten L. Hekkelman	82eae05868	changed b-factor options for structure loading	2025-06-11 14:17:50 +02:00
Maarten L. Hekkelman	e8fb53c49b	Alternate implementation of structure_open_options	2025-06-11 13:35:58 +02:00
Maarten L. Hekkelman	604c97afe1	Merge branch 'develop' into clebreto-feature/enrich_structure_options	2025-06-11 11:43:01 +02:00
Maarten L. Hekkelman	7e60cdf272	remove redundant statement	2025-06-11 11:42:26 +02:00
Maarten L. Hekkelman	9ea7cfcc80	Remove test	2025-06-11 09:56:26 +02:00
Maarten L. Hekkelman	a7a4a16f79	remove debug code	2025-06-11 09:45:45 +02:00
Maarten L. Hekkelman	6717059934	Revert renaming compound_id to mon_id in residue	2025-06-11 09:41:49 +02:00
Maarten L. Hekkelman	714747c280	version bump	2025-06-11 09:32:36 +02:00
Maarten L. Hekkelman	81cd305c80	rename mm::polymer fields and methods to better match mmcif_pdbx naming. fix building mm::structure using pdb_seq_num instead of auth_seq_num	2025-06-11 09:30:54 +02:00
Maarten L. Hekkelman	5de872bbb3	Version bump, update mmcif_pdbx.dic	2025-06-10 09:17:30 +02:00
Maarten L. Hekkelman	ce6a75a920	right...	2025-06-10 09:11:26 +02:00
Maarten L. Hekkelman	874a5cb2f2	missing code added	2025-06-02 15:09:49 +02:00
Maarten L. Hekkelman	6e2202d4f1	More verbose strip	2025-06-02 09:10:58 +02:00
Maarten L. Hekkelman	bcf33df701	Added strip, removed dangerous datablock::is_valid (non-const version)	2025-06-02 08:52:58 +02:00
Maarten L. Hekkelman	3bdcf21c69	Merge commit '4b36bdc' into develop	2025-05-29 16:14:15 +02:00
Maarten L. Hekkelman	4b36bdc58c	work around incorrect mmcif_pdbx name	2025-05-29 16:13:28 +02:00
Maarten L. Hekkelman	6d9008ee8c	Merge branch 'trunk' into develop	2025-05-29 15:15:41 +02:00
Maarten L. Hekkelman	ee93692707	comment formatting	2025-05-29 14:15:13 +02:00
Maarten L. Hekkelman	2bcc368bce	reconstruct when audit_conform is missing	2025-05-29 14:07:45 +02:00
Maarten L. Hekkelman	6cc4467d53	options	2025-05-27 13:55:13 +02:00
Maarten L. Hekkelman	425f98dc07	No longer fail pdb conversion when missing compound info	2025-05-12 12:45:21 +02:00
Maarten L. Hekkelman	cf34a9f3ad	Updated test file	2025-05-12 11:56:42 +02:00
Maarten L. Hekkelman	3b2f347428	Remove reconstruct test again	2025-05-12 11:46:16 +02:00
Maarten L. Hekkelman	bd82c3cc4f	Only reconstruct when needed	2025-05-12 10:24:00 +02:00
Maarten L. Hekkelman	af319866c7	added get_atom_by_atom_id	2025-05-06 14:11:05 +02:00
Maarten L. Hekkelman	b6ab29398e	Fix parsing PDB files	2025-04-15 09:56:51 +02:00
Maarten L. Hekkelman	a5bb1797c0	Merge branch 'trunk' of github.com:PDB-REDO/libcifpp into trunk	2025-04-15 09:17:44 +02:00
Maarten L. Hekkelman	a9647671c4	entity_poly.type is now by default 'other'	2025-04-15 09:17:37 +02:00
Maarten L. Hekkelman	63f784e7da	Reconstruct branches, somewhat	2025-04-14 12:44:27 +02:00
Maarten L. Hekkelman	5da3379e0b	Don't reconstruct invalid files, non-polymer should not have auth_seq_id	2025-04-14 11:06:36 +02:00
Maarten L. Hekkelman	2f3514689d	Default value for b_iso_or_equiv, better sorting of atoms	2025-04-09 15:12:11 +02:00
Maarten L. Hekkelman	89a3ea4e24	update mmcif_pdbx.dic	2025-04-09 15:11:37 +02:00
Maarten L. Hekkelman	467e9555f4	remove duplicate test	2025-04-09 13:11:05 +02:00
Maarten L. Hekkelman	5b32ca15f7	Fix cleanup_empty_categories	2025-04-09 13:06:45 +02:00
Maarten L. Hekkelman	92402817d2	Merge branch 'trunk' into develop	2025-04-09 13:05:12 +02:00
Maarten L. Hekkelman	60ad3031d5	remove warning, hopefully fix docs action	2025-04-09 10:15:15 +02:00
Maarten L. Hekkelman	41c0521480	Merge branch 'feature/enrich_structure_options' of github.com:clebreto/libcifpp into clebreto-feature/enrich_structure_options	2025-04-02 13:58:01 +02:00
Maarten L. Hekkelman	13a97353aa	fix test	2025-04-02 11:18:20 +02:00
Maarten L. Hekkelman	f49c166b9b	Fix for MSVC	2025-03-31 12:52:26 +02:00
Maarten L. Hekkelman	fffa326f80	validator for multiple dictionaries	2025-03-31 12:38:28 +02:00
Maarten L. Hekkelman	da446adbb2	non compiling code	2025-03-28 16:15:58 +01:00
Maarten L. Hekkelman	617fec5c69	Close, but no cigar	2025-03-28 15:06:02 +01:00
Maarten L. Hekkelman	cfefa69c9c	Merge branch 'develop' of github.com:PDB-REDO/libcifpp into develop	2025-03-27 11:40:42 +01:00
Maarten L. Hekkelman	00638a9e23	Fix loading coordinates from converted restraint files	2025-03-27 11:36:46 +01:00
Maarten L. Hekkelman	e241e03a15	Fix loading coordinates from converted restraint files	2025-03-27 11:26:30 +01:00
LE BRETON Come	7d33d56c0e	Update docs	2025-03-07 15:59:19 +01:00
LE BRETON Come	f86f34e5e1	WIP Enrich StructureOpenOptions	2025-03-07 15:54:30 +01:00
Maarten L. Hekkelman	5e7b52b7de	loading dictionaries	2025-02-17 12:57:08 +01:00
Maarten L. Hekkelman	0459d344e9	Fixes in error reporting	2025-02-17 12:32:14 +01:00
Maarten L. Hekkelman	71e525cd76	Refactored dictionary loading	2025-02-17 09:40:36 +01:00
Maarten L. Hekkelman	1480706d8b	change for mingw	2025-02-05 16:05:08 +01:00
Maarten L. Hekkelman	96655b6d80	revert	2025-01-29 17:12:59 +01:00
Maarten L. Hekkelman	eed2aa0d0d	better way to include eigen3	2025-01-29 17:01:44 +01:00
Maarten L. Hekkelman	de0c078a23	Update changelog	2025-01-29 16:08:55 +01:00
Maarten L. Hekkelman	321e995a54	Add some comments	2025-01-29 16:07:03 +01:00
Maarten L. Hekkelman	da9f1f81d7	Fix eigen3 problems on github?	2025-01-29 15:57:16 +01:00
Maarten L. Hekkelman	c6d4477a24	Using eigen quaternions	2025-01-29 15:37:57 +01:00
Maarten L. Hekkelman	523b073cdc	own eigen	2025-01-29 14:25:28 +01:00
Maarten L. Hekkelman	2591bee21b	test for github actions, own eigen library	2025-01-29 13:54:20 +01:00
Maarten L. Hekkelman	d881ca00c9	cleanup	2025-01-29 13:54:00 +01:00
Maarten L. Hekkelman	329dbff474	replace deprecated call	2025-01-29 13:42:37 +01:00
Maarten L. Hekkelman	d84a9fe6dc	Deal with missing entity.type	2025-01-29 13:41:11 +01:00