diff --git a/ColabFold2_preview.ipynb b/ColabFold2_preview.ipynb
new file mode 100644
index 000000000..462a5b4b0
--- /dev/null
+++ b/ColabFold2_preview.ipynb
@@ -0,0 +1,773 @@
+{
+ "cells": [
+ {
+ "cell_type": "markdown",
+ "metadata": {
+ "id": "view-in-github",
+ "colab_type": "text"
+ },
+ "source": [
+ "
"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "header",
+ "metadata": {
+ "id": "header"
+ },
+ "source": [
+ "# ColabFold2 preview\n",
+ "\n",
+ "Predict protein, RNA, DNA and small-molecule structures with\n",
+ "[AlphaFold 3](https://www.nature.com/articles/s41586-024-07487-w) \u2014 and with thirteen\n",
+ "other sets of weights, all through one implementation. Pick a model in the install\n",
+ "cell; the weights download themselves. MSAs come from the\n",
+ "[ColabFold](https://github.com/sokrypton/ColabFold) MMseqs2 server, so no local\n",
+ "databases are needed, and the attention/XLA flags are chosen for whatever GPU you get.\n",
+ "\n",
+ "**Models, licences, download sizes and every option: the Instructions cell at the\n",
+ "bottom.**\n",
+ "\n",
+ "**Please cite** AlphaFold 3 ([Abramson 2024](https://doi.org/10.1038/s41586-024-07487-w)),\n",
+ "ColabFold ([Mirdita 2022](https://doi.org/10.1038/s41592-022-01488-1)), and whichever\n",
+ "model you ran.\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "id": "install",
+ "metadata": {
+ "cellView": "form",
+ "id": "install"
+ },
+ "outputs": [],
+ "source": [
+ "#@title Install dependencies (~35 s)\n",
+ "import os, time, glob, shutil, sys\n",
+ "_T0 = time.time()\n",
+ "\n",
+ "model = \"openbind0\" #@param [\"openbind0\", \"openfold3\", \"boltz2\", \"protenix2\", \"rosettafold3\", \"chai1\", \"intellifold2\", \"opendde\", \"esmfold2\", \"esmfold2_lm600m\", \"esmfold2_lm300m\", \"alphafold3\", \"af2_ptm\", \"af2_multimer\"]\n",
+ "\n",
+ "persist_cache_to_drive = False #@param {type:\"boolean\"}\n",
+ "#@markdown - **persist_cache_to_drive**: keep the compiled model in Drive so the next\n",
+ "#@markdown session skips the recompile -- worth ~53 s (69 s cold vs 16 s warm on a\n",
+ "#@markdown 68-residue input). Never changes a result.\n",
+ "\n",
+ "# Set any form field from the environment, for runs outside Colab:\n",
+ "# AF3_NB_OVERRIDES='{\"model\": \"boltz2\"}'\n",
+ "import json as _json\n",
+ "for _k, _v in _json.loads(os.environ.get('AF3_NB_OVERRIDES', '{}')).items():\n",
+ " if _k in globals():\n",
+ " globals()[_k] = _v\n",
+ " print(f'override: {_k} = {_v!r}')\n",
+ "\n",
+ "VERSION = '3.1.11' # package and run_alphafold.py both come from this tag\n",
+ "NATIVE_DIR = 'af3_native_weights'\n",
+ "AF3_WEIGHTS_URL = 'https://storage.googleapis.com/alphafold3/af3.bin.zst'\n",
+ "AF2_DIR = 'af2_params'\n",
+ "IS_AF3 = (model == 'alphafold3')\n",
+ "IS_AF2 = model.startswith('af2_')\n",
+ "# int8 weights, expanded on load; AF2 and AF3 ship their own float32 files.\n",
+ "PRECISION = 'fp32' if (IS_AF3 or IS_AF2) else 'int8'\n",
+ "\n",
+ "\n",
+ "def _sh(cmd, what):\n",
+ " \"\"\"Run a shell command, raising if it fails.\"\"\"\n",
+ " if os.system(cmd) != 0:\n",
+ " raise RuntimeError(f'{what} failed. The output is above.')\n",
+ "\n",
+ "\n",
+ "if not os.path.isfile('ALPHAFOLD3_READY'):\n",
+ " print('Installing packages...')\n",
+ " # Installed with --no-deps, so the package's own imports are listed here.\n",
+ " # Letting pip resolve them would re-download jax and the CUDA stack.\n",
+ " _sh(\"pip install -q dm-haiku==0.0.17 rdkit==2025.9.4 \"\n",
+ " \"tokamax==0.0.11 ml_collections\", 'installing dependencies')\n",
+ " _sh(\"pip install -q git+https://github.com/sokrypton/py2Dmol.git\", # wheel lags the repo\n",
+ " 'installing py2Dmol')\n",
+ " if IS_AF2:\n",
+ " os.system(\"apt-get -qq install -y aria2 > /dev/null 2>&1\") # AF2's tar is 5.3 GB\n",
+ " # Retried: PyPI's index can lag a just-published release by a few minutes.\n",
+ " for _try in range(4):\n",
+ " if os.system(f'pip install -q --no-deps alphafold3-colabfold=={VERSION}') == 0:\n",
+ " break\n",
+ " print(f'pip could not find {VERSION} yet; retrying in 20 s')\n",
+ " time.sleep(20)\n",
+ " else:\n",
+ " raise RuntimeError(f'could not install alphafold3-colabfold=={VERSION}')\n",
+ " # run_alphafold.py is a top-level script, not part of the package.\n",
+ " _sh(f'wget -q -O run_alphafold.py https://raw.githubusercontent.com'\n",
+ " f'/sokrypton/alphafold3/v{VERSION}/run_alphafold.py', 'fetching run_alphafold.py')\n",
+ " # haiku 0.0.17 still calls the moved jax.core.DropVar.\n",
+ " os.system(\"sed -i 's/jax.core.DropVar/jax.extend.core.DropVar/g' /usr/local/lib/python*/dist-packages/haiku/_src/jaxpr_info.py\")\n",
+ " import alphafold3 # confirms the install before anything depends on it\n",
+ " os.system('touch ALPHAFOLD3_READY')\n",
+ " print(f'Packages installed ({alphafold3.__file__}).')\n",
+ "\n",
+ "# tokamax's Triton kernels need more shared memory than Ada cards have, so\n",
+ "# restrict them to datacenter GPUs (A100 cc 8.0, H100 cc 9.0+).\n",
+ "try:\n",
+ " import tokamax\n",
+ " _gu = os.path.join(os.path.dirname(tokamax.__file__), '_src', 'gpu_utils.py')\n",
+ " _s = open(_gu).read()\n",
+ " _old = 'return float(device.compute_capability) >= 8.0'\n",
+ " _new = ('cc = float(device.compute_capability)\\n'\n",
+ " ' return cc == 8.0 or cc >= 9.0 # datacenter only; Ada/L4 lack shared memory')\n",
+ " if _old in _s:\n",
+ " open(_gu, 'w').write(_s.replace(_old, _new))\n",
+ " print('Patched tokamax: Triton restricted to datacenter GPUs (L4/Ada -> XLA).')\n",
+ "except Exception as _e:\n",
+ " print(f'(tokamax patch skipped: {_e})')\n",
+ "\n",
+ "# Weights, fetched in the background by the same code the run uses.\n",
+ "STAMP = f'WEIGHTS_DONE_{model}_{PRECISION}'\n",
+ "if IS_AF3 and not os.path.isfile(STAMP):\n",
+ " # DeepMind's own release, subject to the AlphaFold 3 terms of use, which\n",
+ " # run_alphafold prints at startup. A copy you already have in NATIVE_DIR is\n",
+ " # used as-is.\n",
+ " os.makedirs(NATIVE_DIR, exist_ok=True)\n",
+ " if glob.glob(f'{NATIVE_DIR}/*.bin.zst'):\n",
+ " open(STAMP, 'w').close()\n",
+ " print(f'Using the AlphaFold 3 parameters already in {NATIVE_DIR}/.')\n",
+ " else:\n",
+ " print('Downloading AlphaFold 3 parameters (~1 GB)...')\n",
+ " os.system(f'(wget -O {NATIVE_DIR}/af3.bin.zst \"{AF3_WEIGHTS_URL}\"'\n",
+ " f' > {STAMP}.log 2>&1 && touch {STAMP}) &')\n",
+ "elif not (IS_AF3 or os.path.isfile(STAMP)):\n",
+ " _script, _args = ('prefetch_af2.py', AF2_DIR) if IS_AF2 else (\n",
+ " 'prefetch_weights.py', f'{model} {PRECISION}')\n",
+ " print(f'Downloading {\"official AlphaFold 2 parameters (CC BY 4.0)\" if IS_AF2 else model} weights...')\n",
+ " with open(_script, 'w') as fh:\n",
+ " fh.write('import sys\\n'\n",
+ " 'from alphafold3.model import weights\\n'\n",
+ " + ('print(weights.ensure_af2_params(sys.argv[1]))\\n' if IS_AF2 else\n",
+ " 'print(weights.ensure_weights(sys.argv[1], None, precision=sys.argv[2]))\\n'))\n",
+ " os.system(f'(python {_script} {_args} > {STAMP}.log 2>&1 && touch {STAMP}) &')\n",
+ "\n",
+ "# /tmp is wiped with the VM, so a fresh session recompiles (~53 s); Drive survives.\n",
+ "CACHE_DIR = '/tmp/af3_cache'\n",
+ "if persist_cache_to_drive:\n",
+ " try:\n",
+ " from google.colab import drive\n",
+ " drive.mount('/content/drive')\n",
+ " CACHE_DIR = '/content/drive/MyDrive/.af3_cache'\n",
+ " os.makedirs(CACHE_DIR, exist_ok=True)\n",
+ " print(f'Compile cache: {CACHE_DIR} (survives this session)')\n",
+ " except Exception as _e:\n",
+ " print(f'(Drive mount failed, using {CACHE_DIR}: {_e})')\n",
+ "\n",
+ "\n",
+ "def _await(sentinel, limit=1200):\n",
+ " \"\"\"Wait for a background job, reporting its log if it never finishes.\"\"\"\n",
+ " t0 = time.time()\n",
+ " while not os.path.isfile(sentinel):\n",
+ " if time.time() - t0 > limit:\n",
+ " log = f'{sentinel}.log'\n",
+ " tail = open(log).read()[-1500:] if os.path.isfile(log) else '(no output captured)'\n",
+ " raise RuntimeError(f'{sentinel} did not appear within {limit} s. '\n",
+ " f'Tail of {log}:\\n{tail}')\n",
+ " time.sleep(5)\n",
+ " print(f'{sentinel} \\u2713 ({time.time() - t0:.0f} s)')\n",
+ "\n",
+ "\n",
+ "_await(STAMP)\n",
+ "\n",
+ "if IS_AF3 and os.path.getsize(f'{NATIVE_DIR}/af3.bin.zst') < 1_000_000:\n",
+ " raise RuntimeError('the AlphaFold 3 download is incomplete - re-run this cell.')\n",
+ "\n",
+ "print(f'Setup complete! Model: {model}.')\n",
+ "if model == 'chai1':\n",
+ " print('NOTE: chai-1 is running WITHOUT ESM2 embeddings, which are most of its token\\n'\n",
+ " ' features. Expect worse structures than chai-lab itself produces.')\n",
+ "print(f'Setup took {time.time() - _T0:.0f} s.')\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "id": "input",
+ "metadata": {
+ "cellView": "form",
+ "id": "input"
+ },
+ "outputs": [],
+ "source": [
+ "#@title Input sequences\n",
+ "import re, os, json, hashlib\n",
+ "\n",
+ "#@markdown ### Molecules\n",
+ "#@markdown Separate multiple chains within a box using `:` (extra colons are fine: `A::::B` == `A:B`). Leave a box empty if unused; full details in the Instructions cell.\n",
+ "protein = 'PIAQIHILEGRSDEQKETLIREVSEAISRSLDAPLTSVRVIITEMAKGHFGIGGELASK' #@param {type:\"string\"}\n",
+ "dna = '' #@param {type:\"string\"}\n",
+ "rna = '' #@param {type:\"string\"}\n",
+ "ligand_ccd = '' #@param {type:\"string\"}\n",
+ "ligand_smiles = '' #@param {type:\"string\"}\n",
+ "\n",
+ "#@markdown ### Run settings\n",
+ "jobname = 'test' #@param {type:\"string\"}\n",
+ "msa_mode = \"mmseqs2_server\" #@param [\"mmseqs2_server\", \"single_sequence\"]\n",
+ "seeds = '1' #@param {type:\"string\"}\n",
+ "on_existing = \"overwrite\" #@param [\"overwrite\", \"skip\"]\n",
+ "#@markdown - `msa_mode`: `single_sequence` skips the MSA (faster, lower accuracy).\n",
+ "#@markdown - `seeds`: comma-separated, e.g. `1,2,3`.\n",
+ "#@markdown - `on_existing`: `overwrite` replaces this job's previous results; `skip` keeps them.\n",
+ "\n",
+ "# Headless overrides -- see the install cell.\n",
+ "import json as _json, os as _os\n",
+ "for _k, _v in _json.loads(_os.environ.get('AF3_NB_OVERRIDES', '{}')).items():\n",
+ " if _k in globals():\n",
+ " globals()[_k] = _v\n",
+ " print(f'override: {_k} = {_v!r}')\n",
+ "\n",
+ "# Split a box into entries: collapse colon runs, drop whitespace, skip empties\n",
+ "def split_entries(s):\n",
+ " s = re.sub(r':+', ':', s).strip(':')\n",
+ " return [e for e in (''.join(tok.split()) for tok in s.split(':')) if e]\n",
+ "\n",
+ "prot_seqs = [e.upper() for e in split_entries(protein)]\n",
+ "dna_seqs = [e.upper() for e in split_entries(dna)]\n",
+ "rna_seqs = [e.upper() for e in split_entries(rna)]\n",
+ "ccd_codes = [e.upper() for e in split_entries(ligand_ccd)]\n",
+ "smiles_strs = split_entries(ligand_smiles) # case-sensitive: leave as typed\n",
+ "\n",
+ "# Fetch chemical definitions for the components this input names, from\n",
+ "# files.rcsb.org (~0.6 s). A code that is not fetched raises when folding.\n",
+ "with open('prefetch_ccd.py', 'w') as fh:\n",
+ " fh.write('import sys, os, importlib.metadata as md\\n'\n",
+ " 'from alphafold3.constants import ccd_fetch\\n'\n",
+ " 'root = os.path.dirname(md.distribution(\"alphafold3-colabfold\")'\n",
+ " '.locate_file(\"alphafold3\"))\\n'\n",
+ " 'conv = os.path.join(root, \"alphafold3\", \"constants\", \"converters\")\\n'\n",
+ " 'os.makedirs(conv, exist_ok=True)\\n'\n",
+ " 'ccd_fetch.write_pickles(ccd_fetch.codes_for_input(extra=sys.argv[1:]),\\n'\n",
+ " ' os.path.join(conv, \"ccd.pickle\"),\\n'\n",
+ " ' os.path.join(conv, \"chemical_component_sets.pickle\"),\\n'\n",
+ " ' libcifpp_dir=os.path.join(root, \"share\", \"libcifpp\"))\\n')\n",
+ "print(f'Fetching the CCD: 35 standard residues'\n",
+ " + (f' + {\", \".join(ccd_codes)}' if ccd_codes else '') + ' ...')\n",
+ "if os.system('python prefetch_ccd.py ' + ' '.join(ccd_codes)) != 0:\n",
+ " raise RuntimeError('could not build the CCD tables; see the output above')\n",
+ "\n",
+ "# Build AF3 chain entities (IDs A, B, C, ... in canonical order)\n",
+ "CHAIN_IDS = list('ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz')\n",
+ "chains, prot_groups, idx = [], {}, 0\n",
+ "\n",
+ "for seq in prot_seqs:\n",
+ " cid = CHAIN_IDS[idx]; idx += 1\n",
+ " if seq in prot_groups: # merge identical seqs -> homo-oligomer\n",
+ " ent = prot_groups[seq]\n",
+ " ids = ent['id'] if isinstance(ent['id'], list) else [ent['id']]\n",
+ " ent['id'] = ids + [cid]\n",
+ " else:\n",
+ " ent = {'id': cid, 'sequence': seq, 'templates': []}\n",
+ " if msa_mode == 'single_sequence':\n",
+ " ent.update({'unpairedMsa': f'>query\\n{seq}\\n', 'pairedMsa': ''})\n",
+ " prot_groups[seq] = ent\n",
+ " chains.append({'protein': ent})\n",
+ "\n",
+ "for seq in rna_seqs:\n",
+ " c = {'id': CHAIN_IDS[idx], 'sequence': seq}\n",
+ " if msa_mode == 'single_sequence':\n",
+ " c['unpairedMsa'] = f'>query\\n{seq}\\n'\n",
+ " chains.append({'rna': c}); idx += 1\n",
+ "\n",
+ "for seq in dna_seqs:\n",
+ " chains.append({'dna': {'id': CHAIN_IDS[idx], 'sequence': seq}}); idx += 1\n",
+ "\n",
+ "for code in ccd_codes:\n",
+ " chains.append({'ligand': {'id': CHAIN_IDS[idx], 'ccdCodes': [code]}}); idx += 1\n",
+ "\n",
+ "for smiles in smiles_strs:\n",
+ " chains.append({'ligand': {'id': CHAIN_IDS[idx], 'smiles': smiles}}); idx += 1\n",
+ "\n",
+ "if not chains:\n",
+ " raise ValueError('No valid input found - fill in at least one box.')\n",
+ "\n",
+ "# Seeds: pull out integers regardless of separators, dedupe, default to [1]\n",
+ "seed_list = []\n",
+ "for tok in re.findall(r'\\d+', seeds):\n",
+ " v = int(tok)\n",
+ " if v not in seed_list:\n",
+ " seed_list.append(v)\n",
+ "if not seed_list:\n",
+ " seed_list = [1]\n",
+ "\n",
+ "# Deterministic, lower-cased job name from inputs+seeds.\n",
+ "# Same input+seeds -> same folder (so re-runs reuse it instead of piling up).\n",
+ "# Lower-cased to match run_alphafold.py's sanitised_name() output directory.\n",
+ "def _flat(mol):\n",
+ " if 'sequence' in mol: return mol['sequence']\n",
+ " if 'ccdCodes' in mol: return ','.join(mol['ccdCodes'])\n",
+ " return mol.get('smiles', '?')\n",
+ "flat = ':'.join(_flat(list(c.values())[0]) for c in chains) + '|seeds=' + ','.join(map(str, seed_list))\n",
+ "basejob = (re.sub(r'\\W+', '', ''.join(jobname.split())) or 'job').lower()\n",
+ "jobname = basejob + '_' + hashlib.sha1(flat.encode()).hexdigest()[:5]\n",
+ "\n",
+ "# Input JSON goes to a temp dir; ALL results land in ONE folder: af3_output//\n",
+ "INPUT_DIR = '/tmp/af3_inputs'\n",
+ "OUTPUT_DIR = 'af3_output'\n",
+ "job_dir = f'{OUTPUT_DIR}/{jobname}'\n",
+ "\n",
+ "fold_input = {\n",
+ " 'name': jobname,\n",
+ " 'sequences': chains,\n",
+ " 'modelSeeds': seed_list,\n",
+ " 'dialect': 'alphafold3',\n",
+ " 'version': 1,\n",
+ "}\n",
+ "os.makedirs(INPUT_DIR, exist_ok=True)\n",
+ "json_path = f'{INPUT_DIR}/{jobname}.json'\n",
+ "with open(json_path, 'w') as f:\n",
+ " json.dump(fold_input, f, indent=2)\n",
+ "\n",
+ "print(f'Job \"{jobname}\" -> results will be written to {job_dir}/')\n",
+ "fold_input\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "id": "run",
+ "metadata": {
+ "cellView": "form",
+ "id": "run"
+ },
+ "outputs": [],
+ "source": [
+ "#@title Run the model\n",
+ "import os, shutil, subprocess, glob, time\n",
+ "_T0 = time.time()\n",
+ "\n",
+ "#@markdown Defaults match AlphaFold 3; raise only if needed.\n",
+ "num_recycles = 10 #@param {type:\"integer\"}\n",
+ "num_diffusion_samples = 5 #@param {type:\"integer\"}\n",
+ "#@markdown - `num_recycles`: refinement passes; more helps hard targets, costs time.\n",
+ "#@markdown - `num_diffusion_samples`: structures per seed, so total = seeds x samples.\n",
+ "\n",
+ "# Headless overrides -- see the install cell.\n",
+ "import json as _json, os as _os\n",
+ "for _k, _v in _json.loads(_os.environ.get('AF3_NB_OVERRIDES', '{}')).items():\n",
+ " if _k in globals():\n",
+ " globals()[_k] = _v\n",
+ " print(f'override: {_k} = {_v!r}')\n",
+ "\n",
+ "num_recycles = max(1, int(num_recycles))\n",
+ "num_diffusion_samples = max(1, int(num_diffusion_samples))\n",
+ "\n",
+ "os.makedirs(OUTPUT_DIR, exist_ok=True)\n",
+ "\n",
+ "# Re-run policy (one folder per job, no timestamped duplicates):\n",
+ "# overwrite -> wipe this job's folder and recompute\n",
+ "# skip -> if a finished result (.cif) is already there, don't recompute\n",
+ "have_results = os.path.isdir(job_dir) and any(f.endswith('.cif') for f in os.listdir(job_dir))\n",
+ "run_it = not (on_existing == 'skip' and have_results)\n",
+ "if run_it:\n",
+ " shutil.rmtree(job_dir, ignore_errors=True) # start clean so exactly one folder is produced\n",
+ "\n",
+ "# Attention implementation and XLA flags, chosen from the device.\n",
+ "def detect_device():\n",
+ " try:\n",
+ " out = subprocess.run(\n",
+ " ['nvidia-smi', '--query-gpu=compute_cap', '--format=csv,noheader'],\n",
+ " capture_output=True, text=True, timeout=15)\n",
+ " caps = [float(x) for x in out.stdout.split() if x.strip()]\n",
+ " if caps:\n",
+ " return 'gpu', min(caps)\n",
+ " except Exception:\n",
+ " pass\n",
+ " return 'cpu', None\n",
+ "\n",
+ "device, cap = detect_device()\n",
+ "nojit = False\n",
+ "xla_flags = [] # extra XLA flags to export for this device (per AlphaFold 3's guidance)\n",
+ "\n",
+ "if device == 'cpu':\n",
+ " flash_impl = 'xla'\n",
+ " nojit = True\n",
+ " print('No GPU detected - running on CPU with XLA attention + --nojit (slow, but avoids the compile).')\n",
+ "elif cap < 8.0:\n",
+ " # T4 / V100: XLA attention, and no custom-kernel fusion pass.\n",
+ " flash_impl = 'xla'\n",
+ " xla_flags = ['--xla_disable_hlo_passes=custom-kernel-fusion-rewriter']\n",
+ " print(f'Pre-Ampere GPU (compute capability {cap}) - XLA attention + custom-kernel fusion disabled.')\n",
+ "elif 8.0 < cap < 9.0:\n",
+ " # L4 / Ada: limited shared memory, so no Triton kernels -- XLA and cuBLAS.\n",
+ " flash_impl = 'xla'\n",
+ " xla_flags = ['--xla_gpu_enable_triton_gemm=false']\n",
+ " print(f'Ada/consumer GPU (compute capability {cap}) - XLA attention + Triton GEMM disabled (shared-memory limit).')\n",
+ "else:\n",
+ " # A100 / H100: Triton flash attention, Triton GEMM off per AlphaFold 3.\n",
+ " flash_impl = 'triton'\n",
+ " xla_flags = ['--xla_gpu_enable_triton_gemm=false']\n",
+ " print(f'Datacenter GPU (compute capability {cap}) - Triton flash attention + Triton GEMM disabled.')\n",
+ "\n",
+ "# Export XLA flags so the child shell (and JAX inside it) inherit them.\n",
+ "cur = os.environ.get('XLA_FLAGS', '')\n",
+ "for f in xla_flags:\n",
+ " if f not in cur:\n",
+ " cur = (cur + ' ' + f).strip()\n",
+ "if cur:\n",
+ " os.environ['XLA_FLAGS'] = cur\n",
+ "\n",
+ "print('XLA_FLAGS =', os.environ.get('XLA_FLAGS', '(unset)'))\n",
+ "\n",
+ "# Ported models find their own cache, so --model_dir is only for AF2 and AF3.\n",
+ "print(f'Model: {model}')\n",
+ "\n",
+ "cmd = [\n",
+ " 'python', 'run_alphafold.py',\n",
+ " f'--json_path={json_path}',\n",
+ " f'--model={model}',\n",
+ " '--norun_data_pipeline',\n",
+ " f'--output_dir={OUTPUT_DIR}',\n",
+ " f'--cache_dir={CACHE_DIR}',\n",
+ " '--force_output_dir', # reuse af3_output// instead of a timestamped copy\n",
+ " f'--flash_attention_implementation={flash_impl}',\n",
+ " f'--num_recycles={num_recycles}',\n",
+ " f'--num_diffusion_samples={num_diffusion_samples}',\n",
+ "]\n",
+ "if msa_mode == 'mmseqs2_server':\n",
+ " cmd.append('--use_msa_server')\n",
+ "# chai-1 and ESMFold2 fold from a language model, downloaded on first use.\n",
+ "# Without it they are a different model, not a slightly worse one.\n",
+ "if model == 'chai1' or model.startswith('esmfold2'):\n",
+ " cmd.append('--use_esm_embeddings')\n",
+ "if nojit:\n",
+ " cmd.append('--nojit')\n",
+ "if IS_AF3 or IS_AF2:\n",
+ " cmd.append(f'--model_dir={AF2_DIR if IS_AF2 else NATIVE_DIR}')\n",
+ "\n",
+ "cmd = ' '.join(cmd)\n",
+ "if run_it:\n",
+ " print(cmd)\n",
+ " # Popen rather than `!`: streams the output and gives an exit status.\n",
+ " _p = subprocess.Popen(cmd, shell=True, stdout=subprocess.PIPE,\n",
+ " stderr=subprocess.STDOUT, text=True, bufsize=1)\n",
+ " for _line in _p.stdout:\n",
+ " print(_line, end='')\n",
+ " _rc = _p.wait()\n",
+ " _cifs = glob.glob(f'{job_dir}/**/*.cif', recursive=True)\n",
+ " if _rc != 0 or not _cifs:\n",
+ " raise RuntimeError(\n",
+ " f'the fold FAILED (exit {_rc}, {len(_cifs)} structures written). '\n",
+ " 'The output above is the whole story; scroll up for the error.')\n",
+ " print(f'\\nDone -> {job_dir}/ ({len(_cifs)} structures, '\n",
+ " f'{time.time() - _T0:.0f} s)')\n",
+ "else:\n",
+ " print(f'Skipping: results already exist in {job_dir}/ (set on_existing=overwrite to recompute).')\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "id": "display3d",
+ "metadata": {
+ "cellView": "form",
+ "id": "display3d"
+ },
+ "outputs": [],
+ "source": [
+ "#@title Display structures + PAE (py2Dmol)\n",
+ "import csv, glob, os, json\n",
+ "import numpy as np\n",
+ "import py2Dmol\n",
+ "\n",
+ "load_as_frames = True #@param {type:\"boolean\"}\n",
+ "viewer_size = 400\n",
+ "#@markdown All predicted models load together, best first (**rank_1, rank_2, ...**), each with its own PAE.\n",
+ "#@markdown - `load_as_frames` **off** -> pick a model from the dropdown.\n",
+ "#@markdown - `load_as_frames` **on** -> models become frames you can play through (press play / drag the slider).\n",
+ "#@markdown - The interactive PAE matrix sits beside the structure; click or drag-box on it to highlight residues.\n",
+ "\n",
+ "# All models in rank order (best first): from the ranking CSV, fall back to globbing.\n",
+ "def collect_models():\n",
+ " ranking_csv = f'{job_dir}/{jobname}_ranking_scores.csv'\n",
+ " cifs = []\n",
+ " if os.path.exists(ranking_csv):\n",
+ " rows = []\n",
+ " with open(ranking_csv) as f:\n",
+ " for r in csv.DictReader(f):\n",
+ " rows.append((float(r['ranking_score']), int(r['seed']), int(r['sample'])))\n",
+ " for _, seed, sample in sorted(rows, reverse=True):\n",
+ " d = f'{job_dir}/seed-{seed}_sample-{sample}'\n",
+ " hit = sorted(glob.glob(f'{d}/*_model.cif')) or sorted(glob.glob(f'{d}/*.cif'))\n",
+ " if hit:\n",
+ " cifs.append(hit[0])\n",
+ " if not cifs:\n",
+ " cifs = (sorted(glob.glob(f'{job_dir}/**/*_model.cif', recursive=True))\n",
+ " or sorted(glob.glob(f'{job_dir}/**/*.cif', recursive=True)))\n",
+ " return cifs\n",
+ "\n",
+ "# Per-model PAE: confidences.json next to the CIF, else the top-level one.\n",
+ "def load_pae(cif):\n",
+ " d = os.path.dirname(cif)\n",
+ " cands = [p for p in glob.glob(f'{d}/*_confidences.json')\n",
+ " if 'summary' not in os.path.basename(p)]\n",
+ " if not cands:\n",
+ " top = f'{job_dir}/{jobname}_confidences.json'\n",
+ " cands = [top] if os.path.exists(top) else []\n",
+ " if cands:\n",
+ " pae = json.load(open(cands[0])).get('pae')\n",
+ " if pae is not None:\n",
+ " return np.asarray(pae, dtype=float)\n",
+ " return None\n",
+ "\n",
+ "cifs = collect_models()\n",
+ "if not cifs:\n",
+ " raise FileNotFoundError(f'No model CIFs found in {job_dir}/')\n",
+ "print(f'Loaded {len(cifs)} model(s) from {job_dir}/'\n",
+ " + (' (frames - press play)' if load_as_frames else ' (use the dropdown to switch)'))\n",
+ "\n",
+ "viewer = py2Dmol.view(size=(viewer_size, viewer_size),\n",
+ " pae=True, autoplay=load_as_frames,\n",
+ " style=\"cartoon\")\n",
+ "for i, cif in enumerate(cifs, start=1):\n",
+ " pae = load_pae(cif)\n",
+ " if load_as_frames:\n",
+ " viewer.add_pdb(cif, name='models', paes=pae) # same name -> frames (play through)\n",
+ " else:\n",
+ " viewer.add_pdb(cif, name=f'rank_{i}', paes=pae) # distinct names -> dropdown of objects\n",
+ "viewer.show()\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "id": "plots",
+ "metadata": {
+ "cellView": "form",
+ "id": "plots"
+ },
+ "outputs": [],
+ "source": [
+ "#@title Quality metrics and plots\n",
+ "import json, os\n",
+ "import numpy as np\n",
+ "import matplotlib.pyplot as plt\n",
+ "\n",
+ "conf_path = f'{OUTPUT_DIR}/{jobname}/{jobname}_confidences.json'\n",
+ "summ_path = f'{OUTPUT_DIR}/{jobname}/{jobname}_summary_confidences.json'\n",
+ "\n",
+ "with open(conf_path) as f:\n",
+ " conf = json.load(f)\n",
+ "with open(summ_path) as f:\n",
+ " summ = json.load(f)\n",
+ "\n",
+ "plddts = np.array(conf.get('atom_plddts', conf.get('token_plddts', [])), dtype=float)\n",
+ "plddt_chain_ids = conf.get('atom_chain_ids', conf.get('token_chain_ids', [])) # pLDDT is per-ATOM\n",
+ "token_chain_ids = conf.get('token_chain_ids', []) # PAE is per-TOKEN\n",
+ "pae = np.array(conf.get('pae', []), dtype=float)\n",
+ "\n",
+ "# \u2500\u2500 Summary (ipTM is None for single-chain jobs \u2014 guard before formatting) \u2500\n",
+ "def fmt(v):\n",
+ " return f'{v:.3f}' if isinstance(v, (int, float)) else 'n/a'\n",
+ "\n",
+ "mean_plddt = summ.get('mean_plddt')\n",
+ "if mean_plddt is None and plddts.size:\n",
+ " mean_plddt = float(np.mean(plddts))\n",
+ "iptm = summ.get('iptm')\n",
+ "\n",
+ "print('=' * 38)\n",
+ "print(f'Mean pLDDT : {fmt(mean_plddt)}')\n",
+ "print(f'pTM : {fmt(summ.get(\"ptm\"))}')\n",
+ "print(f'ipTM : {fmt(iptm)}' + (' (single chain \u2014 no interface)' if iptm is None else ''))\n",
+ "print(f'Ranking score : {fmt(summ.get(\"ranking_score\"))}')\n",
+ "print('=' * 38)\n",
+ "\n",
+ "# \u2500\u2500 Plots \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n",
+ "has_pae = pae.ndim == 2 and pae.size > 0\n",
+ "ncols = 2 if has_pae else 1\n",
+ "fig, axes = plt.subplots(1, ncols, figsize=(13 if has_pae else 6.5, 4))\n",
+ "axes = np.atleast_1d(axes)\n",
+ "\n",
+ "# pLDDT per residue \u2014 a line (coloured per chain when there is more than one)\n",
+ "ax = axes[0]\n",
+ "x = np.arange(len(plddts))\n",
+ "xmax = max(len(plddts) - 1, 1)\n",
+ "ax.set_xlim(0, xmax)\n",
+ "ax.set_ylim(0, 100)\n",
+ "\n",
+ "unique_chains = list(dict.fromkeys(plddt_chain_ids))\n",
+ "if len(unique_chains) > 1:\n",
+ " colors = plt.cm.tab10(np.linspace(0, 1, len(unique_chains)))\n",
+ " tcid = np.array(plddt_chain_ids)\n",
+ " for ch, col in zip(unique_chains, colors):\n",
+ " y = np.where(tcid == ch, plddts, np.nan) # NaN gaps keep chains as separate lines\n",
+ " ax.plot(x, y, lw=1.5, color=col, label=f'Chain {ch}')\n",
+ " for b in [i for i in range(1, len(plddt_chain_ids)) if plddt_chain_ids[i] != plddt_chain_ids[i-1]]:\n",
+ " ax.axvline(b - 0.5, color='grey', lw=0.6, alpha=0.5)\n",
+ " ax.legend(loc='lower right', fontsize=8)\n",
+ "else:\n",
+ " ax.plot(x, plddts, lw=1.5, color='#1f77b4')\n",
+ "\n",
+ "for y in (50, 70, 90):\n",
+ " ax.axhline(y, ls='--', lw=0.7, color='grey', alpha=0.5)\n",
+ " ax.text(xmax, y, f' {y}', va='center', ha='left', fontsize=7, color='grey')\n",
+ "ax.set_xlabel('Atom')\n",
+ "ax.set_ylabel('pLDDT')\n",
+ "ax.set_title('Predicted pLDDT per atom')\n",
+ "\n",
+ "# PAE matrix\n",
+ "if has_pae:\n",
+ " ax = axes[1]\n",
+ " im = ax.imshow(pae, cmap='bwr', vmin=0, vmax=30, interpolation='nearest')\n",
+ " plt.colorbar(im, ax=ax, fraction=0.046, pad=0.04, label='PAE (\u00c5)')\n",
+ " if token_chain_ids:\n",
+ " for b in [i for i in range(1, len(token_chain_ids)) if token_chain_ids[i] != token_chain_ids[i-1]]:\n",
+ " ax.axhline(b - 0.5, c='black', lw=0.8)\n",
+ " ax.axvline(b - 0.5, c='black', lw=0.8)\n",
+ " ax.set_xlabel('Scored residue')\n",
+ " ax.set_ylabel('Aligned residue')\n",
+ " ax.set_title('Predicted Aligned Error (PAE)')\n",
+ "\n",
+ "plt.tight_layout()\n",
+ "plt.show()\n"
+ ]
+ },
+ {
+ "cell_type": "code",
+ "execution_count": null,
+ "id": "download",
+ "metadata": {
+ "cellView": "form",
+ "id": "download"
+ },
+ "outputs": [],
+ "source": [
+ "#@title Download results\n",
+ "from google.colab import files\n",
+ "import os\n",
+ "\n",
+ "results_zip = f'{jobname}.result.zip'\n",
+ "os.system(f'zip -r {results_zip} {OUTPUT_DIR}/{jobname}')\n",
+ "files.download(results_zip)\n"
+ ]
+ },
+ {
+ "cell_type": "markdown",
+ "id": "instructions",
+ "metadata": {
+ "id": "instructions"
+ },
+ "source": [
+ "# Instructions \n",
+ "\n",
+ "**Quick start:** pick a **model** in the install cell, fill in the sequence(s), then\n",
+ "**Runtime \u2192 Run all**. The first run downloads that model's weights; later runs reuse\n",
+ "them, and each model has its own cache so switching back is instant.\n",
+ "\n",
+ "---\n",
+ "\n",
+ "## Models\n",
+ "\n",
+ "Ported weights come from [sokrypton/af3-any-model](https://huggingface.co/sokrypton/af3-any-model).\n",
+ "\"Tower\" is a protein language model fetched separately on first use.\n",
+ "\n",
+ "| model | weights | licence | notes |\n",
+ "|---|---|---|---|\n",
+ "| `openbind0` | [OpenFold3 v0.5.0 \"OpenBind\"](https://github.com/aqlaboratory/openfold-3/releases/tag/v0.5.0) (AlQuraishi Lab) | Apache-2.0 | The current release, and the default here. |\n",
+ "| `openfold3` | [OpenFold3 preview-2](https://github.com/aqlaboratory/openfold) (AlQuraishi Lab) | Apache-2.0 | The earlier preview, kept because earlier results used it. |\n",
+ "| `boltz2` | [Boltz-2](https://github.com/jwohlwend/boltz) (Wohlwend et al.) | MIT | Strong on ligands; keeps a modified residue as one token. |\n",
+ "| `protenix2` | [Protenix-v2](https://github.com/bytedance/Protenix) (ByteDance) | Apache-2.0 | The widest trunk here (pair 256), so the slowest. |\n",
+ "| `rosettafold3` | [RoseTTAFold3](https://github.com/RosettaCommons/foundry) (RosettaCommons) | BSD-3-Clause | Carries chirality features; handles D-amino acids. |\n",
+ "| `chai1` | [chai-1](https://github.com/chaidiscovery/chai-lab) (Chai Discovery) | Apache-2.0 | Folds from ESM2 3B, fetched and run automatically. |\n",
+ "| `intellifold2` | [IntelliFold-v2](https://huggingface.co/intelligenAI/intellifold) (IntelliGen-AI) | Apache-2.0 | Widened channels (pair 512), largest ported download. |\n",
+ "| `opendde` | [OpenDDE](https://huggingface.co/aurekaresearch/OpenDDE) (Aureka Research) | Apache-2.0 | Runs its diffusion on an expanded structural-token set. |\n",
+ "| `esmfold2` | [ESMFold2](https://huggingface.co/biohub/ESMFold2) (Chan Zuckerberg Biohub) | MIT | Folds from ESM-C instead of an MSA \u2014 single sequence, no search. |\n",
+ "| `esmfold2_lm600m` | ESMFold2, 600M tower | MIT | No confidence head. |\n",
+ "| `esmfold2_lm300m` | ESMFold2, 300M tower | MIT | No confidence head. |\n",
+ "| `af2_ptm` | AlphaFold 2 monomer pTM (DeepMind) | CC BY 4.0 | **Protein only** \u2014 a ligand or nucleotide raises rather than quietly folding the rest. Templates use the model_1/model_2 parameter sets. |\n",
+ "| `af2_multimer` | AlphaFold 2 multimer v3 (DeepMind) | CC BY 4.0 | Protein only, as above. |\n",
+ "| `alphafold3` | Google DeepMind's own parameters | [AF3 terms of use](https://github.com/google-deepmind/alphafold3/blob/main/WEIGHTS_TERMS_OF_USE.md) | DeepMind's public release, downloaded on first use (~1 GB). Its terms govern the weights and the outputs; run_alphafold prints them at startup. |\n",
+ "---\n",
+ "\n",
+ "## Input\n",
+ "\n",
+ "Each molecule type has its own box; within a box, separate chains with `:`.\n",
+ "\n",
+ "| box | contents | example |\n",
+ "|---|---|---|\n",
+ "| **protein** | amino-acid sequence(s) | `MKTAY...` or `SEQ1:SEQ2` |\n",
+ "| **dna** | DNA sequence(s) | `CGCGAATTCGCG` |\n",
+ "| **rna** | RNA sequence(s) | `GCGGAUUUA` |\n",
+ "| **ligand_ccd** | ligand(s) by PDB CCD code | `ATP:MG:HEM` |\n",
+ "| **ligand_smiles** | ligand(s) by SMILES | `CC(=O)Oc1ccccc1C(=O)O` |\n",
+ "\n",
+ "Mix boxes freely to build a complex. Chain IDs A, B, C\u2026 follow AlphaFold 3's canonical\n",
+ "order (protein \u2192 RNA \u2192 DNA \u2192 ligand). Identical protein sequences are merged, so\n",
+ "`SEQ:SEQ` is a homodimer. Sequences and CCD codes are upper-cased; **SMILES are left\n",
+ "as typed**. Whitespace and extra colons are forgiven (`SEQ1::::SEQ2` = `SEQ1:SEQ2`) \u2014\n",
+ "which is also why an atom-mapped SMILES containing `:` needs a raw AF3 JSON instead.\n",
+ "\n",
+ "**seeds**: comma-separated, one prediction each (`1,2,3`). Junk and duplicates are\n",
+ "dropped. **msa_mode**: `mmseqs2_server` queries the public\n",
+ "[ColabFold](https://colabfold.mmseqs.com/) API (protein only \u2014 RNA/DNA always run\n",
+ "MSA-free); `single_sequence` skips it, faster and less accurate.\n",
+ "\n",
+ "## Output\n",
+ "\n",
+ "| file | contents |\n",
+ "|---|---|\n",
+ "| `*.cif` | Best-ranked structure. B-factor = pLDDT (0\u2013100). |\n",
+ "| `*_confidences.json` | Per-residue pLDDT, PAE matrix, contact probabilities. |\n",
+ "| `*_summary_confidences.json` | Mean pLDDT, pTM, ipTM, ranking score. |\n",
+ "| `*_ranking_scores.csv` | Every seed \u00d7 sample combination. |\n",
+ "| `seed-N_sample-M/` | One directory per prediction. |\n",
+ "| `TERMS_OF_USE.md` | The licence for whichever weights you ran. |\n",
+ "\n",
+ "pLDDT above 90 is very high, 70\u201390 reliable backbone, 50\u201370 doubtful, below 50 likely\n",
+ "disordered or wrong. Lower PAE means two residues are confidently placed *relative to\n",
+ "each other*, which is what to read for an interface. ipTM above 0.8 is a well-defined\n",
+ "complex interface, and is `n/a` for a single chain \u2014 there is no interface to score.\n",
+ "\n",
+ "## Troubleshooting\n",
+ "\n",
+ "**OOM**: shorter sequence, or a larger GPU (`Runtime \u2192 Change runtime type`).\n",
+ "**MSA server timeout**: the public server is rate-limited \u2014 retry, or use\n",
+ "`single_sequence`. **Download popup blocked**: disable your ad blocker.\n",
+ "\n",
+ "## Licence\n",
+ "\n",
+ "The AlphaFold 3 **source code** is [Apache 2.0](https://www.apache.org/licenses/LICENSE-2.0);\n",
+ "the **weights** are each their own, as listed in the model table. Outputs from the seven\n",
+ "Apache/MIT/BSD-licensed ported models are **not** subject to DeepMind's AlphaFold 3 Output\n",
+ "Terms of Use and may be used freely, including commercially. `alphafold3` is the exception:\n",
+ "its parameters and outputs carry DeepMind's own terms. Every run writes a\n",
+ "`TERMS_OF_USE.md` naming the licence that actually applies to it.\n",
+ "\n",
+ "## Bugs / feedback\n",
+ "\n",
+ "https://github.com/sokrypton/alphafold3/issues\n"
+ ]
+ }
+ ],
+ "metadata": {
+ "accelerator": "GPU",
+ "colab": {
+ "gpuType": "T4",
+ "provenance": [],
+ "include_colab_link": true
+ },
+ "kernelspec": {
+ "display_name": "Python 3 (ipykernel)",
+ "language": "python",
+ "name": "python3"
+ },
+ "language_info": {
+ "codemirror_mode": {
+ "name": "ipython",
+ "version": 3
+ },
+ "file_extension": ".py",
+ "mimetype": "text/x-python",
+ "name": "python",
+ "nbconvert_exporter": "python",
+ "pygments_lexer": "ipython3",
+ "version": "3.10.12"
+ }
+ },
+ "nbformat": 4,
+ "nbformat_minor": 5
+}
\ No newline at end of file
diff --git a/colabfold/batch.py b/colabfold/batch.py
index e0359d073..7d0d0bfcf 100644
--- a/colabfold/batch.py
+++ b/colabfold/batch.py
@@ -38,6 +38,14 @@
try:
import alphafold
except ModuleNotFoundError:
+ if "--version" in sys.argv[1:]:
+ # `colabfold_batch --version` must still answer in a base install
+ # (without the `alphafold` extra). argparse would handle the flag, but
+ # main() is never reached because the imports below fail first.
+ from colabfold.utils import get_version
+
+ print(f"{os.path.basename(sys.argv[0])} {get_version()}")
+ sys.exit(0)
raise RuntimeError(
"\n\nalphafold is not installed. Please run `pip install colabfold[alphafold]`\n"
)
@@ -68,6 +76,7 @@
NO_GPU_FOUND,
CIF_REVISION_DATE,
get_commit,
+ get_version,
setup_logging,
CFMMCIFIO,
AF3Utils,
@@ -1814,6 +1823,7 @@ def generate_af3_input(
def main():
parser = ArgumentParser(formatter_class=ArgumentDefaultsHelpFormatter)
+ parser.add_argument("--version", action="version", version=f"%(prog)s {get_version()}")
parser.add_argument(
"input",
default="input",
@@ -2233,10 +2243,7 @@ def comma_separated_list(arg_string):
setup_logging(Path(args.results).joinpath("log.txt"), verbose=args.debug_logging)
- version = importlib_metadata.version("colabfold")
- commit = get_commit()
- if commit:
- version += f" ({commit})"
+ version = get_version()
logger.info(f"Running colabfold {version}")
diff --git a/colabfold/mmseqs/search.py b/colabfold/mmseqs/search.py
index 9c3be7f03..530be8902 100644
--- a/colabfold/mmseqs/search.py
+++ b/colabfold/mmseqs/search.py
@@ -14,7 +14,7 @@
from typing import List, Union
from colabfold.input import get_queries, msa_to_str, safe_filename
-from colabfold.utils import AF3Utils
+from colabfold.utils import AF3Utils, get_version
logger = logging.getLogger(__name__)
@@ -291,6 +291,7 @@ def mmseqs_search_pair(
def main():
parser = ArgumentParser(formatter_class=ArgumentDefaultsHelpFormatter)
+ parser.add_argument("--version", action="version", version=f"%(prog)s {get_version()}")
parser.add_argument(
"query",
type=Path,
diff --git a/colabfold/mmseqs/split_msas.py b/colabfold/mmseqs/split_msas.py
index b0e6e67f8..5c0eb80da 100644
--- a/colabfold/mmseqs/split_msas.py
+++ b/colabfold/mmseqs/split_msas.py
@@ -8,6 +8,8 @@
from tqdm import tqdm
+from colabfold.utils import get_version
+
logger = logging.getLogger(__name__)
@@ -36,6 +38,7 @@ def main():
parser = ArgumentParser(
description="Take an a3m database from the colabdb search and turn it into a folder of a3m files"
)
+ parser.add_argument("--version", action="version", version=f"%(prog)s {get_version()}")
parser.add_argument(
"search_folder",
help="The search folder in which you ran colabfold_search with the final.a3m",
diff --git a/colabfold/utils.py b/colabfold/utils.py
index 0d3314f21..1345fa216 100644
--- a/colabfold/utils.py
+++ b/colabfold/utils.py
@@ -77,6 +77,15 @@ def get_commit() -> Optional[str]:
return direct_url["vcs_info"]["commit_id"]
+def get_version() -> str:
+ """Installed colabfold version, with the git commit appended when installed from VCS."""
+ version = distribution("colabfold").version
+ commit = get_commit()
+ if commit:
+ version += f" ({commit})"
+ return version
+
+
# Copied from Bio.PDB to override _save_dict method
# https://github.com/biopython/biopython/blob/biopython-179/Bio/PDB/mmcifio.py
# We add poly_seq and revision_date so that AF2 can read these cif files