From 7f1683cdc4a7f584b451a3dca92a3cac539eae63 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Muras?= Date: Tue, 1 Sep 2026 15:12:01 +0200 Subject: [PATCH 1/5] Add generic `attributes` JSON column to machines for backend-agnostic per-machine metadata --- lib/cuckoo/core/data/machines.py | 34 ++++++++++++++ lib/cuckoo/core/database.py | 2 +- .../versions/4. add_machine_attributes.py | 47 +++++++++++++++++++ 3 files changed, 82 insertions(+), 1 deletion(-) create mode 100644 utils/db_migration/versions/4. add_machine_attributes.py diff --git a/lib/cuckoo/core/data/machines.py b/lib/cuckoo/core/data/machines.py index 6f6127f38a3..5568414aff8 100644 --- a/lib/cuckoo/core/data/machines.py +++ b/lib/cuckoo/core/data/machines.py @@ -23,12 +23,15 @@ Select, String, ) + from sqlalchemy.dialects.postgresql import JSONB from sqlalchemy.orm import ( Mapped, + flag_modified, mapped_column, relationship, selectinload, ) + from sqlalchemy.types import JSON except ImportError: # pragma: no cover raise CuckooDependencyError("Unable to import sqlalchemy (install with `poetry install`)") @@ -63,6 +66,15 @@ class Machine(Base): resultserver_port: Mapped[str] = mapped_column(String(255), nullable=False) reserved: Mapped[bool] = mapped_column(Boolean(), nullable=False, default=False) + # Generic key/value bag for per-machine semi-structured data. + # Use get_attribute / set_attribute for reads and writes. Never mutate the dict in place — in-place changes + # are not detected by the ORM and will be silently lost on flush. Reads via direct attribute access are fine. + attributes: Mapped[Optional[dict]] = mapped_column( + JSON().with_variant(JSONB(), "postgresql"), + nullable=True, + default=dict, + ) + def __repr__(self): return f"" @@ -88,6 +100,28 @@ def to_json(self): """ return json.dumps(self.to_dict()) + # ---- attributes helpers ---- + + def get_attribute(self, key: str, default=None): + """Return an attribute value, or *default* if not set.""" + if not self.attributes: + return default + return self.attributes.get(key, default) + + def set_attribute(self, key: str, value) -> None: + """Set an attribute. Passing None removes the key. + + Always use this instead of mutating the dict in-place — in-place changes + are not detected by the ORM and will be silently lost on flush. + """ + if self.attributes is None: + self.attributes = {} + if value is None: + self.attributes.pop(key, None) + else: + self.attributes[key] = value + flag_modified(self, "attributes") + def __init__(self, name, label, arch, ip, platform, interface, snapshot, resultserver_ip, resultserver_port, reserved): self.name = name self.label = label diff --git a/lib/cuckoo/core/database.py b/lib/cuckoo/core/database.py index 9b7b9b5e3a7..0582092c71d 100644 --- a/lib/cuckoo/core/database.py +++ b/lib/cuckoo/core/database.py @@ -49,7 +49,7 @@ -SCHEMA_VERSION = "3a1b_tenant_visibility" +SCHEMA_VERSION = "4a6c2b_machine_attributes" log = logging.getLogger(__name__) conf = Config("cuckoo") diff --git a/utils/db_migration/versions/4. add_machine_attributes.py b/utils/db_migration/versions/4. add_machine_attributes.py new file mode 100644 index 00000000000..c67c58f5292 --- /dev/null +++ b/utils/db_migration/versions/4. add_machine_attributes.py @@ -0,0 +1,47 @@ +# Copyright (C) 2010-2015 Cuckoo Foundation. +# This file is part of Cuckoo Sandbox - http://www.cuckoosandbox.org +# See the file 'docs/LICENSE' for copying permission. + +"""Add attributes JSON column to machines + +Revision ID: 4a6c2b_machine_attributes +Revises: 3a1b_tenant_visibility +Create Date: 2026-09-01 + +Adds a generic JSONB column to the machines table to hold per-machine +semi-structured data (e.g. provider metadata, admin notes, runtime state) +without requiring ALTER TABLE for each new field. + +On Postgres the column is stored as real JSONB; on SQLite/MySQL it falls +back to plain JSON. Matches the Machine model which declares it with +JSON().with_variant(JSONB(), "postgresql") so fresh installs on Postgres +get the optimal type automatically via Base.metadata.create_all(). +Existing installs require this migration. +""" + +import sqlalchemy as sa +from alembic import op +from sqlalchemy.dialects.postgresql import JSONB + +# revision identifiers, used by Alembic. +revision = "4a6c2b_machine_attributes" +down_revision = "3a1b_tenant_visibility" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + # Real JSONB on Postgres (matches the Machine model), plain JSON elsewhere. + # Alembic compiles with_variant against the connection dialect at runtime. + op.add_column( + "machines", + sa.Column( + "attributes", + sa.JSON().with_variant(JSONB(), "postgresql"), + nullable=True, + ), + ) + + +def downgrade() -> None: + op.drop_column("machines", "attributes") \ No newline at end of file From e302f909897bc1b3c05c0ef3150a15c423c5bad8 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Muras?= Date: Tue, 1 Sep 2026 17:17:44 +0200 Subject: [PATCH 2/5] use MutableDict to allow in-place mutation --- lib/cuckoo/core/data/machines.py | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/lib/cuckoo/core/data/machines.py b/lib/cuckoo/core/data/machines.py index 5568414aff8..4b58e3baae6 100644 --- a/lib/cuckoo/core/data/machines.py +++ b/lib/cuckoo/core/data/machines.py @@ -24,9 +24,9 @@ String, ) from sqlalchemy.dialects.postgresql import JSONB + from sqlalchemy.ext.mutable import MutableDict from sqlalchemy.orm import ( Mapped, - flag_modified, mapped_column, relationship, selectinload, @@ -66,11 +66,13 @@ class Machine(Base): resultserver_port: Mapped[str] = mapped_column(String(255), nullable=False) reserved: Mapped[bool] = mapped_column(Boolean(), nullable=False, default=False) - # Generic key/value bag for per-machine semi-structured data. - # Use get_attribute / set_attribute for reads and writes. Never mutate the dict in place — in-place changes - # are not detected by the ORM and will be silently lost on flush. Reads via direct attribute access are fine. + # Generic key/value bag for per-machine semi-structured data: last_error, + # last_boot_on/last_shutdown_on and config_version (populated uniformly by + # base machinery), plus backend/provider data such as the VNC console URL + # or provider metadata (region, availability zone, hypervisor). Changes to + # the dict are auto-tracked via MutableDict (no flag_modified needed). attributes: Mapped[Optional[dict]] = mapped_column( - JSON().with_variant(JSONB(), "postgresql"), + MutableDict.as_mutable(JSON().with_variant(JSONB(), "postgresql")), nullable=True, default=dict, ) @@ -109,10 +111,11 @@ def get_attribute(self, key: str, default=None): return self.attributes.get(key, default) def set_attribute(self, key: str, value) -> None: - """Set an attribute. Passing None removes the key. + """Set an attribute. Passing None removes the key. Lazy-inits the dict. - Always use this instead of mutating the dict in-place — in-place changes - are not detected by the ORM and will be silently lost on flush. + MutableDict tracks changes automatically so no flag_modified is needed. + get_attribute / set_attribute are convenience wrappers; direct in-place + mutation (machine.attributes["k"] = v) is also fine. """ if self.attributes is None: self.attributes = {} @@ -120,7 +123,6 @@ def set_attribute(self, key: str, value) -> None: self.attributes.pop(key, None) else: self.attributes[key] = value - flag_modified(self, "attributes") def __init__(self, name, label, arch, ip, platform, interface, snapshot, resultserver_ip, resultserver_port, reserved): self.name = name From 85241934992a1c659a66cc396d8caed005995c1b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Muras?= Date: Tue, 1 Sep 2026 17:27:46 +0200 Subject: [PATCH 3/5] adjust comments --- lib/cuckoo/core/data/machines.py | 13 ++++--------- .../versions/4. add_machine_attributes.py | 8 ++++---- 2 files changed, 8 insertions(+), 13 deletions(-) diff --git a/lib/cuckoo/core/data/machines.py b/lib/cuckoo/core/data/machines.py index 4b58e3baae6..346adc50e43 100644 --- a/lib/cuckoo/core/data/machines.py +++ b/lib/cuckoo/core/data/machines.py @@ -66,11 +66,7 @@ class Machine(Base): resultserver_port: Mapped[str] = mapped_column(String(255), nullable=False) reserved: Mapped[bool] = mapped_column(Boolean(), nullable=False, default=False) - # Generic key/value bag for per-machine semi-structured data: last_error, - # last_boot_on/last_shutdown_on and config_version (populated uniformly by - # base machinery), plus backend/provider data such as the VNC console URL - # or provider metadata (region, availability zone, hypervisor). Changes to - # the dict are auto-tracked via MutableDict (no flag_modified needed). + # Generic key/value bag for per-machine semi-structured data attributes: Mapped[Optional[dict]] = mapped_column( MutableDict.as_mutable(JSON().with_variant(JSONB(), "postgresql")), nullable=True, @@ -105,7 +101,7 @@ def to_json(self): # ---- attributes helpers ---- def get_attribute(self, key: str, default=None): - """Return an attribute value, or *default* if not set.""" + """Return an attribute value, or default if not set.""" if not self.attributes: return default return self.attributes.get(key, default) @@ -113,9 +109,8 @@ def get_attribute(self, key: str, default=None): def set_attribute(self, key: str, value) -> None: """Set an attribute. Passing None removes the key. Lazy-inits the dict. - MutableDict tracks changes automatically so no flag_modified is needed. - get_attribute / set_attribute are convenience wrappers; direct in-place - mutation (machine.attributes["k"] = v) is also fine. + set_attribute is convenience wrapper - MutableDict tracks changes automatically so + direct in-place mutation (machine.attributes["k"] = v) is also fine. """ if self.attributes is None: self.attributes = {} diff --git a/utils/db_migration/versions/4. add_machine_attributes.py b/utils/db_migration/versions/4. add_machine_attributes.py index c67c58f5292..ed724cdc2b4 100644 --- a/utils/db_migration/versions/4. add_machine_attributes.py +++ b/utils/db_migration/versions/4. add_machine_attributes.py @@ -8,15 +8,15 @@ Revises: 3a1b_tenant_visibility Create Date: 2026-09-01 -Adds a generic JSONB column to the machines table to hold per-machine -semi-structured data (e.g. provider metadata, admin notes, runtime state) -without requiring ALTER TABLE for each new field. +Adds a generic JSONB column to the machines table to store extra per-machine data - whether + generic (e.g. last shutdown date) or machinery-specific (e.g. VNC connection details) +semi-structured data (e.g. provider metadata, admin notes, runtime state) - without + requiring ALTER TABLE for each new field. On Postgres the column is stored as real JSONB; on SQLite/MySQL it falls back to plain JSON. Matches the Machine model which declares it with JSON().with_variant(JSONB(), "postgresql") so fresh installs on Postgres get the optimal type automatically via Base.metadata.create_all(). -Existing installs require this migration. """ import sqlalchemy as sa From 70fb7112c61d7787d656cefebf1a583470e09f42 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Muras?= Date: Wed, 2 Sep 2026 07:53:07 +0200 Subject: [PATCH 4/5] dedicated unit test --- tests/test_database.py | 31 +++++++++++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/tests/test_database.py b/tests/test_database.py index 0da02ba471c..0722df9be8d 100644 --- a/tests/test_database.py +++ b/tests/test_database.py @@ -409,6 +409,7 @@ def test_add_machine(self, db: _Database): "resultserver_port": "2043", "arch": "x64", "reserved": False, + "attributes": {}, } assert m2.to_dict() == { @@ -428,8 +429,38 @@ def test_add_machine(self, db: _Database): "tags": ["tag1tag2"], "arch": "x64", "reserved": True, + "attributes": {}, } + def test_machine_attributes(self, db: _Database): + # 1) default=dict — each new row gets its own dict, not shared. + # Flush to apply the default (default fires on flush, not on construction + # for mutable/JSON columns); verify both rows get distinct dict objects. + with db.session.begin(): + m1 = self.add_machine(db, name="attr1", label="attr1") + m2 = self.add_machine(db, name="attr2", label="attr2") + db.session.flush() + assert m1.attributes is not m2.attributes + + # 3) get_attribute / set_attribute helpers. + with db.session.begin(): + m1.set_attribute("last_error", "boom") + m1.set_attribute("vnc_url", "vnc://x") + assert m1.get_attribute("last_error") == "boom" + assert m1.get_attribute("vnc_url") == "vnc://x" + m1.set_attribute("temp", "x") + m1.set_attribute("temp", None) # value=None removes the key + assert m1.get_attribute("temp") is None + + # 2) MutableDict auto-tracks in-place mutation — no flag_modified needed. + with db.session.begin(): + m = db.view_machine("attr1") + m.attributes["direct"] = "via_inplace_mutation" # no set_attribute, no flag_modified + with db.session.begin(): + m = db.view_machine("attr1") + assert m.get_attribute("last_error") == "boom" + assert m.get_attribute("direct") == "via_inplace_mutation" + def test_find_machine_to_service_task_tags_reserved(self, db: _Database): with db.session.begin(): self.add_machine(db, name="name0", label="label0", tags="tag1,x64", reserved=False) From 354eb08cdf43060e51d8bb4288e17016baa7c522 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Muras?= Date: Wed, 2 Sep 2026 18:33:54 +0200 Subject: [PATCH 5/5] format: add missing newline to machine attributes migration (ruff W292) --- utils/db_migration/versions/4. add_machine_attributes.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/utils/db_migration/versions/4. add_machine_attributes.py b/utils/db_migration/versions/4. add_machine_attributes.py index ed724cdc2b4..2e3ae542e73 100644 --- a/utils/db_migration/versions/4. add_machine_attributes.py +++ b/utils/db_migration/versions/4. add_machine_attributes.py @@ -44,4 +44,4 @@ def upgrade() -> None: def downgrade() -> None: - op.drop_column("machines", "attributes") \ No newline at end of file + op.drop_column("machines", "attributes")