diff --git a/.claude/review-guidelines.md b/.claude/review-guidelines.md index 0c78b168..901ede60 100644 --- a/.claude/review-guidelines.md +++ b/.claude/review-guidelines.md @@ -1,6 +1,6 @@ # Review Guidelines -Apply the full checklist in `docs/pr-reviews/PR_REVIEW_CHECKLIST.md`. Priorities: +Apply the shared review process maintained outside the repo. Priorities for this runtime: - Dual-process boundaries hold: PLC core (C/C++) and webserver (Python) communicate only via the documented IPC commands (`core/src/plc_app/unix_socket.c`). - Real-time safety in the scan cycle: no blocking calls, allocation, or logging in the hot path. diff --git a/.github/pull_request_template.md b/.github/pull_request_template.md index 9bed8461..b804f506 100644 --- a/.github/pull_request_template.md +++ b/.github/pull_request_template.md @@ -15,4 +15,4 @@ - [ ] `bash scripts/run-pytest.sh` passes - [ ] `pre-commit run` clean - [ ] Docs updated if behavior changed (README, CLAUDE.md, docs/) -- [ ] Follows `docs/pr-reviews/PR_REVIEW_CHECKLIST.md` +- [ ] Follows the shared review process (summary in `.claude/review-guidelines.md`) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index a4d5cac4..911cc36a 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -24,4 +24,5 @@ If your change alters documented behavior (commands, endpoints, env vars, archit ## Review -PRs are reviewed against `docs/pr-reviews/PR_REVIEW_CHECKLIST.md` (summary in `.claude/review-guidelines.md`). +PRs are reviewed against the shared review process maintained outside the repo. The in-tree +summary lives at `.claude/review-guidelines.md`. diff --git a/bootloader/internal/api/authz_test.go b/bootloader/internal/api/authz_test.go index 1a31ebc6..1d68066a 100644 --- a/bootloader/internal/api/authz_test.go +++ b/bootloader/internal/api/authz_test.go @@ -12,12 +12,9 @@ import ( "github.com/Autonomy-Logic/openplc-runtime/bootloader/internal/runtimeauth" ) -// The routes that can change what this device runs are admin-only. -// -// The runtime treats `user` as a restricted role, but the bootloader checked -// only the signature: any runtime account could change the runtime version or -// self-update the bootloader -- and a self-update starts a container with the -// Docker socket bound, which is host root. +// Routes that change what the device runs are admin-only. Self-update +// starts a container with the Docker socket bound (host root), so a +// signature-only check is not enough. func TestARestrictedAccountCannotChangeWhatTheDeviceRuns(t *testing.T) { cases := []struct { name, method, path, body string diff --git a/bootloader/internal/api/server.go b/bootloader/internal/api/server.go index bf59d161..dc89c738 100644 --- a/bootloader/internal/api/server.go +++ b/bootloader/internal/api/server.go @@ -2,18 +2,10 @@ // Copyright (c) 2026 Autonomy® // Package api is the bootloader's control API on port 8445. -// -// Deliberately small. This is the interface to the component that recovers a -// device, so its surface is the shortest list that does the job: say what -// state you are in, show me the runtime's logs, restart it, change its -// version, wipe its data. It accepts no programs and does not control the PLC -// -- those belong to the runtime, and a bootloader that could do them would be -// a second, less-reviewed path to the same capability. -// -// Every route except login and capabilities requires a token from the -// runtime's own account set. Capabilities is unauthenticated for the same -// reason the runtime's is: a client has to be able to tell what it is talking -// to before it has credentials. +// Surface is intentionally small: state, runtime logs, restart, +// version change, wipe. No program upload, no PLC control — those +// belong to the runtime. Every route except login and capabilities +// requires a token from the runtime's own account set. package api import ( @@ -65,34 +57,23 @@ type Updater interface { Progress() updater.Progress } -// SelfUpdater replaces the bootloader with a newer version of itself. -// -// Start returns once the helper that performs the swap is running: this -// process is about to be stopped by it, so there is no completion to report -// and nothing to poll -- the client reconnects and reads the new version from -// capabilities. +// SelfUpdater replaces the bootloader with a newer version. Start returns +// once the helper is running; this process is about to be stopped by it, +// so the client reconnects and reads the new version from capabilities. type SelfUpdater interface { Start(ctx context.Context, version string) error } -// HostReporter answers for the machine the runtime runs on. -// -// The bootloader is the right place for this. It exists on every device that -// can be updated from an editor, including one running a runtime far older -// than these endpoints -- so a Runtime Status screen fed from here is -// populated regardless of which runtime version is installed, which is not -// true of anything served by the runtime itself. +// HostReporter answers for the machine the runtime runs on. Lives in the +// bootloader so the Runtime Status screen is populated independent of +// which runtime version (or none) is installed. type HostReporter interface { SystemInfo(ctx context.Context) (*dockerapi.Info, error) } -// Authenticator resolves credentials against the runtime's account set, and -// serves the signing secret behind the tokens it issues. -// -// Secrets() is read per request rather than snapshotted at start-up: on a -// fresh install the runtime writes .env and restapi.db AFTER the bootloader is -// already running, and a snapshot taken before that left every authenticated -// route answering 503 until the container was restarted. +// Authenticator resolves credentials against the runtime's accounts and +// serves the token signing secret. Secrets() is read per request so a +// first-install .env written after start-up does not leave routes at 503. type Authenticator interface { Authenticate(ctx context.Context, username, password, pepper string) (*runtimeauth.User, error) CountUsers(ctx context.Context) (int, error) @@ -219,12 +200,9 @@ func (s *Server) ListenAndServe(ctx context.Context) error { // --- middleware ---------------------------------------------------------- -// authenticated wraps a handler with bearer-token verification. -// -// It also enforces the no-users rule: with no accounts on the device the -// bootloader accepts nothing at all. First-user bootstrap is a sensitive flow -// that lives in the runtime alone, and a bootloader that could mint the first -// admin would be a second path to owning the device. +// authenticated wraps a handler with bearer-token verification. With no +// accounts on the device, every authenticated route is refused: +// first-user bootstrap belongs to the runtime, never here. func (s *Server) authenticated(next func(http.ResponseWriter, *http.Request)) http.HandlerFunc { return func(w http.ResponseWriter, r *http.Request) { count, err := s.cfg.Users.CountUsers(r.Context()) @@ -277,18 +255,9 @@ func subjectFrom(ctx context.Context) string { // device. Matched against the runtime's own value (webserver/restapi.py). const RoleAdmin = "admin" -// adminOnly restricts a route to administrators. -// -// Applied to the routes that can change what this device runs. Without it any -// runtime account -- including one the runtime itself treats as restricted -- -// could change the runtime version or self-update the bootloader, and a -// self-update starts a container with the Docker socket bound, which is host -// root. The runtime distinguishes these roles; the component that can replace -// the runtime must not be the one that ignores the distinction. -// -// The role is read from the database per request. The token carries none, and -// a role claim would mean a demotion did not take effect until the token -// expired. +// adminOnly restricts a route to administrators. Role is read from the DB +// per request (not from a token claim) so a demotion takes effect +// immediately rather than at token expiry. func (s *Server) adminOnly(next func(http.ResponseWriter, *http.Request)) http.HandlerFunc { return s.authenticated(func(w http.ResponseWriter, r *http.Request) { subject := subjectFrom(r.Context()) @@ -326,20 +295,8 @@ func bearerToken(r *http.Request) (string, bool) { // --- handlers ------------------------------------------------------------ -// handleDeviceInfo reports the machine the runtime runs on. -// -// Sourced from the Docker daemon, which runs on the host and answers for it. -// The obvious alternative -- have the runtime report on itself -- is what this -// replaces: that endpoint exists only in runtimes new enough to have it, so -// every device in the field today answered it with a catch-all body and the -// screen had nothing to show. The bootloader is present wherever an update is -// possible at all, which makes it the one source that is always there. -// -// Deliberately only facts that VARY between devices. "This runtime runs in a -// container" and "this device updates itself" were both here at one point and -// are neither: a client reaching this handler at all has already learned them -// from the bootloader answering, so reporting them again was a field that -// could only ever hold one value. +// handleDeviceInfo reports the machine the runtime runs on. Sourced from +// the Docker daemon. Only facts that VARY between devices are reported. func (s *Server) handleDeviceInfo(w http.ResponseWriter, r *http.Request) { payload := map[string]any{ "bootloaderVersion": s.cfg.Version, @@ -583,16 +540,9 @@ func (s *Server) handleUpdateProgress(w http.ResponseWriter, r *http.Request) { writeJSON(w, http.StatusOK, s.cfg.Updater.Progress()) } -// handleSelfUpdate replaces the bootloader itself. -// -// Separate from the runtime update on purpose: they change different things -// and fail differently. A bootloader that will not come back costs the ability -// to manage the device; a runtime that will not come back stops the plant. The -// runtime container is untouched here, so a PLC keeps running throughout. -// -// There is no progress to poll. This process is replaced as part of the -// operation, so the client's connection ends with it -- reconnecting and -// reading /capabilities is how you learn the outcome. +// handleSelfUpdate replaces the bootloader itself. The runtime container +// is untouched so a running PLC survives. No progress endpoint: this +// process ends; reconnect and read /capabilities for the outcome. func (s *Server) handleSelfUpdate(w http.ResponseWriter, r *http.Request) { if s.cfg.SelfUpdater == nil { writeError(w, http.StatusNotImplemented, "this bootloader cannot update itself") diff --git a/bootloader/internal/api/server_test.go b/bootloader/internal/api/server_test.go index f490f60a..bb984b04 100644 --- a/bootloader/internal/api/server_test.go +++ b/bootloader/internal/api/server_test.go @@ -218,10 +218,6 @@ func TestProtectedRoutesRejectATokenSignedWithAnotherSecret(t *testing.T) { } func TestATokenTheBootloaderIssuedIsAccepted(t *testing.T) { - // The bootloader owns its own sessions: the editor logs in here with the - // credentials it already holds, and this token is only ever presented - // back to the bootloader. Cross-service acceptance is deliberately not a - // contract -- the two services may resolve different .env files. srv := newTestServer(t, &fakeUsers{count: 1}, healthySupervisor(), &fakeLogs{}) resp, _ := get(t, srv, "/api/bootloader/status", validToken(t)) if resp.StatusCode != http.StatusOK { diff --git a/bootloader/internal/api/throttle.go b/bootloader/internal/api/throttle.go index 7f7d8f07..cdb63b6b 100644 --- a/bootloader/internal/api/throttle.go +++ b/bootloader/internal/api/throttle.go @@ -10,21 +10,9 @@ import ( "time" ) -// Login throttling. -// -// Every attempt, including one for a username that does not exist, runs a full -// 600k-iteration PBKDF2 -- deliberately, so response timing does not enumerate -// accounts. That makes the endpoint expensive by design, and this component -// runs on the host network with no CPU limit, beside a PLC whose real-time -// headroom must not be eaten. A loop of login POSTs from any host on the LAN -// was therefore both a brute-force path to Docker-socket access and a cheap -// denial of service against the scan cycle. -// -// Two independent limits, because they address different things: a global -// concurrency cap bounds the CPU an attacker can command at any instant, and -// per-source backoff makes sustained guessing impractical. Neither replaces -// the other -- one attacker with two connections defeats a cap alone, and a -// distributed source set defeats backoff alone. +// Login throttling. Every attempt runs 600k PBKDF2 iterations (timing-safe +// against enumeration), so the endpoint needs both a global concurrency cap +// and per-source backoff to resist brute force and DoS. const ( // maxConcurrentVerifications is small on purpose. Two verifications in // flight is more than a legitimate operator ever needs, and it leaves the @@ -142,12 +130,8 @@ func (t *loginThrottle) recordSuccess(source string) { delete(t.sources, source) } -// requestSource identifies the caller for backoff purposes. -// -// The remote address only. There is no proxy in front of this: it is reached -// directly on the LAN, or through the orchestrator agent on the same host, so -// an X-Forwarded-For here would be attacker-controlled and trusting it would -// hand out a way to reset someone else's backoff. +// requestSource identifies the caller by remote address only. No proxy +// sits in front, so X-Forwarded-For would be attacker-controlled. func requestSource(r *http.Request) string { host, _, err := net.SplitHostPort(r.RemoteAddr) if err != nil { diff --git a/bootloader/internal/api/tls.go b/bootloader/internal/api/tls.go index 0c9b2b4a..8539527a 100644 --- a/bootloader/internal/api/tls.go +++ b/bootloader/internal/api/tls.go @@ -19,19 +19,9 @@ import ( "time" ) -// The bootloader serves HTTPS with its own self-signed certificate, generated -// once into its state directory and reused thereafter. -// -// Its own, rather than the runtime's: the runtime generates its certificate -// inside its image (webserver/certOPENPLC.pem), so it is not in the shared -// volume and there is nothing to share. Reusing it would also mean the -// bootloader could not serve TLS at all before the runtime had ever started, -// which is exactly the case recovery exists for. -// -// Self-signed is the same posture the runtime already has, so the editor's -// handling is unchanged. Persisting it matters: regenerating on every boot -// would change the fingerprint each time the device restarted, training -// operators to click through certificate warnings. +// Self-signed HTTPS cert, generated once into the state dir and persisted +// so the fingerprint stays stable. Owned by the bootloader because +// recovery must work before the runtime has ever started. const ( certFileName = "bootloader-cert.pem" keyFileName = "bootloader-key.pem" @@ -72,11 +62,8 @@ func LoadOrCreateCertificate(stateDir string) (tls.Certificate, error) { return cert, nil } -// generateSelfSigned writes a new P-256 certificate and key. -// -// ECDSA rather than RSA: a 2048-bit RSA keygen on a Pi-class CPU takes long -// enough to notice at first boot, and P-256 is both faster and universally -// supported by anything that will talk to this port. +// generateSelfSigned writes a new P-256 ECDSA certificate and key. +// ECDSA because 2048-bit RSA keygen is slow enough on a Pi to notice. func generateSelfSigned(certPath, keyPath string) error { key, err := ecdsa.GenerateKey(elliptic.P256(), rand.Reader) if err != nil { diff --git a/bootloader/internal/discovery/responder.go b/bootloader/internal/discovery/responder.go index 73773ef5..02c1d441 100644 --- a/bootloader/internal/discovery/responder.go +++ b/bootloader/internal/discovery/responder.go @@ -1,23 +1,11 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// Package discovery answers the editor's LAN discovery probe while the runtime -// is not running. -// -// The runtime has its own responder (webserver/discovery/network_discovery.py) -// and normally owns this port. The bootloader's exists for one situation: the -// runtime is down, so nothing is answering, and a device that cannot be found -// cannot be repaired. Without this, a failed update makes a device vanish from -// the editor's list at exactly the moment somebody needs to reach it. -// -// It runs ONLY in recovery mode, which is what keeps the two responders from -// ever competing. Recovery is defined as "the runtime container is stopped" -- -// the supervisor stops it before entering that state -- so exclusivity holds -// by construction rather than by coordination. Two services answering the same -// broadcast would give the editor two different answers for one device. -// -// The protocol is the runtime's, byte for byte: a fixed magic string in, one -// JSON datagram back, unicast to the sender. +// Package discovery answers the editor's LAN discovery probe while +// the runtime is down. Runs ONLY in recovery mode (the supervisor +// stops the runtime first), so it never races the runtime's own +// responder. Protocol is byte-for-byte the runtime's: fixed magic in, +// one JSON datagram back, unicast to the sender. package discovery import ( @@ -51,13 +39,8 @@ const ( perIPRateLimit = 100 * time.Millisecond ) -// Reply is what a probing editor receives. -// -// service says "openplc-bootloader", not "openplc-runtime". Being honest here -// costs an older editor the ability to see a device in recovery -- but an -// older editor could not have done anything about it either, and the -// alternative is a client that thinks it is talking to a working runtime and -// then fails against every endpoint it tries. +// Reply is what a probing editor receives. service says +// "openplc-bootloader" so a client cannot mistake it for a working runtime. type Reply struct { Service string `json:"service"` ProtocolVersion int `json:"protocol_version"` @@ -109,12 +92,9 @@ func New(port int, provider ReplyProvider, log *slog.Logger) *Responder { } } -// Enable starts answering probes. Safe to call when already enabled. -// -// A bind failure is logged and swallowed. Discovery is a convenience: losing -// it must not stop the bootloader serving its control API, which is the -// primary way in. The most likely cause is the runtime still holding the port, -// and in that case the device is findable anyway. +// Enable starts answering probes; idempotent. A bind failure is logged +// and swallowed: discovery is a convenience, losing it must not stop the +// control API from serving. func (r *Responder) Enable() { r.mu.Lock() if r.conn != nil { @@ -122,11 +102,9 @@ func (r *Responder) Enable() { return } var conn *net.UDPConn - // SO_REUSEADDR and SO_REUSEPORT, matching how the runtime binds the same - // port. Linux shares a UDP port only when EVERY socket asked to, so - // without these a lingering bootloader socket makes the runtime's own bind - // fail -- and the runtime does not retry. The release now happens before - // the runtime starts; this is the safety net for a race in between. + // SO_REUSEADDR/REUSEPORT must match the runtime; Linux shares a UDP + // port only when every socket asked to, otherwise a lingering socket + // here blocks the runtime's bind. listener := net.ListenConfig{Control: reusePort} generic, err := listener.ListenPacket( context.Background(), "udp", ":"+strconv.Itoa(r.port)) diff --git a/bootloader/internal/discovery/reuse_linux.go b/bootloader/internal/discovery/reuse_linux.go index 90297f4c..e1ebbf13 100644 --- a/bootloader/internal/discovery/reuse_linux.go +++ b/bootloader/internal/discovery/reuse_linux.go @@ -11,12 +11,8 @@ import ( "golang.org/x/sys/unix" ) -// reusePort sets SO_REUSEADDR and SO_REUSEPORT on the listening socket. -// -// The runtime sets both when it binds the discovery port. Linux shares a UDP -// port only when every socket involved asked to, so the bootloader has to ask -// too -- otherwise its lingering socket makes the runtime's bind fail, and the -// runtime binds once at start-up and never retries. +// reusePort sets SO_REUSEADDR and SO_REUSEPORT so the port can be shared +// with the runtime, which also sets both. func reusePort(_, _ string, c syscall.RawConn) error { var setErr error err := c.Control(func(fd uintptr) { diff --git a/bootloader/internal/dockerapi/client.go b/bootloader/internal/dockerapi/client.go index 98503bcc..9f9f5dc3 100644 --- a/bootloader/internal/dockerapi/client.go +++ b/bootloader/internal/dockerapi/client.go @@ -1,14 +1,10 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// Package dockerapi is a minimal client for the Docker Engine API over the -// host's unix socket. -// -// Hand-rolled rather than using the official SDK on purpose: the bootloader needs -// eight calls, and the SDK brings a dependency tree into the one component -// whose job is to still work when everything else is broken. The Engine API is -// JSON over HTTP; the only unusual part is dialing a unix socket instead of a -// TCP address, which the transport below handles. +// Package dockerapi is a minimal client for the Docker Engine API +// over the host's unix socket. Hand-rolled to avoid pulling the +// official SDK's dependency tree into the one component that must +// still work when everything else is broken. package dockerapi import ( @@ -30,10 +26,8 @@ import ( // which is why the bootloader stays small enough to audit. const DefaultSocket = "/var/run/docker.sock" -// apiVersion is pinned low enough to work on the oldest engine we support. -// The SLM-RP4 test device ships Docker 20.10 (API 1.41), and every call this -// package makes has been stable since well before that. Pinning avoids a -// daemon upgrade silently changing a response shape under us. +// apiVersion is pinned low enough for Docker 20.10 (SLM-RP4). Pinning +// avoids a daemon upgrade silently changing a response shape. const apiVersion = "v1.41" // Client talks to the Docker daemon. Safe for concurrent use: the embedded @@ -45,14 +39,9 @@ type Client struct { socket string } -// New returns a client bound to socket. A zero-value socket means -// DefaultSocket. -// -// The timeout applies to unary calls only. Streaming calls (events, image -// pull) must not be bounded by it -- an events stream is meant to stay open -// for the life of the process -- so they run on a separate, timeout-free -// client. Using one client for both is the classic way to end up with an -// events stream that dies silently after 30 seconds. +// New returns a client bound to socket (empty means DefaultSocket). The +// 30s timeout applies to unary calls only; streaming (events, pull) runs +// on a separate timeout-free client. func New(socket string) *Client { if socket == "" { socket = DefaultSocket @@ -75,18 +64,9 @@ func New(socket string) *Client { } } -// newStreamClient builds the timeout-free client used for long-lived response -// bodies. Called ONCE, from New. -// -// It used to be built per call, from stream() and doLongRunning(). Each -// throwaway Transport kept its own idle connection pool with no -// IdleConnTimeout, and a drained-and-closed body returns its connection to -// that pool -- where the read and write goroutines pin it forever. Every -// StopContainer, PullImage and ContainerLogs therefore leaked a unix socket -// and two goroutines, fastest while an operator reads logs in recovery, which -// is the state this component exists for. It ends in EMFILE. -// -// An http.Client is safe for concurrent use, so one is all that is needed. +// newStreamClient builds the timeout-free client used for long-lived +// response bodies. One instance (not per-call) so idle sockets are +// shared and bounded, instead of pinned by throwaway transports. func newStreamClient(socket string) *http.Client { dial := func(ctx context.Context, _, _ string) (net.Conn, error) { var d net.Dialer @@ -223,10 +203,9 @@ func checkResponse(resp *http.Response, path string) error { return &APIError{Status: resp.StatusCode, Message: message, Path: path} } -// doLongRunning issues a request whose duration the caller bounds with the -// context, rather than the shared client's fixed timeout. For calls the daemon -// legitimately holds open -- stopping a container waits out its grace period -- -// a fixed client timeout is a race the caller cannot widen. +// doLongRunning issues a request bounded by the caller's context, not by +// the shared client's fixed timeout. Needed for calls the daemon holds +// open (stop with a grace period, image pull). func (c *Client) doLongRunning(ctx context.Context, method, path string, body any) error { req, err := c.newRequest(ctx, method, path, body) if err != nil { @@ -276,13 +255,9 @@ func encodeQuery(params url.Values) string { return "?" + params.Encode() } -// Reason extracts the most useful human-readable part of a daemon error. -// -// Errors here accumulate layers on the way up -- "could not download X: -// pulling X: docker /images/create?fromImage=X&tag=Y: HTTP 500: pull access -// denied" -- and every layer but the last is machinery. The daemon's own -// message is the only part that tells an operator what to do about it, so -// that is what gets shown; the full chain still goes to the log. +// Reason extracts the daemon's own message from a wrapped error chain: +// everything above the APIError is transport machinery, and operators +// need the bottom layer. The full chain still goes to the log. func Reason(err error) string { if err == nil { return "" diff --git a/bootloader/internal/dockerapi/containers.go b/bootloader/internal/dockerapi/containers.go index e4c3f3a1..2abe7e9c 100644 --- a/bootloader/internal/dockerapi/containers.go +++ b/bootloader/internal/dockerapi/containers.go @@ -28,17 +28,13 @@ type ContainerState struct { } `json:"Health"` } -// ContainerHostConfig is the part of a container's host configuration the -// bootloader needs to reproduce when it replaces itself. -// -// Captured from the RUNNING container rather than reconstructed from defaults: -// an operator may have installed with extra mounts or a different port, and -// a self-update that silently dropped them would leave a device subtly -// misconfigured in a way nobody would connect to "the bootloader updated". +// ContainerHostConfig is the host-config slice captured from a RUNNING +// container so a self-update can reproduce it, including operator-added +// mounts and port overrides that defaults would silently drop. type ContainerHostConfig struct { Binds []string `json:"Binds"` NetworkMode string `json:"NetworkMode"` - // "host", or empty for Docker's private default (RTOP-292). + // "host", or empty for Docker's private default. UTSMode string `json:"UTSMode"` Privileged bool `json:"Privileged"` RestartPolicy RestartPolicy `json:"RestartPolicy"` @@ -111,34 +107,24 @@ func (c *Client) StartContainer(ctx context.Context, name string) error { return c.do(ctx, http.MethodPost, "/containers/"+url.PathEscape(name)+"/start", nil, nil) } -// StopContainer sends SIGTERM and, after the grace period, SIGKILL. -// -// The grace period matters: the runtime shuts the PLC down and flushes retained -// variables on SIGTERM, so cutting it short risks losing the retain image. The -// daemon returns 304 when the container is already stopped, which is inside the -// 2xx-or-not check and so surfaces as success. +// StopContainer sends SIGTERM, then SIGKILL after the grace period. The +// runtime flushes retained variables on SIGTERM; cutting the grace short +// risks losing the retain image. func (c *Client) StopContainer(ctx context.Context, name string, grace time.Duration) error { params := url.Values{} params.Set("t", strconv.Itoa(int(grace.Seconds()))) path := "/containers/" + url.PathEscape(name) + "/stop" + encodeQuery(params) - // The daemon holds this request open for the whole grace period before it - // resorts to SIGKILL, so the client must be allowed to wait longer than - // the grace itself. The shared unary client's fixed 30s timeout is exactly - // equal to the default grace, so every swap raced it: observed on the - // SLM-RP4 as "Client.Timeout exceeded while awaiting headers" on a stop - // that was proceeding perfectly well, leaving the runtime to be killed by - // the force-remove path instead of shut down cleanly -- which for a PLC - // means skipping the SIGTERM handler that flushes retained variables. + // Daemon holds this open for the whole grace; the client needs more + // than the shared 30s unary timeout, which equals the default grace + // and would truncate every clean shutdown. stopCtx, cancel := context.WithTimeout(ctx, grace+stopTimeoutMargin) defer cancel() err := c.doLongRunning(stopCtx, http.MethodPost, path, nil) if err != nil && (IsNotFound(err) || hasStatus(err, http.StatusNotModified)) { - // Nothing was running, so nothing will exit. Reported rather than - // swallowed: a caller that suppresses crash accounting for the exit it - // is about to cause must know when that exit is never coming, or the - // suppression outlives the stop and eats the next real crash. + // Reported, not swallowed: a caller suppressing crash accounting + // around this stop must know when no exit will follow. return ErrNotRunning } return err @@ -166,9 +152,9 @@ func (c *Client) RemoveContainer(ctx context.Context, name string, force bool) e return nil } -// ContainerLogs returns the tail of a container's combined output. Used by the -// bootloader's status endpoint so an operator can see why a runtime would not -// start without needing shell access -- which is the entire point of RTOP-283. +// ContainerLogs returns the tail of a container's combined output. +// Used by the bootloader status endpoint so an operator can see why a +// runtime would not start without needing shell access on the device. func (c *Client) ContainerLogs(ctx context.Context, name string, tail int) (string, error) { params := url.Values{} params.Set("stdout", "true") @@ -183,11 +169,9 @@ func (c *Client) ContainerLogs(ctx context.Context, name string, tail int) (stri return readMultiplexed(body, 512*1024) } -// RenameContainer gives an existing container a new name. -// -// Used by the self-update so a replacement can be created under a temporary -// name and only then take over the real one -- which means a failed create -// leaves the old container untouched instead of removing it first and hoping. +// RenameContainer gives an existing container a new name. Used by the +// self-update to swap atomically: a failed create leaves the old +// container untouched instead of removed-then-missing. func (c *Client) RenameContainer(ctx context.Context, name, newName string) error { params := url.Values{} params.Set("name", newName) diff --git a/bootloader/internal/dockerapi/events.go b/bootloader/internal/dockerapi/events.go index 3de0c362..5d20115c 100644 --- a/bootloader/internal/dockerapi/events.go +++ b/bootloader/internal/dockerapi/events.go @@ -57,17 +57,9 @@ func (e *Event) HealthStatus() string { return "" } -// StreamEvents delivers container events for the named container to handle -// until ctx is cancelled or the stream breaks. -// -// It returns the error that ended the stream, always non-nil -- a broken -// events stream is never a normal end of work, and the caller is expected to -// reconnect and re-reconcile. That reconnect matters: the daemon restarting -// closes this stream, and any state change during the gap is missed, so the -// caller must re-inspect rather than assume it saw everything. -// -// Filtering happens daemon-side so an unrelated busy host does not push -// thousands of irrelevant events through this process. +// StreamEvents delivers container events until ctx is cancelled or the +// stream breaks (always returns non-nil). Caller must reconnect AND +// re-inspect, since state changes during the gap are missed. func (c *Client) StreamEvents(ctx context.Context, containerName string, handle func(Event)) error { filters := map[string][]string{ "type": {"container"}, @@ -103,17 +95,9 @@ func (c *Client) StreamEvents(ctx context.Context, containerName string, handle } } -// readMultiplexed decodes Docker's non-TTY stream framing into plain text. -// -// Without a TTY the daemon interleaves stdout and stderr as frames with an -// 8-byte header: [stream byte, 3 zero bytes, 4-byte big-endian length]. Reading -// the body raw would splice those headers into the middle of log lines, which -// is exactly the kind of small wrongness that makes an operator distrust the -// recovery screen. Both streams are kept, in arrival order, because a runtime -// that failed to start says why on stderr. -// -// Output is capped at limit bytes; the tail is kept, since the end of the log -// is where the failure is. +// readMultiplexed decodes Docker's non-TTY framing: 8-byte header +// [stream byte, 3 zero, 4-byte BE length]. Both streams kept in arrival +// order. Output capped at limit bytes, keeping the tail. func readMultiplexed(r io.Reader, limit int) (string, error) { reader := bufio.NewReader(r) var out strings.Builder diff --git a/bootloader/internal/dockerapi/images.go b/bootloader/internal/dockerapi/images.go index aa0f445b..8e331e5a 100644 --- a/bootloader/internal/dockerapi/images.go +++ b/bootloader/internal/dockerapi/images.go @@ -45,17 +45,9 @@ type pullEvent struct { } `json:"errorDetail"` } -// StallTimeout is how long a pull may go without any progress before it is -// abandoned. -// -// A stall timeout rather than a total timeout, deliberately. The Docker -// client's streaming pull takes no timeout at all, so a half-open connection -// to the registry leaves the decoder parked forever -- the failure -// orchestrator-agent documents in pull_runtime_image.py, where the entry stuck -// in "pulling" refused every retry for the life of the process. A total -// timeout would instead punish a slow-but-working link, which on a plant -// network is the normal case: a 1.4 GB image over a poor connection can -// legitimately take a very long time while never stalling. +// StallTimeout aborts a pull after this long with no progress. Stall +// (not total) timeout so a slow-but-working link is not punished: a +// 1.4 GB image over a plant network can legitimately take hours. const StallTimeout = 5 * time.Minute // PullImage pulls ref, reporting progress until it completes. @@ -219,12 +211,8 @@ func (c *Client) InspectImage(ctx context.Context, ref string) (*ImageInfo, erro return &out, nil } -// RemoveImage deletes an image by reference. -// -// A missing image is success: the goal is "not present". A conflict is NOT -// swallowed -- it means a container still references the image, and silently -// ignoring that would leave an operator believing disk was reclaimed when it -// was not. +// RemoveImage deletes an image by reference. Missing is success. A +// conflict (container still references it) is NOT swallowed. func (c *Client) RemoveImage(ctx context.Context, ref string, force bool) error { params := url.Values{} if force { @@ -237,11 +225,9 @@ func (c *Client) RemoveImage(ctx context.Context, ref string, force bool) error return nil } -// splitImageRef separates a reference into name and tag. -// -// Only the last colon counts as the tag separator, and only when it appears -// after the final slash: a registry with a port ("host:5000/image") puts a -// colon in the name, and splitting on the first would produce nonsense. +// splitImageRef separates a reference into name and tag. Only the last +// colon after the final slash is the tag separator, so a registry port +// like "host:5000/image" is not misparsed. func splitImageRef(ref string) (name, tag string) { lastSlash := strings.LastIndex(ref, "/") lastColon := strings.LastIndex(ref, ":") diff --git a/bootloader/internal/dockerapi/info.go b/bootloader/internal/dockerapi/info.go index 33ab3eac..2ae33be7 100644 --- a/bootloader/internal/dockerapi/info.go +++ b/bootloader/internal/dockerapi/info.go @@ -8,15 +8,9 @@ import ( "net/http" ) -// Info is the subset of the daemon's /info the bootloader reports. -// -// These are HOST facts, not container ones, and that is the whole reason this -// exists. The bootloader runs in a container: its own uname reports the shared -// kernel correctly but its hostname is a container id, and reading /etc/os-release -// from the image would describe the image rather than the device. The daemon -// runs on the host and answers for it, over a socket the bootloader already -// holds -- so no extra mounts, no extra privileges, and nothing that has to be -// kept in step with how the container happens to be launched. +// Info is the subset of /info the bootloader reports. These are HOST +// facts (hostname, OS), not the container's: the daemon answers for the +// host over the socket the bootloader already holds. type Info struct { // Name is the host's hostname. Name string `json:"Name"` diff --git a/bootloader/internal/health/prober.go b/bootloader/internal/health/prober.go index 7a3bf28e..64cd5057 100644 --- a/bootloader/internal/health/prober.go +++ b/bootloader/internal/health/prober.go @@ -1,14 +1,10 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// Package health probes the runtime webserver. -// -// Scope is deliberately narrow: "is the webserver answering". Whether plc_main -// is running, whether a program is loaded, and whether that program is in -// ERROR are all the webserver's own business -- runtimemanager._monitor() -// already restarts plc_main and drops it into safe mode on rapid crashes. A -// probe that cared about PLC state would let a bad user program trigger a -// runtime rollback, turning a logic bug into a device outage. +// Package health probes the runtime webserver. Scope is narrow: "is +// the webserver answering". PLC state is the webserver's business +// (runtimemanager._monitor() handles plc_main). Caring about PLC +// state here would let a bad user program trigger a runtime rollback. package health import ( @@ -20,12 +16,8 @@ import ( "time" ) -// Prober checks the runtime's unauthenticated version endpoint. -// -// /api/version, not /api/ping: ping sits behind @jwt_required(), so the -// bootloader has no credentials for it and a probe there would report a healthy -// runtime as dead. (The healthcheck example in docs/DOCKER.md has this wrong -// and always gets a 401.) +// Prober checks the runtime's unauthenticated /api/version endpoint. +// /api/ping is JWT-required and would always 401 from here. type Prober struct { url string client *http.Client @@ -34,12 +26,9 @@ type Prober struct { // DefaultURL is where the runtime listens on a host-network container. const DefaultURL = "https://127.0.0.1:8443/api/version" -// New returns a prober for url, or DefaultURL when empty. -// -// TLS verification is off by design. The runtime generates a self-signed -// certificate at start-up, and this connection is to 127.0.0.1 inside the same -// host -- there is no name to verify and no network path to intercept. Turning -// it on would simply make the probe always fail. +// New returns a prober for url, or DefaultURL when empty. TLS verify is +// off: the connection is to 127.0.0.1 against the runtime's self-signed +// cert, so there is no name to verify and no network path to intercept. func New(url string, timeout time.Duration) *Prober { if url == "" { url = DefaultURL diff --git a/bootloader/internal/runtimeauth/password.go b/bootloader/internal/runtimeauth/password.go index bc028e17..756c556c 100644 --- a/bootloader/internal/runtimeauth/password.go +++ b/bootloader/internal/runtimeauth/password.go @@ -15,14 +15,9 @@ import ( "strings" ) -// Werkzeug's generate_password_hash writes -// -// pbkdf2:sha256:$$ -// -// with the salt used as raw bytes of the ASCII string, not decoded. The runtime -// pins iterations at 600000 (User.derivation_method) but the count is read from -// the stored hash rather than assumed, so a future change on the Python side -// keeps verifying instead of silently rejecting every password. +// Werkzeug format: pbkdf2:sha256:$$. Salt is +// used as raw ASCII bytes, not decoded. Iterations are read from the +// hash so a future Python-side change keeps verifying. const ( pbkdf2Prefix = "pbkdf2:" // maxIterations bounds work from a malformed or hostile hash: 600k is the @@ -36,12 +31,8 @@ const ( // runtime hashes differently" instead of "wrong password". var ErrUnsupportedHash = errors.New("unsupported password hash format") -// VerifyPassword checks password against a Werkzeug PBKDF2 hash. -// -// The pepper is appended before hashing, exactly as User.set_password does -// (“password = password + PEPPER“). Getting the order wrong would fail every -// login while looking entirely reasonable, which is why the shared test vector -// exists. +// VerifyPassword checks password against a Werkzeug PBKDF2 hash. Pepper +// is appended before hashing, matching User.set_password. func VerifyPassword(storedHash, password, pepper string) (bool, error) { if !strings.HasPrefix(storedHash, pbkdf2Prefix) { return false, fmt.Errorf("%w: %q", ErrUnsupportedHash, firstField(storedHash)) diff --git a/bootloader/internal/runtimeauth/provider.go b/bootloader/internal/runtimeauth/provider.go index 84debc8e..a0f2c27d 100644 --- a/bootloader/internal/runtimeauth/provider.go +++ b/bootloader/internal/runtimeauth/provider.go @@ -11,19 +11,9 @@ import ( "sync" ) -// Provider serves the runtime's credentials, reloading them when they change. -// -// Loading once at start-up was wrong in the case that matters most: on a fresh -// install the bootloader starts BEFORE the runtime has ever run, so neither -// `.env` nor `restapi.db` exists yet. Every authenticated route then answered -// 503 until someone restarted the bootloader container -- while -// /capabilities answered happily, so the editor offered "Change runtime -// version" and the login behind it failed. The same staleness applies whenever -// the runtime regenerates its secrets. -// -// So the files are re-examined on use. A stat of each per request is cheap -// next to the PBKDF2 verification it precedes, and it means the bootloader -// becomes usable the moment the runtime has written them, with no restart. +// Provider serves the runtime's credentials, re-examined on use so the +// bootloader becomes usable the moment the runtime first writes .env and +// restapi.db (which may happen after the bootloader has started). type Provider struct { dataDir string log *slog.Logger diff --git a/bootloader/internal/runtimeauth/runtimeauth_test.go b/bootloader/internal/runtimeauth/runtimeauth_test.go index 32703a10..794ed9c3 100644 --- a/bootloader/internal/runtimeauth/runtimeauth_test.go +++ b/bootloader/internal/runtimeauth/runtimeauth_test.go @@ -18,17 +18,10 @@ import ( "time" ) -// Shared test vector, generated by the runtime's OWN libraries (werkzeug -// generate_password_hash and flask_jwt_extended create_access_token) and -// pinned here verbatim. -// -// The Go side of the bootloader reimplements two formats the Python side owns. -// That is the same hazard as the ctypes mirror in shared/plugin_runtime_args.py: -// the two can drift apart silently, and the symptom is every login failing on a -// device nobody can log into to diagnose. The identical values are asserted -// from Python in tests/pytest/restapi/test_bootloader_auth_vector.py, so a -// werkzeug or PyJWT upgrade that changes either format breaks a test on the -// side that changed rather than a device in the field. +// Shared test vector pinned from werkzeug / flask_jwt_extended. The same +// values are re-asserted from Python in +// tests/pytest/restapi/test_bootloader_auth_vector.py so an upstream +// upgrade that changes a format breaks a test, not a field device. const ( vectorPepper = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" vectorPassword = "correct horse battery staple" @@ -105,10 +98,8 @@ func TestAnAbsurdIterationCountIsRefused(t *testing.T) { // --- tokens -------------------------------------------------------------- func TestOurHMACAgreesWithPyJWTOnTheWireFormat(t *testing.T) { - // The cross-language check, kept because it is what proves the base64url - // variant and the "header.payload" framing agree with flask_jwt_extended. - // It signs with the RUNTIME's secret directly, which is what PyJWT did -- - // this is a statement about the encoding, not about which tokens we accept. + // Encoding check: signs with the RUNTIME's secret directly to prove + // base64url and header.payload framing match flask_jwt_extended. parts := strings.Split(vectorToken, ".") if len(parts) != 3 { t.Fatalf("vector token is malformed: %d segments", len(parts)) @@ -122,12 +113,9 @@ func TestOurHMACAgreesWithPyJWTOnTheWireFormat(t *testing.T) { } func TestARuntimeTokenIsNotAcceptedByTheBootloader(t *testing.T) { - // The two services share a credential database, NOT a session. They - // previously shared JWT_SECRET_KEY directly, which made a 2-hour - // bootloader token a valid runtime token -- eight times the runtime's own - // TTL, and revoked by neither side's logout. The bootloader now signs with - // a key derived from that secret, so a real runtime token fails here on - // the signature. + // Bootloader signs with a key derived from the runtime secret, so a + // real runtime token fails here on the signature. Prevents a long-TTL + // bootloader token from doubling as a runtime token. if _, err := VerifyToken(vectorSecret, vectorToken); err == nil { t.Fatal("a runtime-issued token was accepted by the bootloader") } @@ -503,11 +491,8 @@ func TestAMissingRoleColumnValueDefaultsToAdmin(t *testing.T) { // --- absent database ----------------------------------------------------- func TestANilStoreReportsRatherThanPanics(t *testing.T) { - // On a device whose runtime has never started there is no restapi.db, and - // the bootloader still has to come up so an operator can find out why. A - // typed nil assigned to an interface is not nil at the call site, so - // without these guards the first request would panic the process into a - // Docker restart loop -- on precisely the device that most needs a way in. + // First-run devices have no restapi.db; the bootloader must still boot. + // A typed-nil UserStore in an interface is not nil at the call site. var store *UserStore ctx := context.Background() diff --git a/bootloader/internal/runtimeauth/secrets.go b/bootloader/internal/runtimeauth/secrets.go index 572b0790..6a140f80 100644 --- a/bootloader/internal/runtimeauth/secrets.go +++ b/bootloader/internal/runtimeauth/secrets.go @@ -2,18 +2,9 @@ // Copyright (c) 2026 Autonomy® // Package runtimeauth authenticates callers against the runtime's own -// credentials. -// -// The bootloader deliberately does not keep a second user database. It reads the -// runtime's “.env“ and “restapi.db“ from the shared data directory -- -// mounted read-only, because it only ever needs to read them -- so there is -// exactly one set of accounts on the device and no second thing to keep in -// sync or forget to revoke. -// -// The formats here mirror the runtime's and must stay byte-compatible with it, -// the same hazard as the ctypes mirror in shared/plugin_runtime_args.py. Both -// sides are pinned by a shared test vector: tests/pytest/restapi generates a -// hash and a token, and the Go tests verify the identical values. +// credentials. Reads `.env` and `restapi.db` from the shared data +// directory (mounted read-only). The hash and token formats mirror +// the runtime's — pinned byte-for-byte by a shared pytest/Go vector. package runtimeauth import ( @@ -23,14 +14,9 @@ import ( "strings" ) -// Secrets are the two values the runtime generates once, in -// webserver/config.py::generate_env_file, and never rotates: changing either -// invalidates every stored password hash, which is why that function deletes -// the database when it writes a new .env. -// -// The pepper is what the bootloader genuinely needs, since it is required to -// verify a password against a stored hash. The JWT secret is used only to sign -// the bootloader's own tokens -- the two services do not share sessions. +// Secrets generated once by generate_env_file and never rotated. +// Changing either invalidates every stored password hash. Pepper verifies +// passwords; JWTSecret signs the bootloader's own tokens. type Secrets struct { // JWTSecret signs and verifies access tokens (HS256). JWTSecret string @@ -38,12 +24,8 @@ type Secrets struct { Pepper string } -// LoadSecrets reads the runtime's .env. -// -// A hand-rolled parser rather than a dotenv library: the file is written by -// generate_env_file with four fixed KEY=VALUE lines and no quoting, expansion -// or multi-line values, so a dependency would buy nothing in the component -// that most wants none. +// LoadSecrets reads the runtime's .env. Hand-rolled parser: the file has +// four fixed KEY=VALUE lines with no quoting or expansion. func LoadSecrets(path string) (*Secrets, error) { file, err := os.Open(path) if err != nil { diff --git a/bootloader/internal/runtimeauth/token.go b/bootloader/internal/runtimeauth/token.go index 39d21068..bd5af800 100644 --- a/bootloader/internal/runtimeauth/token.go +++ b/bootloader/internal/runtimeauth/token.go @@ -17,44 +17,16 @@ import ( "time" ) -// Tokens are HS256 JWTs in the same shape as the runtime's, but they are NOT -// the runtime's tokens and neither service will accept the other's. -// -// What the two share is the credential database, not a session: the editor -// keeps the user's credentials after login and signs in to the bootloader -// separately when it needs to. -// -// Separation is enforced by the signing key, not by a claim. Both services -// read the same JWT_SECRET_KEY from the same .env, so signing with it -// directly made the two token spaces identical: a 2-hour bootloader token was -// a valid runtime token, eight times the runtime's own 15-minute TTL, and the -// runtime's /logout revoked neither. The bootloader therefore signs with a key -// DERIVED from that secret (see bootloaderKey), which the runtime does not -// know how to compute. A runtime token fails the signature check here, and a -// bootloader token fails it there -- with no change required in the runtime, -// and no reliance on a verifier bothering to check an audience claim. -// -// The `aud` claim below is belt and braces: it makes the intent legible in a -// decoded token and would catch a future signing change that reunified the -// keys by accident. -// -// The rest of the claim set still mirrors flask_jwt_extended's -- "sub", -// "type", "iat", "nbf", "exp", "jti" -- so the two are recognisable to the -// same tooling. A hand-rolled implementation rather than a JWT library -// because HS256 is an HMAC over two base64url segments, and the -// library-shaped risk here (accepting "alg": "none", or letting the token -// choose its own algorithm) is precisely what an explicit implementation -// avoids: the algorithm below is a constant, never read from the header. +// Bootloader HS256 JWTs. Signed with a key DERIVED from JWT_SECRET_KEY +// (see bootloaderKey) so the runtime cannot verify these. Hand-rolled: +// HS256 is a constant, never read from the header. const ( // TokenType is flask_jwt_extended's discriminator. A refresh token // presented as an access token must not be accepted. TokenType = "access" - // DefaultTokenTTL is deliberately longer than the runtime's 15-minute - // default: a version change involves an image pull that can run for many - // minutes on a slow device, and having the caller's token expire midway - // through would strand a device mid-update. The bootloader owns its own - // sessions, so this does not have to match the runtime's. + // DefaultTokenTTL is longer than the runtime's 15 min default so a + // slow image pull during an update does not strand the token. DefaultTokenTTL = 2 * time.Hour // clockSkew tolerates a small disagreement between the editor's clock and // the device's, which on an industrial box without NTP is routine. @@ -66,13 +38,9 @@ const ( keyDomain = "openplc-bootloader/token/v1" ) -// bootloaderKey derives the bootloader's signing key from the runtime's -// secret. -// -// HMAC with a fixed domain string: a one-way function of the shared secret -// that the runtime never computes, so neither service can verify the other's -// tokens. Anyone who can read .env can derive it, which is the point -- this -// separates two token spaces on one device, it is not a secret from the host. +// bootloaderKey derives the bootloader signing key from the runtime +// secret via HMAC with a fixed domain string. One-way, so the runtime +// cannot compute it and cannot verify bootloader tokens. func bootloaderKey(secret string) []byte { mac := hmac.New(sha256.New, []byte(secret)) mac.Write([]byte(keyDomain)) @@ -106,12 +74,8 @@ type jwtHeader struct { Typ string `json:"typ"` } -// IssueToken mints an access token for the given user id. -// -// The subject is the user's numeric id rendered as a string, matching the -// runtime's user_identity_lookup (“return str(user.id)“). A username here -// would produce a token the runtime accepts structurally but then fails to -// resolve to a user, which is a confusing way to be broken. +// IssueToken mints an access token for the given user id. Subject is +// the user's numeric id as a string, matching user_identity_lookup. func IssueToken(secret, userID string, ttl time.Duration) (string, error) { if secret == "" { return "", errors.New("cannot issue a token without a signing secret") @@ -159,11 +123,8 @@ func VerifyToken(secret, token string) (*Claims, error) { } signingInput := parts[0] + "." + parts[1] - // The algorithm is NOT taken from the header. Trusting the header is the - // classic JWT vulnerability: a token claiming "alg": "none" or "HS256" - // against an RSA key gets verified against attacker-chosen rules. Here - // HS256 is the only thing that is ever computed, so a header saying - // otherwise simply fails the comparison below. + // Algorithm is a constant (HS256), never read from the header, so a + // token claiming "alg": "none" fails on the signature comparison. expected := sign(secret, signingInput) if subtle.ConstantTimeCompare([]byte(expected), []byte(parts[2])) != 1 { return nil, ErrInvalidToken diff --git a/bootloader/internal/runtimeauth/users.go b/bootloader/internal/runtimeauth/users.go index 43f33754..38bc9bde 100644 --- a/bootloader/internal/runtimeauth/users.go +++ b/bootloader/internal/runtimeauth/users.go @@ -14,13 +14,9 @@ import ( _ "modernc.org/sqlite" // pure-Go SQLite driver: no cgo, cross-compiles ) -// The runtime's users table, from webserver/restapi.py::User. -// -// Read-only, and opened read-only. The bootloader authenticates against these -// accounts but must never create, modify or promote one -- user management -// stays entirely in the runtime, including the first-user bootstrap. A bootloader -// that could write here would be a second, less-reviewed path to an admin -// account on the device. +// Users table from the runtime (webserver/restapi.py::User), opened +// read-only. User management (including first-user bootstrap) stays in +// the runtime; the bootloader never writes. const ( usersTable = "users" openTimeout = 5 * 1000 // busy_timeout, milliseconds @@ -31,24 +27,13 @@ const ( // unauthenticated caller which usernames exist. var ErrNoSuchUser = errors.New("no such user") -// ErrNoUsers means the runtime has never had an account created. -// -// The bootloader refuses every command in that state, deliberately. First-user -// bootstrap is a sensitive flow and it lives in the runtime alone; duplicating -// it here would mean two places that can mint the first admin on a device. -// The practical consequence is narrow: it only bites if the very first runtime -// start fails before anyone has logged in, and install.sh runs with shell -// access anyway. +// ErrNoUsers means no account has been created. Bootloader refuses every +// command in that state so it cannot mint a first admin. var ErrNoUsers = errors.New("no users have been created yet") -// ErrNoDatabase means there is no account database to read. -// -// A nil UserStore is a legitimate state, not a programming error: on a device -// whose runtime has never started there is no restapi.db yet, and the -// bootloader must still come up so an operator can find out why. Every method -// below tolerates a nil receiver, because a typed nil assigned to an interface -// is NOT nil at the call site -- without these guards the first request on -// such a device would panic the bootloader into a restart loop. +// ErrNoDatabase: the runtime has never started so restapi.db is absent. +// A nil UserStore is a legitimate state; every method tolerates a nil +// receiver (typed nil in an interface is not nil at the call site). var ErrNoDatabase = errors.New("the runtime account database is not available") // User is the subset of an account the bootloader needs. @@ -64,13 +49,9 @@ type UserStore struct { db *sql.DB } -// OpenUserStore opens the runtime database read-only. -// -// mode=ro is what makes a read-only bind mount work: SQLite would otherwise -// want to create a rollback journal beside the file and fail on the mount -// rather than on the query. immutable is NOT set -- the runtime writes to this -// database while we read it, and immutable would tell SQLite the file can -// never change, which would serve stale pages after a password change. +// OpenUserStore opens the runtime database read-only. mode=ro so SQLite +// does not try to create a rollback journal. immutable is NOT set since +// the runtime writes while we read (password change). func OpenUserStore(dbPath string) (*UserStore, error) { dsn := fmt.Sprintf("file:%s?mode=ro&_pragma=busy_timeout(%d)", url.PathEscape(dbPath), openTimeout) @@ -93,11 +74,8 @@ func (s *UserStore) Close() error { return s.db.Close() } -// CountUsers reports how many accounts exist. -// -// Used to answer "is this device bootstrapped". A missing table counts as -// zero rather than an error: a runtime that has never started leaves the file -// present but empty, and that is the no-users case, not a broken database. +// CountUsers reports how many accounts exist. A missing table counts as +// zero: a never-started runtime leaves the file empty, not broken. func (s *UserStore) CountUsers(ctx context.Context) (int, error) { if s == nil || s.db == nil { return 0, ErrNoDatabase @@ -145,12 +123,9 @@ func (s *UserStore) FindUser(ctx context.Context, username string) (*User, error return &user, nil } -// Authenticate verifies a username and password, returning the account. -// -// Both a missing user and a bad password come back as ErrNoSuchUser so the -// caller cannot accidentally answer differently for the two. The password is -// still hashed for an unknown user -- see below -- so the two paths cost -// roughly the same time. +// Authenticate verifies username and password. Both missing user and +// bad password return ErrNoSuchUser; the dummy hash below equalises +// timing so an unknown username does not return faster. func (s *UserStore) Authenticate(ctx context.Context, username, password, pepper string) (*User, error) { if s == nil || s.db == nil { return nil, ErrNoDatabase @@ -178,12 +153,8 @@ func (s *UserStore) Authenticate(ctx context.Context, username, password, pepper return user, nil } -// RoleByID reports the role of the account a token was issued for. -// -// Read at request time rather than carried in the token. The token has no role -// claim, and adding one would mean a role change only took effect when the -// token expired -- a demoted account would keep administrative access to the -// component that can pull and run any image on the device. +// RoleByID reads the account role at request time (not from the token) +// so a demotion takes effect immediately, not at token expiry. func (s *UserStore) RoleByID(ctx context.Context, userID string) (string, error) { if s == nil || s.db == nil { return "", ErrNoDatabase diff --git a/bootloader/internal/runtimespec/spec.go b/bootloader/internal/runtimespec/spec.go index aaa99d6a..2e83a487 100644 --- a/bootloader/internal/runtimespec/spec.go +++ b/bootloader/internal/runtimespec/spec.go @@ -2,51 +2,27 @@ // Copyright (c) 2026 Autonomy® // Package runtimespec decides how the runtime container is run. +// Every flag below is load-bearing: // -// This is the ONE place those flags exist. The plan settled on a single -// privilege level rather than a matrix of profiles, because multiple profiles -// mean multiple ways to be misconfigured and a support matrix nobody can hold -// in their head. Every flag below is load-bearing: +// - Privileged + /dev bind: parity with the host-root install so +// SPI_IOC_MESSAGE and GPIO line-handle ioctls reach real devices +// and hot-plugged serial adapters appear without mknod. +// - NetworkMode host: NICs under their real names for EtherCAT +// AF_PACKET and UDP discovery broadcasts. +// - UTSMode host: the device's live hostname, so discovery does +// not report a container id captured at image build time. +// - No CPU limits, ever: Cpus/CpuQuota/CpuPeriod/Memory enable the +// cgroup CPU controller, and under CONFIG_RT_GROUP_SCHED a +// non-root cgroup starts at rt_runtime_us=0, which makes +// sched_setscheduler(SCHED_FIFO) fail. No fields exist for them. +// - rtprio/memlock ulimits: redundant under Privileged (CAP_SYS_NICE +// bypasses RLIMIT_RTPRIO, CAP_IPC_LOCK bypasses RLIMIT_MEMLOCK), +// kept so a de-privileged container still works. +// - RestartPolicy "no": the supervisor owns the lifecycle; a Docker +// restart would race crash-loop accounting. // -// - Privileged + /dev bind: exact parity with the current root install. -// Verified against the SLM-RP4 HAL, which drives /dev/spidev6.0 through -// SPI_IOC_MESSAGE and /dev/gpiochip0 through the GPIO line-handle ioctls. -// Binding the host's live devtmpfs also means hot-plugged serial adapters -// appear without mknod or device cgroup rules. -// -// - NetworkMode host: every NIC visible under its real name in the host's -// own namespace. EtherCAT needs AF_PACKET and SIOCSIFFLAGS on a real -// interface, and the UDP discovery responder needs to see broadcasts. -// Deliberately NOT the orchestrator's dedicated-NIC mechanism, which moves -// a host NIC into a container namespace and removes it from the host. -// -// - UTSMode host: the device's hostname, live, which is what discovery -// reports. NetworkMode host alone copies it once at CREATE time, so an -// image built in a container ships that container's id (RTOP-292). -// -// - No CPU limits, ever. This is the one trap that survives "just make it -// privileged", because it is not a privilege. Setting Cpus/CpuQuota/ -// CpuPeriod/Memory enables the cgroup CPU controller, and with -// CONFIG_RT_GROUP_SCHED a non-root cgroup starts at rt_runtime_us = 0 -- -// at which point sched_setscheduler(SCHED_FIFO) fails outright and the -// runtime silently loses real-time scheduling. There is no field for them -// in this package, so they cannot be set by accident. CpusetCpus would be -// safe (pinning is not bandwidth throttling) but nothing needs it yet. -// -// - rtprio/memlock ulimits are redundant under Privileged, since -// CAP_SYS_NICE bypasses RLIMIT_RTPRIO and CAP_IPC_LOCK bypasses -// RLIMIT_MEMLOCK. They stay as documented intent, and they are what saves -// the deployment if anyone ever de-privileges the container. -// -// - RestartPolicy "no": the supervisor owns the lifecycle. Letting Docker -// also restart it would race the crash-loop accounting and hide exactly -// the signal recovery mode depends on. -// -// Board-specific additions come from a JSON file in the bootloader's own volume, -// written by install.sh. That file may only ADD binds and environment; it can -// never remove privilege, change the network mode, or introduce a CPU limit. -// Validation is strict because the file is the one operator-supplied input to -// a component that runs as host root. +// Board-specific JSON additions may only ADD binds and environment; +// they cannot remove privilege, change network mode, or add CPU limits. package runtimespec import ( @@ -93,14 +69,9 @@ type CreatePayload struct { type Config struct { // Repository is the image repository, without a tag. Repository string `json:"repository"` - // Version is the tag currently desired. The bootloader rewrites this when an - // update succeeds, which is what makes the choice survive a reboot. - // - // Read and written from different goroutines -- the updater writes it, the - // API and discovery replies read it, and the supervisor's event loop reads - // it through ImageRef -- so it goes through Version()/SetVersion() and - // the mutex below. Touching the field directly is a data race; the tests - // only passed under -race because nothing in them read it concurrently. + // Version is the tag currently desired. Writes come from the updater, + // reads from the API, discovery and supervisor, so only touch through + // Version()/SetVersion()+mu below -- direct access is a data race. Version string `json:"version"` // DataDir is the host path holding the runtime's persistent data. Bound // into the container at the same path so the runtime's own defaults apply @@ -130,10 +101,9 @@ const ( UTSModeHost = "host" ) -// forbiddenBindTargets are host paths that must never be handed to the runtime -// container. The docker socket is the important one: mounting it would give -// the runtime's HTTP API control of every container on the host, which is -// precisely the privilege the bootloader exists to keep away from it. +// Host paths the runtime container must never mount. The docker socket +// is the one that matters: mounting it would hand the runtime's HTTP API +// control of every container on the host. var forbiddenBindSources = []string{ "/var/run/docker.sock", "/run/docker.sock", @@ -273,11 +243,8 @@ func (c *Config) ImageRef() string { return c.Repository + ":" + c.DesiredVersion() } -// DesiredVersion reports the tag currently desired. -// -// Every read outside (de)serialisation goes through here. The Version field -// stays exported because encoding/json needs it to be, but reading it -// directly from a goroutine other than the one that wrote it is a data race. +// DesiredVersion returns the current desired tag under the mutex. The +// Version field stays exported for encoding/json; direct access races. func (c *Config) DesiredVersion() string { c.mu.RLock() defer c.mu.RUnlock() @@ -309,32 +276,9 @@ func (c *Config) ContainerSpec(imageRef string) any { binds = append(binds, c.ExtraBinds...) env := []string{ - // OPENPLC_UPDATE_POLICY and OPENPLC_BOOTLOADER_PORT used to be set - // here for /api/capabilities to echo back. The runtime side of that - // was removed as dead weight -- the editor learns both facts from the - // bootloader answering at all -- so setting them told nobody - // anything. Passing environment a runtime does not read is how a - // reader ends up believing a feature exists. - // Point the runtime's persistent data at the bind mount. - // - // This is load-bearing and NOT redundant with the bind. The runtime - // resolves its own data directory by DETECTION, not by what is - // mounted: webserver/config.py::get_persistent_data_dir() returns - // /var/run/runtime whenever is_running_in_container() is true. Without - // this override the runtime writes a fresh .env and restapi.db inside - // the container and never touches the mounted ones -- so users, - // credentials, the stored project, retained variables and any VPP - // licenses would all be discarded on every single version swap, which - // is precisely what persisting them outside the container is for. - // - // Confirmed on hardware before this line existed: the container held - // its own .env under /var/run/runtime while the mounted restapi.db, - // project_snapshot/ and retain.bin sat unused beside it. - // - // Only the PERSISTENT dir is redirected. RUNTIME_DIR keeps its default - // so the command and log sockets stay container-internal, which is - // correct -- they are ephemeral and both endpoints live in the same - // container. + // Override the runtime's auto-detected persistent dir so it writes + // to the bind mount instead of /var/run/runtime inside the + // container. RUNTIME_DIR keeps its default (sockets are ephemeral). "OPENPLC_PERSISTENT_DATA_DIR=" + c.DataDir, } env = append(env, c.ExtraEnv...) diff --git a/bootloader/internal/runtimespec/spec_test.go b/bootloader/internal/runtimespec/spec_test.go index e64f007a..72fd95b9 100644 --- a/bootloader/internal/runtimespec/spec_test.go +++ b/bootloader/internal/runtimespec/spec_test.go @@ -189,11 +189,9 @@ func TestContainerSpecCarriesTheParityFlags(t *testing.T) { } func TestContainerSpecNeverSetsACPULimit(t *testing.T) { - // The one trap that survives "just make it privileged", because it is not - // a privilege: any of these enables the cgroup CPU controller, and with - // CONFIG_RT_GROUP_SCHED a non-root cgroup starts at rt_runtime_us = 0, so - // sched_setscheduler(SCHED_FIFO) fails and the runtime loses real-time - // scheduling silently. + // Any CPU-limit field activates the cgroup CPU controller; with + // CONFIG_RT_GROUP_SCHED that sets rt_runtime_us=0 and + // sched_setscheduler(SCHED_FIFO) silently fails. cfg := &Config{Version: "v4.2.1"} cfg.applyDefaults() spec := decodeSpec(t, cfg) @@ -228,11 +226,9 @@ func TestContainerSpecSetsTheRealTimeUlimits(t *testing.T) { } func TestContainerSpecSetsNoEnvironmentTheRuntimeIgnores(t *testing.T) { - // OPENPLC_UPDATE_POLICY and OPENPLC_BOOTLOADER_PORT were set here for - // /api/capabilities to echo back. That runtime-side reporting was removed - // as dead weight -- a client learns both facts from the bootloader - // answering at all -- so these told nobody anything, while reading like a - // feature that existed. + // Guard against re-adding env vars the runtime does not read. The + // previous OPENPLC_UPDATE_POLICY/BOOTLOADER_PORT were removed when + // /api/capabilities stopped echoing them. cfg := &Config{ Repository: "ghcr.io/x/runtime", Version: "v4.2.1", @@ -326,13 +322,9 @@ func TestSaveLeavesNoTempFileBehind(t *testing.T) { } func TestTheRuntimeIsPointedAtTheMountedDataDirectory(t *testing.T) { - // The bind alone is not enough, and this is the bug that proved it on - // hardware. The runtime resolves its persistent data directory by - // DETECTION -- config.py returns /var/run/runtime whenever it thinks it - // is containerized -- so without this override it writes a fresh .env and - // restapi.db inside the container and ignores the mounted ones. Every - // version swap would then discard users, credentials, the stored project, - // retained variables and any VPP licenses. + // Env override is required: the runtime detects its data dir and + // defaults to /var/run/runtime inside the container, ignoring the + // bind mount unless OPENPLC_PERSISTENT_DATA_DIR redirects it. cfg := &Config{Version: "v4.2.1", DataDir: "/var/lib/openplc-runtime"} cfg.applyDefaults() spec := decodeSpec(t, cfg) diff --git a/bootloader/internal/selfupdate/selfupdate.go b/bootloader/internal/selfupdate/selfupdate.go index cf559d9e..d1d63d14 100644 --- a/bootloader/internal/selfupdate/selfupdate.go +++ b/bootloader/internal/selfupdate/selfupdate.go @@ -1,25 +1,12 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// Package selfupdate replaces the bootloader with a newer version of itself. -// -// A container cannot replace itself: removing it kills the process doing the -// removing, halfway through. So the running bootloader spawns a ONE-SHOT child -// from the new image, and that child does the work from outside -- the same -// shape orchestrator-agent uses in tools/upgrade_self.py, which is proven in -// production. -// -// The runtime container is never touched. A bootloader update must not -// interrupt a running PLC: losing the ability to manage a device is a bad -// afternoon, stopping its plant is a different category of problem. That is -// also why the failure mode is acceptable -- if the new bootloader will not -// start, Docker's restart policy keeps trying while the runtime carries on. -// -// The child reproduces the parent's configuration from the RUNNING container -// rather than from defaults. An operator may have installed with extra mounts -// or a non-standard port, and a self-update that quietly dropped them would -// leave a device subtly wrong in a way nobody would connect to "the bootloader -// updated itself". +// Package selfupdate replaces the bootloader with a newer version of +// itself. A container cannot remove itself mid-process, so the parent +// spawns a one-shot child from the new image that does the swap from +// outside. The runtime container is never touched. The child +// reproduces the parent's container config (binds, port, env) from +// the RUNNING container, not from defaults. package selfupdate import ( @@ -147,11 +134,8 @@ func Start(ctx context.Context, docker DockerClient, repository, version string, return nil } -// Execute is the child's side: replace the parent and exit. -// -// Idempotent by design. A parent that is already gone -- because a previous -// attempt got that far before dying -- is not an error; the goal is that a -// bootloader on the new image is running when this finishes. +// Execute replaces the parent and exits. Idempotent: a parent already +// gone from an interrupted attempt is not an error. func Execute(ctx context.Context, docker DockerClient, log *slog.Logger) error { target := os.Getenv(EnvTarget) newImage := os.Getenv(EnvNewImage) @@ -187,13 +171,8 @@ func Execute(ctx context.Context, docker DockerClient, log *slog.Logger) error { case <-time.After(settleDelay): } - // Create the replacement FIRST, under a temporary name. - // - // Removing the parent first meant a rejected create -- an invalid - // HostConfig on an older daemon, a full disk, an image pruned between the - // pull and the create -- left the device with no bootloader at all, on - // hardware this feature exists because it has no SSH. The helper runs with - // RestartPolicy: no, so nothing would have come back for it. + // Create the replacement under a temporary name BEFORE removing the + // parent, so a rejected create leaves the old bootloader intact. staging := target + "-next" // A leftover from an interrupted attempt would take the name. if err := docker.RemoveContainer(ctx, staging, true); err != nil { @@ -228,12 +207,9 @@ func Execute(ctx context.Context, docker DockerClient, log *slog.Logger) error { return nil } -// replacementSpec rebuilds the parent's create payload with the new image. -// -// The parent's own environment is carried over except the self-update -// variables: leaving those in would put the NEW bootloader straight back into -// child mode on start-up, and it would immediately try to replace itself in a -// loop. +// replacementSpec rebuilds the parent's create payload with the new +// image, dropping the self-update env vars so the new container does +// not immediately re-enter helper mode. func replacementSpec(parent *dockerapi.ContainerInspect, newImage string) map[string]any { env := make([]string, 0, len(parent.Config.Env)) for _, entry := range parent.Config.Env { @@ -266,7 +242,8 @@ func replacementSpec(parent *dockerapi.ContainerInspect, newImage string) map[st "HostConfig": map[string]any{ "Binds": parent.HostConfig.Binds, "NetworkMode": parent.HostConfig.NetworkMode, - // Set, never inherited: a pre-RTOP-292 parent would pass the bug on. + // Set explicitly, never inherited: an older parent with a + // private UTS namespace would otherwise propagate it. "UTSMode": runtimespec.UTSModeHost, "Privileged": parent.HostConfig.Privileged, "RestartPolicy": map[string]any{"Name": restart}, @@ -298,13 +275,9 @@ func defaultSpec(newImage string) map[string]any { } } -// findSelf identifies the container this process is running in. -// -// HOSTNAME is the container's short id under Docker's defaults, which is the -// most direct answer. It can be overridden (--hostname), so a miss falls back -// to the name install.sh uses -- and a miss on both is reported rather than -// guessed at, because every caller of this is about to delete whatever it -// names. +// findSelf identifies the current container by HOSTNAME, falling back +// to the conventional name install.sh uses. A miss on both is reported, +// never guessed: the caller is about to delete whatever this names. func findSelf(ctx context.Context, docker DockerClient) (*dockerapi.ContainerInspect, error) { if hostname := os.Getenv("HOSTNAME"); hostname != "" { if found, err := docker.InspectContainer(ctx, hostname); err == nil { diff --git a/bootloader/internal/selfupdate/selfupdate_test.go b/bootloader/internal/selfupdate/selfupdate_test.go index 09dd44bc..dfe2df1b 100644 --- a/bootloader/internal/selfupdate/selfupdate_test.go +++ b/bootloader/internal/selfupdate/selfupdate_test.go @@ -388,7 +388,7 @@ func TestTheChildRecreatesEvenIfTheParentIsAlreadyGone(t *testing.T) { } } -// A pre-RTOP-292 parent must not pass its namespace on. +// A parent with a private UTS namespace must not pass it on. func TestAParentWithAPrivateUTSNamespaceDoesNotPassItOn(t *testing.T) { // "" is Docker's private default; "private" covers the set-not-defaulted // field an earlier revision inherited. @@ -472,13 +472,7 @@ func TestTheRuntimeContainerIsNeverTouched(t *testing.T) { } } -// A create that fails must leave the device with the bootloader it has. -// -// The parent used to be force-removed first. If the create was then rejected -// -- an invalid HostConfig on an older daemon, a full disk, an image pruned -// between the pull and the create -- the helper exited with RestartPolicy: no -// and the device had no bootloader at all, on hardware that by this feature's -// own framing has no SSH. +// A create that fails must leave the old bootloader running. func TestAFailedCreateLeavesTheOldBootloaderRunning(t *testing.T) { docker := newFake() docker.containers["openplc-bootloader"] = parentContainer() diff --git a/bootloader/internal/supervisor/crashwindow.go b/bootloader/internal/supervisor/crashwindow.go index 62ae47e6..660b5425 100644 --- a/bootloader/internal/supervisor/crashwindow.go +++ b/bootloader/internal/supervisor/crashwindow.go @@ -8,12 +8,8 @@ import ( "time" ) -// Defaults mirror webserver/runtimemanager.py's MAX_RAPID_CRASHES / -// RAPID_CRASH_WINDOW one layer up. That module already does this for -// plc_main: restart it, count crashes in a window, and stop restarting when -// the fault is clearly not transient. The bootloader applies the same shape to -// the container, so the two layers behave predictably alike and neither -// masks the other's failure. +// Defaults mirror the runtime's own MAX_RAPID_CRASHES/RAPID_CRASH_WINDOW +// so container-level and process-level crash accounting behave alike. const ( DefaultMaxCrashes = 3 DefaultCrashWindow = 5 * time.Minute @@ -24,14 +20,9 @@ const ( MaxRestartDelay = 30 * time.Second ) -// crashWindow counts unexpected container exits inside a sliding window. -// -// Only UNEXPECTED exits belong here. A runtime that exits because we asked it -// to -- an update handshake, a stop we issued -- is not evidence of a fault, -// and counting those would make the first update look like a crash-loop and -// drop a perfectly healthy device into recovery. Callers gate on -// Supervisor.expectStop rather than filtering by exit code, because a -// deliberate stop and a genuine crash can both exit non-zero. +// crashWindow counts UNEXPECTED container exits inside a sliding +// window. Callers gate on Supervisor.expectStop rather than exit code +// since a deliberate stop and a crash can both exit non-zero. type crashWindow struct { mu sync.Mutex times []time.Time diff --git a/bootloader/internal/supervisor/supervisor.go b/bootloader/internal/supervisor/supervisor.go index 05f09799..d9e083f7 100644 --- a/bootloader/internal/supervisor/supervisor.go +++ b/bootloader/internal/supervisor/supervisor.go @@ -1,28 +1,15 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// Package supervisor owns the runtime container's lifecycle. +// Package supervisor owns the runtime container's lifecycle. Reconciles +// the container at boot, blocks on the Docker events stream, restarts +// on crash, enters recovery on crash-loop. // -// This is the part of the bootloader that decides what the runtime container -// should be doing: at boot it reconciles that container into existence, then -// sits blocked on the Docker events stream and does nothing until something -// happens. When the runtime dies it restarts it, and when it dies repeatedly -// it stops trying and enters recovery, so an operator can reach the device -// from the editor instead of the bootloader hammering a runtime that will -// never come up. -// -// Two boundaries are deliberate and easy to get wrong: -// -// - Health means the runtime WEBSERVER came up. Whether plc_main is running, -// whether a program is loaded, and whether that program errors are all the -// webserver's concern -- it already restarts plc_main and drops to safe -// mode on rapid crashes. If the bootloader looked at PLC state, a user -// uploading broken logic would trigger a runtime recovery, which would be -// a spectacular way to turn a program bug into a device outage. -// -// - There is no automatic rollback. A failed update or a crash-loop stops -// and waits for a human. Choosing a version is a decision with physical -// consequences, and guessing wrong twice is worse than stopping once. +// - Health means the runtime webserver came up. PLC state is the +// webserver's concern; caring about it here would turn a bad +// program into a device outage. +// - No automatic rollback. A failed update or crash-loop stops and +// waits for a human. package supervisor import ( @@ -72,13 +59,8 @@ type Status struct { HealthSource string `json:"healthSource,omitempty"` } -// DockerClient is the slice of the Docker API the supervisor uses. -// -// An interface rather than *dockerapi.Client so the state machine can be -// tested directly. The subtle logic here is crash accounting -- distinguishing -// an exit we asked for from one we did not -- and that is exactly the kind of -// thing that is wrong in a way no integration test notices until a device -// drops into recovery during its first successful update. +// DockerClient is the Docker API slice the supervisor uses. An +// interface so the crash-accounting state machine is unit-testable. type DockerClient interface { Ping(ctx context.Context) error InspectContainer(ctx context.Context, name string) (*dockerapi.ContainerInspect, error) @@ -87,10 +69,8 @@ type DockerClient interface { StopContainer(ctx context.Context, name string, grace time.Duration) error RemoveContainer(ctx context.Context, name string, force bool) error StreamEvents(ctx context.Context, name string, handle func(dockerapi.Event)) error - // Image access, so the supervisor can fetch a runtime it has been told to - // run but does not have. Needed on a fresh install -- install.sh writes - // the spec and starts the bootloader without pulling anything -- and - // after a data wipe or an operator editing the spec by hand. + // Image access: the supervisor fetches a runtime it has been told to + // run but does not have (fresh install, data wipe, hand-edited spec). InspectImage(ctx context.Context, ref string) (*dockerapi.ImageInfo, error) PullImage(ctx context.Context, ref string, onProgress func(dockerapi.PullProgress)) error } @@ -170,10 +150,8 @@ type Supervisor struct { mu sync.Mutex // status is the current externally visible condition. status Status - // expectStop suppresses crash accounting while we are deliberately taking - // the container down. Counted rather than boolean: an update stops the - // container and a concurrent reconcile must not clear the suppression - // early, which would make our own stop look like a crash. + // expectStop suppresses crash accounting during a deliberate stop. + // Counted so a concurrent reconcile cannot clear it early. expectStop int // preUpdateState is what BeginUpdate displaced, so an update that changes @@ -281,12 +259,9 @@ func (s *Supervisor) Run(ctx context.Context) error { return s.watch(ctx) } -// watch consumes the events stream, reconnecting on failure. -// -// Every reconnect re-reconciles. The stream can only report what happened -// while it was open, so a gap -- most often the daemon restarting -- may hide -// a container exit. Re-inspecting is the only way to be sure the world still -// matches what we believe. +// watch consumes the events stream and reconnects on failure. Every +// reconnect also re-reconciles: the stream can hide a container exit +// during a gap (daemon restart). func (s *Supervisor) watch(ctx context.Context) error { const reconnectDelay = 2 * time.Second for { @@ -308,15 +283,9 @@ func (s *Supervisor) watch(ctx context.Context) error { case <-time.After(reconnectDelay): } - // Re-sync before trusting the new stream -- but NOT while recovery or - // an update owns the runtime. - // - // Reconcile starts any stopped container it finds. In recovery that - // would restart a runtime the supervisor deliberately stopped, flip - // the state to healthy and disable discovery, with no operator - // involved -- a daemon restart or a socket hiccup was enough. During - // an update it would race the updater's own Reconcile, and the loser - // of the create/remove contention fails the update into recovery. + // Re-sync, but NOT while recovery or an update owns the runtime: + // Reconcile would restart a deliberately-stopped container or + // race the updater's own Reconcile. s.mu.Lock() state := s.status.State s.mu.Unlock() @@ -426,13 +395,9 @@ func (s *Supervisor) handleWedged(ctx context.Context) { } } -// Reconcile brings the runtime container to the desired state and is safe to -// call at any time. -// -// Adoption is the important property: a running healthy container is left -// exactly as it is. The bootloader restarts (its own crash, a self-update) far -// more often than the runtime does, and a reconcile that recreated or bounced -// a working runtime would turn a bootloader hiccup into a plant outage. +// Reconcile brings the runtime container to the desired state; safe to +// call at any time. Adoption is the point: a running healthy container +// is left exactly as it is, so a bootloader restart does not bounce it. func (s *Supervisor) Reconcile(ctx context.Context) error { inspect, err := s.docker.InspectContainer(ctx, s.cfg.ContainerName) switch { @@ -453,19 +418,9 @@ func (s *Supervisor) Reconcile(ctx context.Context) error { s.status.Image = inspect.Config.Image s.mu.Unlock() - // Does the existing container actually match what the spec now asks for? - // - // This is what makes Reconcile reconcile rather than merely "start - // whatever is there". A container is created from an image reference and - // keeps it for life, so after a version change the existing one is the OLD - // version -- and the branch below would have happily restarted it and - // reported success. The update then appeared to work while the device kept - // running the version it started with, which is precisely the bug the - // integration suite caught. - // - // It also covers an operator editing the spec by hand (a board mount, an - // env var) and restarting the bootloader: the container is rebuilt from - // the spec instead of silently keeping the old configuration. + // A container keeps its image for life. After a version change the + // existing one is the OLD version, so Reconcile must replace it, not + // just restart it, otherwise the update silently has no effect. desired := s.spec.ImageRef() if stale, reason := s.containerIsStale(ctx, inspect, desired); stale { s.log.Info("runtime container needs replacing, recreating", @@ -512,26 +467,19 @@ func (s *Supervisor) Reconcile(ctx context.Context) error { } } -// containerIsStale reports whether the running container needs replacing, and -// why. The reason is logged, so an unexpected recreate can be traced. -// -// Compares resolved image IDs, not tag strings. A container pins its image by -// ID, so a re-pull of the same tag can leave the container on the OLD layers -// while Config.Image still reads as a match -- which made "reinstall the -// current version", the documented repair path, silently a no-op that started -// the very layers the operator was trying to replace. -// -// The tag comparison stays as the first check because it is free and catches -// the ordinary version change; the ID lookup only runs when the tags agree. +// containerIsStale reports whether the running container needs +// replacing, and why. Compares resolved image IDs (not tags) because a +// re-pull of the same tag can leave the container on the OLD layers. func (s *Supervisor) containerIsStale( ctx context.Context, inspect *dockerapi.ContainerInspect, desired string, ) (bool, string) { if inspect.Config.Image != desired { return true, "image tag differs from the desired one" } - // A pre-RTOP-292 container reports a container id and the image can match, - // so nothing else notices. Latched: it alone reads what the daemon reports - // back, which an engine could omit and loop forever. + // A container with a private UTS namespace reports a container id, + // and the image can still match, so nothing else notices. Latched: + // it alone reads what the daemon reports back, which an engine + // could omit and loop forever. if !s.utsRecreateDone() && inspect.HostConfig.UTSMode != runtimespec.UTSModeHost { return true, "container does not share the host UTS namespace, so " + "discovery would report a container id as the device name" @@ -565,11 +513,9 @@ func (s *Supervisor) markUTSRecreated() { s.utsRecreated = true } -// recreate replaces the container, stopping it gracefully first if it runs. -// -// create() force-removes, and a SIGKILL skips the runtime's SIGTERM handler -// that flushes retained variables. The kill's exit would also arrive as an -// unmarked death and be counted as a crash. +// recreate replaces the container, stopping gracefully first so the +// runtime's SIGTERM handler flushes retained variables (create's +// force-remove sends SIGKILL and would also look like a crash). func (s *Supervisor) recreate(ctx context.Context, inspect *dockerapi.ContainerInspect) error { if inspect != nil && inspect.State.Running { if err := s.Stop(ctx); err != nil { @@ -608,17 +554,9 @@ func (s *Supervisor) create(ctx context.Context) error { return nil } -// ensureImage pulls imageRef when it is not already present. -// -// The bootloader is what fetches the runtime on a fresh device: install.sh -// writes the spec and starts the bootloader without pulling anything, so -// without this a brand-new install would go straight to recovery with "No -// such image". It also covers a spec that names a version whose image was -// retired, or one an operator edited by hand. -// -// Only pulls when the image is absent. A present image is never re-pulled -- -// that would turn every restart into a network round trip, and on a slow link -// into minutes of delay before a PLC that was working comes back. +// ensureImage pulls imageRef only when it is not already present. +// Fresh installs rely on this (install.sh pulls nothing). Never +// re-pulls a present image: a working PLC must not pay for a restart. func (s *Supervisor) ensureImage(ctx context.Context, imageRef string) error { if _, err := s.docker.InspectImage(ctx, imageRef); err == nil { return nil @@ -649,15 +587,9 @@ func (s *Supervisor) ensureImage(ctx context.Context, imageRef string) error { func (s *Supervisor) startAndConfirm(ctx context.Context) error { s.setState(StateStarting, "starting runtime") - // Release anything the bootloader holds that the runtime is about to - // claim -- the UDP discovery port -- BEFORE the container starts. - // - // Waiting for the Healthy transition was too late: the runtime binds - // 33333 once at start-up with SO_REUSEADDR and never retries, the - // bootloader's responder binds it without, and Linux only shares a UDP - // port when every socket asked to. So the runtime got EADDRINUSE, logged a - // warning, and the device was undiscoverable after every recovery until - // its next restart. + // Release the UDP discovery port BEFORE the container starts: the + // runtime binds it once with SO_REUSEADDR and never retries, so a + // late release leaves the device undiscoverable until next restart. if s.onRuntimeStarting != nil { s.onRuntimeStarting() } @@ -668,11 +600,9 @@ func (s *Supervisor) startAndConfirm(ctx context.Context) error { return s.awaitHealthy(ctx) } -// awaitHealthy polls until the runtime is healthy or StartTimeout elapses. -// -// Polling, not events: a container that never becomes healthy emits no event -// to wait for, so a timeout is the only way to notice. The poll is on the -// bootloader's own clock and touches nothing in the scan path. +// awaitHealthy polls until the runtime is healthy or StartTimeout +// elapses. Polling (not events) because a never-healthy container emits +// no event. func (s *Supervisor) awaitHealthy(ctx context.Context) error { deadline := time.Now().Add(s.cfg.StartTimeout) const pollInterval = 2 * time.Second @@ -734,16 +664,9 @@ func (s *Supervisor) confirmByProbe(ctx context.Context) error { return nil } -// markHealthy records steady state and clears the restart backoff. -// -// It deliberately does NOT clear the crash window. A crash-loop is a runtime -// that dies, comes back up fine, and dies again -- which is the common shape, -// because a program that faults on load lets the webserver start before it -// takes the process down. Resetting the count on every healthy start would -// zero the evidence between each crash, so the loop could never reach the -// threshold and the supervisor would restart forever instead of handing the -// device to an operator. The window forgets by aging entries out, which is all -// the forgetting that is wanted: crashes weeks apart never accumulate. +// markHealthy records steady state and clears restart backoff. It does +// NOT clear the crash window: the window ages entries out, which is the +// only forgetting wanted (otherwise die/healthy/die loops never trip). func (s *Supervisor) markHealthy(source string) { s.mu.Lock() s.consecutiveFailures = 0 @@ -752,12 +675,9 @@ func (s *Supervisor) markHealthy(source string) { s.setState(StateHealthy, "") } -// enterRecovery stops the runtime and switches to recovery mode. -// -// Stopping first is what makes UDP discovery exclusive: only one service on -// the host may answer the broadcast, and recovery is defined as "the runtime -// is not running", so the responder can be switched on without ever racing -// the runtime's own. +// enterRecovery stops the runtime, then switches to recovery mode. The +// stop first makes the UDP discovery responder exclusive (only one +// service on the host answers the broadcast). func (s *Supervisor) enterRecovery(ctx context.Context, reason string) { s.markExpectedStop() if err := s.docker.StopContainer(ctx, s.cfg.ContainerName, s.cfg.StopGrace); err != nil { @@ -778,10 +698,8 @@ func (s *Supervisor) EnterRecovery(ctx context.Context, reason string) { s.enterRecovery(ctx, reason) } -// BeginUpdate claims the supervisor for a version change, suppressing crash -// accounting for the stop that is about to happen. It returns an error when an -// update is already running: two concurrent swaps of the same container is not -// a situation worth trying to make safe. +// BeginUpdate claims the supervisor for a version change and suppresses +// crash accounting for the imminent stop. Refuses concurrent updates. func (s *Supervisor) BeginUpdate() error { s.mu.Lock() defer s.mu.Unlock() @@ -799,14 +717,9 @@ func (s *Supervisor) BeginUpdate() error { return nil } -// AbortUpdate undoes BeginUpdate for an update that changed nothing. -// -// The alternative was calling Reconcile, which re-derived the state by acting -// on the device: from recovery -- the common case, where an operator typed a -// version that does not exist -- it found the stopped container and started -// it, leaving recovery with no operator decision. With the container absent it -// set "starting", the pull failed again, and the device reported starting -// indefinitely. "Nothing was changed" has to include the supervisor's state. +// AbortUpdate restores the pre-BeginUpdate state. Does not Reconcile: +// re-deriving state would act on the device and could silently exit +// recovery (where the operator typed a non-existent version). func (s *Supervisor) AbortUpdate() { s.mu.Lock() if s.expectStop > 0 { diff --git a/bootloader/internal/supervisor/supervisor_test.go b/bootloader/internal/supervisor/supervisor_test.go index c7dca240..6c5229ae 100644 --- a/bootloader/internal/supervisor/supervisor_test.go +++ b/bootloader/internal/supervisor/supervisor_test.go @@ -35,8 +35,9 @@ type fakeDocker struct { // image so existing tests keep describing a matching container. configImage string - // privateUTS models a pre-RTOP-292 container. Defaults off, so every other - // test still describes a container to leave alone. + // privateUTS models an older container with a private UTS namespace. + // Defaults off so every other test still describes a container + // the supervisor should leave alone. privateUTS bool // neverReportsUTS models an engine that accepts UTSMode and never echoes @@ -388,11 +389,8 @@ func TestRecoveryStopsTheRuntimeSoDiscoveryStaysExclusive(t *testing.T) { } func TestASuccessfulRestartDoesNotEraseTheCrashHistory(t *testing.T) { - // The common crash-loop shape is: die, come back up fine, die again -- a - // program that faults on load lets the webserver start before it takes the - // process down. If a healthy start cleared the count, the evidence would be - // zeroed between every crash, the threshold could never be reached, and the - // supervisor would restart forever instead of handing the device over. + // Classic crash loop: die, healthy, die. A healthy start must not + // reset the window or the threshold could never be reached. docker := &fakeDocker{exists: true, running: true, health: "healthy", startMakesHealthy: true, imagePresent: true} sup := newTestSupervisor(docker, &fakeProbe{}) ctx := context.Background() @@ -550,8 +548,8 @@ func TestStartTimeoutIsReportedRatherThanHanging(t *testing.T) { } func TestRunEntersRecoveryWhenTheRuntimeCannotStart(t *testing.T) { - // Recovery must be reachable precisely when the runtime will not come up: - // that is the case RTOP-283 exists for. + // Recovery must be reachable precisely when the runtime will not + // come up — the whole reason the bootloader exists. docker := &fakeDocker{startErr: errors.New("no such image"), imagePresent: true} sup := newTestSupervisor(docker, &fakeProbe{err: errors.New("down")}) @@ -567,10 +565,8 @@ func TestRunEntersRecoveryWhenTheRuntimeCannotStart(t *testing.T) { // --- image acquisition --------------------------------------------------- func TestAFreshInstallPullsTheRuntimeImage(t *testing.T) { - // The bootloader is what fetches the runtime on a new device: install.sh - // writes the spec and starts the bootloader without pulling anything. - // Without this the very first boot goes straight to recovery with - // "No such image", which is what the integration suite caught. + // install.sh starts the bootloader without pulling, so the first + // boot would go straight to recovery without this pull. docker := &fakeDocker{startMakesHealthy: true, imagePresent: false} sup := newTestSupervisor(docker, &fakeProbe{}) @@ -617,11 +613,8 @@ func TestAFailedImagePullIsReportedWithTheImageName(t *testing.T) { } func TestTheDownloadIsVisibleInTheStatusWhileItRuns(t *testing.T) { - // On a slow device this pull runs for minutes. The difference between - // "downloading 50%" and an apparently hung device is whether the editor - // has anything to show, so the reason has to be updated DURING the pull, - // not merely at the end. Captured from inside the progress callback, - // because by the time Reconcile returns the state is already healthy. + // Reason must be updated DURING the pull, not at the end. Captured + // from inside onPullProgress because Reconcile returns healthy. docker := &fakeDocker{startMakesHealthy: true, imagePresent: false} sup := newTestSupervisor(docker, &fakeProbe{}) docker.onPullProgress = func() { docker.observed = sup.Status() } @@ -642,11 +635,8 @@ func TestTheDownloadIsVisibleInTheStatusWhileItRuns(t *testing.T) { } func TestAContainerOnTheWrongImageIsRecreated(t *testing.T) { - // The bug the integration suite caught. A container keeps the image - // reference it was created from for life, so after a version change the - // existing container is the OLD version. Reconcile used to see "exists but - // stopped" and simply start it again -- so the update reported success - // while the device carried on running the version it started with. + // A container keeps its image for life; after a version change + // Reconcile must recreate, not just restart. docker := &fakeDocker{ exists: true, running: false, imagePresent: true, startMakesHealthy: true, // fakeSpec serves "test:1"; this container was built from something else. @@ -665,7 +655,8 @@ func TestAContainerOnTheWrongImageIsRecreated(t *testing.T) { } } -// RTOP-292: the image checks see nothing wrong, so a pinned board keeps it. +// Image checks see nothing wrong with a private-UTS container, so the +// supervisor has to recreate it on its own. func TestAContainerWithAPrivateUTSNamespaceIsRecreated(t *testing.T) { docker := &fakeDocker{ exists: true, running: true, health: "healthy", imagePresent: true, @@ -768,15 +759,9 @@ func TestAContainerOnTheRightImageIsStillAdopted(t *testing.T) { // A crash after a stop that did nothing must still count as a crash. // -// The leak this pins: enterRecovery and Stop suppress crash accounting for the -// exit they are about to cause, but a container that has ALREADY exited never -// emits a `die` event -- and the daemon answers the stop with 304/404. The -// suppression therefore outlived the stop and was spent on the next genuine -// crash instead, which the supervisor then read as "stopped as expected" and -// did not restart. The PLC stayed down while Status() still said healthy. -// -// The crash-loop path always ends in that state: the third die is what calls -// enterRecovery, so by then the container is already gone. +// Pins: a stop against an already-exited container (no die event, 304 +// from the daemon) must not leave the expectStop counter armed to +// suppress the NEXT genuine crash. func TestAStopThatDidNothingDoesNotSuppressTheNextCrash(t *testing.T) { docker := &fakeDocker{ exists: true, running: true, health: "healthy", diff --git a/bootloader/internal/updater/disk_linux.go b/bootloader/internal/updater/disk_linux.go index 5d1e707e..9bfc4c6a 100644 --- a/bootloader/internal/updater/disk_linux.go +++ b/bootloader/internal/updater/disk_linux.go @@ -11,18 +11,9 @@ import ( ) // freeBytes reports free space on the filesystem holding path. -// -// Statfs on the bootloader's own state directory, not the Docker data root: -// the bootloader has no mount of /var/lib/docker, and Docker's API exposes -// image sizes but no free-space figure at all. On the layout install.sh -// creates both live under /var/lib, so this is the same filesystem. That -// assumption is why the pre-check is a warning-with-a-number rather than a -// hard gate -- an operator who has moved Docker's data-root elsewhere (as the -// AM62xx Yocto board does, to /persist) would otherwise be blocked by a -// measurement of the wrong disk. -// -// Bavail, not Bfree: Bfree counts blocks reserved for root that an ordinary -// write cannot use, so it would overstate what is actually available. +// Measured on the bootloader's state dir (not Docker's data root, which +// is not mounted here). Uses Bavail, not Bfree, so root-reserved blocks +// are excluded. func freeBytes(path string) (int64, error) { var stat syscall.Statfs_t if err := syscall.Statfs(path, &stat); err != nil { diff --git a/bootloader/internal/updater/disk_other.go b/bootloader/internal/updater/disk_other.go index b526931f..68269db1 100644 --- a/bootloader/internal/updater/disk_other.go +++ b/bootloader/internal/updater/disk_other.go @@ -5,13 +5,8 @@ package updater -// freeBytes has no non-Linux implementation. -// -// The bootloader only ever runs on Linux -- it manages Linux containers -// through a Linux daemon. This file exists purely so the package still builds -// and its tests still run on a developer's machine; returning 0 makes the -// pre-check skip rather than fail, which is the right behaviour when the -// measurement is simply unavailable. +// freeBytes stub: exists so the package builds on non-Linux dev +// machines. Returning 0 makes the pre-check skip rather than fail. func freeBytes(_ string) (int64, error) { return 0, nil } diff --git a/bootloader/internal/updater/updater.go b/bootloader/internal/updater/updater.go index 7ca2d2fe..8e028129 100644 --- a/bootloader/internal/updater/updater.go +++ b/bootloader/internal/updater/updater.go @@ -2,26 +2,11 @@ // Copyright (c) 2026 Autonomy® // Package updater changes which runtime version a device runs. -// -// The whole flow, and the reasoning behind its order: -// -// pull new -> stop old -> start new -> health-gate -> remove old -// -// Pull first because `docker pull` is non-destructive: it does not touch the -// existing image, so until the explicit removal at the end the device still -// has a working version on disk. That costs nothing in the steady state -- -// only one image remains afterwards -- and it means a link that dies mid-pull, -// or a new image that will not start, leaves something to fall back to. -// Removing first would save nothing at the moment that matters, since you -// cannot start the new version without having downloaded it anyway. -// -// Upgrade and downgrade are the same operation. There is no version floor: a -// user may deliberately pair an older runtime with an older editor, and the -// bootloader stays reachable either way, so nothing is gained by refusing. -// -// There is no automatic rollback. A failure stops and hands the device to an -// operator in recovery mode, because choosing a version has physical -// consequences and guessing wrong twice is worse than stopping once. +// Flow: pull new -> stop old -> start new -> health-gate -> remove old. +// Pull before stop so a mid-pull failure or an image that will not +// start leaves the old image on disk. Upgrade and downgrade are the +// same operation; no version floor. No automatic rollback — a failure +// stops in recovery mode and waits for an operator. package updater import ( @@ -187,21 +172,12 @@ func (u *Updater) run(ctx context.Context, targetVersion string) { if err != nil { u.cfg.Log.Error("update failed", "from", previousVersion, "to", targetVersion, "error", err) - // Only a failure that got as far as touching the container hands the - // device to an operator. A bad version name, a full disk or an - // unreachable registry changed nothing -- the runtime is still - // running the version it was, and stopping it would turn a harmless - // refusal into a plant outage. Observed on the SLM-RP4: a failed pull - // stopped a RUNNING PLC. + // Failures that never touched the container leave the running + // runtime intact; recovery would be an unforced plant outage. var beforeSwap errBeforeSwap if errors.As(err, &beforeSwap) { u.cfg.Log.Info("nothing was changed; leaving the runtime alone", "version", previousVersion) - // Put back exactly the state this attempt found. Reconcile used to - // do the re-deriving, but it re-derives by ACTING: from recovery - // it restarted the stopped container and called the device - // healthy, and with the container absent it left the state on - // "starting" forever after the pull failed again. u.cfg.Supervisor.AbortUpdate() return } @@ -240,17 +216,8 @@ func (u *Updater) execute(ctx context.Context, previousVersion, targetVersion st u.setPhase(StatePulling, p.Phase, p.Percent) }) if err != nil { - // A pull can fail for a reason that does not matter: the image is - // already here. That covers an air-gapped device with a side-loaded - // image, a locally built one, and a registry that is merely - // unreachable right now. Refusing in that case would make a version - // the device already holds uninstallable -- which is exactly what - // happened on the SLM-RP4, where a locally tagged image produced - // "pull access denied" and failed an update that could not have - // been more ready to succeed. - // - // Same policy as orchestrator-agent's _pull_runtime_image: only a - // confirmed local copy excuses a failed pull. + // A pull failure is excused only when the image is already present + // locally (air-gapped, side-loaded, offline registry). if _, inspectErr := u.cfg.Docker.InspectImage(ctx, targetRef); inspectErr != nil { // The full chain goes to the log; the operator gets one sentence. u.cfg.Log.Error("pull failed", "image", targetRef, "error", err) @@ -301,20 +268,9 @@ func (u *Updater) execute(ctx context.Context, previousVersion, targetVersion st return nil } -// checkDiskSpace reports, with numbers, when the target is unlikely to fit. -// -// Advisory, and now actually advisory: it used to return an error that the -// caller turned into a failed update, contradicting this comment and refusing -// updates on any device whose Docker data-root had been moved -- the estimate -// is taken from the bootloader's own filesystem, which is only the same disk -// on a default install. The operator on the moved-data-root board was the one -// who lost. -// -// So a tight measurement is surfaced as a warning on the progress the editor -// polls, and the pull goes ahead. If the estimate was right, Docker's own pull -// fails with a clear ENOSPC, which is a legible failure rather than a silent -// one -- and if it was about the wrong disk, nothing was refused for no -// reason. +// checkDiskSpace is advisory: a tight measurement becomes a progress +// warning, and the pull proceeds. The estimate measures the bootloader's +// own disk, which only matches Docker's data-root on a default install. func (u *Updater) checkDiskSpace(ctx context.Context, targetRef string) { free, err := freeBytes(u.cfg.StateDir) if err != nil { @@ -361,12 +317,9 @@ func (u *Updater) setPhase(state State, phase string, percent *int) { u.progress.Percent = percent } -// validateVersion rejects a tag the daemon would refuse or that could be used -// to reach an image other than the one intended. -// -// The reference is always built as repository + ":" + version by -// runtimespec.ImageRefFor, so a version containing a slash or a colon could -// otherwise redirect the pull to a different repository or registry entirely. +// validateVersion rejects a tag the daemon would refuse. The reference +// is built as repository + ":" + version, so a slash or colon in the +// version could redirect the pull to a different repository. func validateVersion(version string) error { if version == "" { return errors.New("a version is required") @@ -411,17 +364,9 @@ func humanBytes(n int64) string { return fmt.Sprintf("%.1f PiB", value/unit) } -// describePullFailure turns a failed pull into a sentence an operator can act -// on. -// -// The default is the daemon's own reason, which is usually specific ("manifest -// unknown", "pull access denied"). What it cannot know is the trap behind the -// most confusing case: a device whose spec names a repository with no registry -// host -- "openplc-runtime" rather than "ghcr.io/autonomy-logic/openplc-runtime" -// -- sends every pull to Docker Hub, where none of these images exist. That -// happens on a device installed from a side-loaded image, and the resulting -// "repository does not exist" points nowhere near the actual problem, so the -// configured repository is named explicitly. +// describePullFailure turns a failed pull into an operator-actionable +// sentence. The special case is an unqualified repository (no registry +// host), which Docker silently routes to Docker Hub. func describePullFailure(repository, ref string, err error) string { reason := dockerapi.Reason(err) if isUnqualifiedRepository(repository) { diff --git a/bootloader/internal/updater/updater_test.go b/bootloader/internal/updater/updater_test.go index ce78fb99..17797409 100644 --- a/bootloader/internal/updater/updater_test.go +++ b/bootloader/internal/updater/updater_test.go @@ -450,12 +450,9 @@ func (b *blockingDocker) RemoveImage(context.Context, string, bool) error { retu // --- disk pre-check ------------------------------------------------------ func TestATightDiskIsWarnedAboutRatherThanRefused(t *testing.T) { - // This used to refuse the update. The measurement is of the bootloader's - // filesystem, which is only Docker's on a default install -- so on a - // device whose data-root had been moved, a perfectly possible update was - // blocked by a figure about the wrong disk. It is a warning now, carried - // on the progress the editor polls, and the pull goes ahead: if the - // estimate was right, Docker reports ENOSPC in its own words. + // The disk measurement is of the bootloader's filesystem, which is + // only Docker's on a default install. A tight check is a warning, + // not a refusal: if too tight, Docker reports ENOSPC itself. docker := &fakeDocker{inspectSize: 1 << 62} // larger than any real disk sup := &fakeSupervisor{} u, _, _ := newTestUpdater(t, docker, sup) @@ -539,12 +536,7 @@ func TestHumanBytesReadsLikeAnErrorMessage(t *testing.T) { } func TestAnImageAlreadyPresentSurvivesAFailedPull(t *testing.T) { - // A pull can fail for a reason that does not matter: the image is already - // here. That covers an air-gapped device with a side-loaded image, a - // locally built one, and a registry that is merely unreachable. Refusing - // would make a version the device already holds uninstallable -- which is - // what happened on the SLM-RP4, where a locally tagged image produced - // "pull access denied" and failed an update that was entirely ready. + // A failed pull must be forgiven when the image is already present. docker := &fakeDocker{ inspectSize: 100, pullErr: errors.New("pull access denied for openplc-runtime"), @@ -606,17 +598,8 @@ func TestAFailureAfterTheSwapBeginsDoesEnterRecovery(t *testing.T) { } func TestARefusedUpdateLeavesTheSupervisorReportingReality(t *testing.T) { - // BeginUpdate moves the supervisor to "updating" and EndUpdate only - // releases the claim, so without restoring state a device that merely - // refused a bad version reports itself as mid-update forever. Seen on the - // SLM-RP4: a failed pull left the bootloader stuck on "updating" while the - // PLC ran happily underneath. - // - // It restores rather than reconciles. Reconcile re-derives the state by - // ACTING on the container: from recovery it would start the runtime that - // recovery deliberately stopped and call the device healthy, and with the - // container absent it would leave the state on "starting" for good after - // the pull failed again. + // A refused update must RESTORE the supervisor's prior state (not + // Reconcile), otherwise a recovery-state device gets started. docker := &fakeDocker{ inspectErr: errors.New("no such image"), pullErr: errors.New("manifest unknown"), diff --git a/bootloader/main.go b/bootloader/main.go index 4b6ab3a0..168c038f 100644 --- a/bootloader/main.go +++ b/bootloader/main.go @@ -1,27 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// Command openplc-bootloader brings up and maintains one local OpenPLC runtime -// container (RTOP-283). -// -// It plays the same role a bootloader plays on an embedded target, and the name -// is meant literally. A bootloader is the small, rarely-changed program that -// starts the real firmware, and that stays reachable to flash a new image when -// the firmware is broken or missing. This does exactly that for the runtime: it -// starts the runtime container, and when the runtime will not run it remains -// available so a new version can be installed from the editor. That is the -// whole reason it exists -- many vendors do not allow SSH, so without something -// that survives a bad runtime there is no way back onto the device. -// -// The analogy holds on the other axis too. A bootloader is kept deliberately -// dumb and stable because it is the one thing that cannot be recovered by any -// other means, so it does the minimum: it does not accept programs, control the -// PLC, or look at PLC state. It is always resident and, in steady state, does -// nothing at all -- after confirming the runtime came up it blocks on the -// Docker events stream, with no timers and no polling. -// -// Docker is the only dependency. Docker's own restart policy starts this -// process, so nothing of ours goes into systemd. +// Command openplc-bootloader brings up and maintains one OpenPLC runtime +// container, and stays reachable to install a new version. Minimal: +// blocks on the Docker events stream, no PLC control or state inspection. package main import ( @@ -51,10 +33,8 @@ import ( // it to every runtime release would produce a long series of identical images. var version = "dev" -// DefaultStateDir is the bootloader's own volume -- separate from the runtime's -// data directory on purpose. "Erase all data" wipes the runtime's volume, and -// the board's device mounts must survive that; a board that came back with no -// SPI after a data wipe would be a miserable failure mode. +// DefaultStateDir is the bootloader's own volume, deliberately separate +// from the runtime's data dir so "erase all data" leaves hardware mounts. const DefaultStateDir = "/var/lib/openplc-bootloader" func main() { @@ -79,14 +59,9 @@ func main() { log := newLogger(*logLevel) - // Self-update helper mode. - // - // A container cannot replace itself, so a bootloader being updated spawns - // a one-shot child from the NEW image and that child does the swap from - // outside. This is that child: it replaces its parent and exits, and it - // must never fall through into ordinary bootloader operation -- two - // bootloaders supervising one runtime is exactly the race this design - // exists to avoid. + // Self-update helper mode. A container cannot replace itself, so the + // NEW image spawns this one-shot child to swap from outside. Must + // never fall through into normal bootloader operation. if selfupdate.IsChild() { log.Info("running as a self-update helper", "version", version) ctx, stop := signal.NotifyContext(context.Background(), syscall.SIGINT, syscall.SIGTERM) @@ -149,22 +124,13 @@ func run(log *slog.Logger, cfg runConfig) error { CrashWindow: cfg.crashWindow, }, log.With("component", "supervisor")) - // Authentication reads the runtime's own credentials out of the shared data - // directory. Missing or unreadable is not fatal: the control API still - // needs to come up so an operator can see WHY, and every authenticated - // route refuses cleanly until the files appear. - // Re-read on use rather than snapshotted: on a fresh install the runtime - // creates .env and restapi.db only once the bootloader has already started - // it, and a snapshot from before that made every login fail until the - // container was restarted. + // Credentials re-read on use: the runtime may create .env and + // restapi.db only after the bootloader is already up. creds := runtimeauth.NewProvider(spec.DataDir, log.With("component", "auth")) defer creds.Close() - // LAN discovery, answered ONLY while in recovery. A device that cannot be - // found cannot be repaired, and without this a failed update makes the - // device vanish from the editor's list at exactly the wrong moment. The - // runtime owns this port the rest of the time; exclusivity holds because - // entering recovery stops the runtime first. + // LAN discovery, answered ONLY in recovery. Exclusive with the + // runtime's own responder because recovery stops the runtime first. responder := discovery.New(discovery.Port, func() discovery.Reply { status := sup.Status() return discovery.Reply{ @@ -245,22 +211,17 @@ func run(log *slog.Logger, cfg runConfig) error { return nil } -// bootloaderSelfUpdater adapts the selfupdate package to the API's interface. -// -// The repository is left empty so the package's default applies: a bootloader -// pulling its replacement from somewhere an API caller chose would be a way to -// run arbitrary images as host root. +// bootloaderSelfUpdater adapts selfupdate to the API interface. Leaves +// the repository empty so the package default applies (never from an +// API caller's input: that would run arbitrary images as host root). type bootloaderSelfUpdater struct { docker *dockerapi.Client log *slog.Logger } func (b bootloaderSelfUpdater) Start(ctx context.Context, version string) error { - // The repository is NOT taken from the API request: a bootloader pulling - // its replacement from wherever a caller named would be a way to run an - // arbitrary image as host root. The env override exists for the - // integration harness, which has no route to ghcr.io, and is set at - // install time rather than per request. + // Repository is NEVER taken from the API request. The env override + // exists for the integration harness and is set at install time. return selfupdate.Start(ctx, b.docker, os.Getenv("OPENPLC_BOOTLOADER_REPOSITORY"), version, b.log) } diff --git a/core/src/drivers/plugin_config.c b/core/src/drivers/plugin_config.c index b0401482..5fe73ddd 100644 --- a/core/src/drivers/plugin_config.c +++ b/core/src/drivers/plugin_config.c @@ -25,21 +25,14 @@ static void remove_newline(char *str) } /** - * Reject a plugin path that could point outside the runtime tree. + * @brief Reject a plugin path that could point outside the runtime tree. * - * The `path` field of a plugin config is handed straight to dlopen (VPP) or - * used as a Python module location, so a config the runtime did not write is - * arbitrary-code selection. vpp_plugins.conf IS such a config: it arrives - * verbatim inside the user's upload. The Python side contains it too - * (webserver/plcapp_management.py validate_vpp_plugins_conf); this check is - * here so containment does not depend on one language alone. + * `path` is handed to dlopen (VPP) or used as a Python module location, so + * an upload-supplied config is arbitrary-code selection. * - * @param require_contained 0 to only reject `..` traversal, 1 to also reject - * absolute paths. Config files the runtime itself owns (plugins.conf) - * pass 0: an operator with a hand-written absolute path there is not - * the threat, and refusing it would break working installations. Only - * the upload-supplied config is parsed with 1. - * @return 1 when the path is acceptable, 0 when it must be rejected. + * @param require_contained 0 rejects only `..` traversal; 1 also rejects + * absolute paths (used for upload-supplied configs). + * @return 1 if acceptable, 0 if rejected. */ static int plugin_path_is_acceptable(const char *path, int require_contained) { diff --git a/core/src/drivers/plugin_config.h b/core/src/drivers/plugin_config.h index 267c6505..343d1623 100644 --- a/core/src/drivers/plugin_config.h +++ b/core/src/drivers/plugin_config.h @@ -27,13 +27,9 @@ typedef struct int parse_plugin_config(const char *config_file, plugin_config_t *configs, int max_configs); /** - * Parse a plugin config that came from an upload (vpp_plugins.conf). - * - * As above, plus absolute (and Windows drive-prefixed) paths are rejected: the - * `path` field of this file is chosen by whoever produced the upload and is fed - * to dlopen, so it must stay inside the runtime tree. Mirrors the Python-side - * containment in webserver/plcapp_management.py so neither side is the only - * thing standing between an upload and dlopen. + * @brief Parse a plugin config from an upload (vpp_plugins.conf). Rejects + * `..` traversal AND absolute/Windows-drive-prefixed paths so the + * dlopen target stays inside the runtime tree. */ int parse_plugin_config_contained(const char *config_file, plugin_config_t *configs, int max_configs); diff --git a/core/src/drivers/plugin_driver.c b/core/src/drivers/plugin_driver.c index d55d04c0..e384ee2a 100644 --- a/core/src/drivers/plugin_driver.c +++ b/core/src/drivers/plugin_driver.c @@ -100,11 +100,9 @@ static int plugin_journal_write_lint(int type, int index, unsigned long long val return journal_write_lint((journal_buffer_type_t)type, (uint16_t)index, (uint64_t)value); } -// STruC++ debugger thunks. Forward to ext_strucpp_debug_* function -// pointers resolved from the program .so by image_tables symbols_init. -// All five tolerate ext_*==NULL (program not yet loaded) and return a -// safe sentinel: counts → 0, debug_set/debug_write → STATUS_OUT_OF_BOUNDS -// (0x81), debug_read → 0 bytes written. +// STruC++ debugger thunks forwarding to ext_strucpp_debug_* loaded from +// the program .so. All tolerate ext_*==NULL (no program) and return a +// safe sentinel. static uint8_t plugin_debug_array_count(void) { @@ -126,14 +124,9 @@ static uint16_t plugin_debug_read(uint8_t arr, uint16_t elem, uint8_t *dest) return ext_strucpp_debug_read ? ext_strucpp_debug_read(arr, elem, dest) : 0; } -// debug_set / debug_write are called from the PLUGIN's own thread (the OPC-UA -// asyncio thread, the BACnet poll thread, ...). Poking the IECVar there races -// the IEC task workers (OpenPLC bug #3, and the mechanism behind the OPC-UA -// global-variable corruption). Both now ENQUEUE through the debug-write -// journal; the dispatcher applies them at the no-task-running window — -// race-free, and (for located vars) through the image journal + forced-slot -// bitmap. Return 0x7E (SUCCESS) once queued, 0x82 (OUT_OF_MEMORY) if the queue -// is momentarily full, 0x81 (OUT_OF_BOUNDS) when no program is loaded. +// debug_set/debug_write run on PLUGIN threads. ENQUEUE through the +// debug-write journal; dispatcher applies at the no-task-running window. +// Returns 0x7E ok, 0x82 queue full, 0x81 no program. static uint8_t plugin_debug_set(uint8_t arr, uint16_t elem, bool forcing, const uint8_t *bytes, uint16_t len) @@ -154,39 +147,24 @@ static uint8_t plugin_debug_write(uint8_t arr, uint16_t elem, return (rc == 0) ? 0x7E : 0x82; } -// Plugin-invoked async PLC stop. Logs the reason at error level and kicks -// off a detached state-transition worker via the same path the unix-socket -// STOP command uses — the transition flag blocks overlapping commands, and -// all plugins get their stop_loop / cleanup hooks called in the normal -// order. Non-blocking by design: the caller's I/O thread returns -// immediately, then continues running for the brief window until the -// plugin's own stop_loop is invoked. Plugins that enter fault-stopped state -// are expected to short-circuit their I/O during that window. -// -// No pre-check on plc_get_state() here: plc_claim_transition does the check -// atomically under the state lock (the state being TRANSITIONING is itself what -// prevents concurrent transitions), so doing it again outside would just -// re-introduce the check-then-act race that interlock exists to close. +// Plugin-invoked async PLC stop. Routes through the same transition +// path as the socket STOP; non-blocking. The claim is atomic under the +// state lock, so no pre-check here. static void plugin_request_plc_stop(const char *reason) { log_info("[PLUGIN] stop requested: %s", reason ? reason : "(no reason given)"); if (!plc_begin_transition(PLC_STATE_STOPPED)) { - // Either the PLC is already stopping/stopped or another stop - // is already in flight — either way, nothing to do. - // - // A stop dropped because another transition was in flight is recovered - // by the switch-movement reconciliation once that transition lands (see + // Already stopping/stopped, or another transition is in flight; + // the switch-movement reconciliation picks it up later (see // transition_worker in unix_socket.c), so a switch-driven stop cannot be // lost here. log_warn("[PLUGIN] stop request collapsed (already transitioning or not running)"); } } -// Mirror of plugin_request_plc_stop, routed through the same transition path -// the socket START command uses. Gated on the mode switch: hardware is -// authoritative no matter who asks, so a plugin cannot start a PLC whose switch -// reads STOP (which also keeps a buggy plugin from defeating the interlock). +// Plugin-invoked PLC start. Gated on the hardware mode switch: a plugin +// cannot start a PLC whose switch reads STOP. static void plugin_request_plc_start(const char *reason) { if (!plc_switch_allows_run()) @@ -210,11 +188,8 @@ static void plugin_set_switch_position(int position) plc_set_switch_position(position == PLC_SWITCH_STOP ? PLC_SWITCH_STOP : PLC_SWITCH_RUN); } -// Map the runtime's PLCState onto the values FC 0x49 reports on baremetal -// targets (0 = STOPPED, 1 = RUNNING, 2 = ERROR) so vendor code driving a -// status LED can share one mapping across both target types. INIT and EMPTY -// are v4-only and have no physical meaning for an indicator, so they report as -// STOPPED — the PLC is not executing. +// Map PLCState to FC 0x49 LED values (0 STOPPED, 1 RUNNING, 2 ERROR). +// INIT/EMPTY report as STOPPED (not executing). static int plugin_get_plc_state(void) { switch (plc_get_state()) @@ -228,7 +203,6 @@ static int plugin_get_plc_state(void) } } - // Python capsule destructor for runtime args // Breakpoint here to debug capsule issues static void plugin_runtime_args_capsule_destructor(PyObject *capsule) @@ -262,19 +236,10 @@ static PyObject *create_python_runtime_args_capsule(plugin_runtime_args_t *args) return capsule; } -/* Tear down a single plugin instance: cleanup hook (if init() ran), - * close native handles, release Python refs. Leaves the slot zeroed. - * - * IMPORTANT: dispatched on the slot's CURRENT stored type, not on the - * incoming config's type — that's the bug fix for the slot-positional - * reload issue. Closing by slot index assumed configs[w].type matched - * driver->plugins[w].config.type, which falls apart whenever the user - * reorders / replaces / changes the type of an entry in plugins.conf. - * - * Caller must ensure the plugin is not running (this function is called - * from update_config which only runs after STOP). The dlclose is unsafe - * on a live plugin — once the .so is unmapped, any in-flight call into - * its function pointers segfaults. */ +/* Tear down one plugin instance (cleanup hook, dlclose, free). Dispatch + * on the slot's CURRENT stored type, not the incoming config's, so a + * reordered or type-swapped plugins.conf does not leak handles. Caller + * must ensure the plugin is not running. */ static void teardown_plugin_instance(plugin_instance_t *plugin) { if (!plugin) return; @@ -288,11 +253,8 @@ static void teardown_plugin_instance(plugin_instance_t *plugin) if (plugin->config.type == PLUGIN_TYPE_PYTHON && plugin->python_plugin) { - // python_plugin_cleanup invokes the optional cleanup() if the - // plugin had been initialised, then DECREFs all module refs and - // frees the python_plugin bundle (sets the field to NULL). - // Calling it on an uninitialised plugin still releases module - // refs cleanly, so always call it as long as python_plugin is set. + // Also safe on an uninitialised plugin: it still releases + // module refs and NULLs python_plugin. python_plugin_cleanup(plugin); } else if (plugin->config.type == PLUGIN_TYPE_NATIVE && plugin->native_plugin) @@ -365,35 +327,10 @@ int plugin_driver_update_config(plugin_driver_t *driver, const char *config_file return -1; } - /* Tear down ALL old plugins, dispatched by their CURRENT stored type - * (not the new config's type). This fixes the slot-positional reload - * bug: previously a slot whose type changed Native→Python would skip - * the dlclose (because the new type was Python) and leak the old .so - * handle; the converse direction would force-free a Python instance's - * native_plugin (which is NULL) but leave python_plugin orphaned. - * - * After this loop every slot 0..old plugin_count-1 is zeroed; we can - * safely rebuild from configs[] without worrying about stale state. - * This function is only called from load_plc_program (post-STOP) and - * plc_main.c boot, both of which guarantee no plugin is running, so - * dlclose is safe. - * - * GIL: this function does Python work in two places — - * 1) teardown_plugin_instance → python_plugin_cleanup (Py_XDECREF, - * PyObject_CallFunctionObjArgs) for any old Python slot; - * 2) python_plugin_get_symbols (PyImport_ImportModule, etc.) for - * each new Python entry in the rebuild loop. - * Both require the GIL. plc_main releases the GIL after the initial - * plugin init, so the second update_config call (from - * load_plc_program) lands here without it. - * - * The teardown loop is the only stage that strictly needs an explicit - * ensure — if Python is initialized and we have Python plugins to - * tear down, we MUST hold the GIL or Py_XDECREF will SIGSEGV. The - * rebuild loop's python_plugin_get_symbols handles the cold-start - * case itself (it calls Py_Initialize if needed and is implicitly - * GIL-holding after that), so for that loop we just need to make - * sure we don't release the GIL we acquired here. */ + /* Tear down ALL old plugins by their CURRENT stored type (not the + * new config's type), so a slot whose type changed does not leak + * the old handle. Requires the GIL for any Python teardown; the + * rebuild loop's get_symbols handles cold-start itself. */ PyGILState_STATE plugin_gstate = PyGILState_LOCKED; int plugin_have_gil = Py_IsInitialized(); if (plugin_have_gil) @@ -407,10 +344,8 @@ int plugin_driver_update_config(plugin_driver_t *driver, const char *config_file teardown_plugin_instance(&driver->plugins[w]); } - /* Reset has_python_plugin before rebuilding — it'll be set again below - * for any Python entries in the new config. Without resetting, removing - * the last Python plugin from plugins.conf would leave the flag at 1 - * and cause unnecessary GIL acquires throughout the driver. */ + /* Reset so a plugins.conf that drops its last Python entry does + * not keep the flag at 1 and acquire the GIL for no reason. */ has_python_plugin = 0; int degraded_count = 0; @@ -426,19 +361,9 @@ int plugin_driver_update_config(plugin_driver_t *driver, const char *config_file if (configs[w].type == PLUGIN_TYPE_PYTHON) { has_python_plugin = 1; - /* Re-import Python module symbols here. The teardown loop - * above ran python_plugin_cleanup, which zeros python_plugin. - * Without re-importing, plugin_driver_init's Python branch - * (which requires plugin->python_plugin && pFuncInit) would - * silently skip every Python plugin on the second invocation - * of update_config — the modbus_slave / modbus_master / opcua - * plugins would never re-init after a PLC restart. - * - * python_plugin_get_symbols handles cold-start itself: if - * Python isn't initialized yet, it calls Py_Initialize which - * implicitly puts the current thread in possession of the GIL, - * so subsequent Python plugins in this loop also run safely - * without an explicit ensure. */ + /* Re-import symbols: teardown zeroed python_plugin, so init + * would skip without this. get_symbols cold-starts Python + * if needed. */ if (plugin->config.path[0] != '\0') { if (python_plugin_get_symbols(plugin) != 0) @@ -470,17 +395,10 @@ int plugin_driver_update_config(plugin_driver_t *driver, const char *config_file { if (plugin->config.enabled) { - /* Fail-safe: an enabled native plugin that cannot load - * its .so (e.g. a missing runtime dependency such as - * Npcap for the EtherCAT plugin on Windows) is marked - * degraded and skipped, but does NOT abort the whole - * runtime. native_plugin stays NULL, so init/start/cycle - * skip it; commands routed to it return a clear - * "unavailable" response. This keeps the runtime out of - * ERROR so the PLC can still reach RUNNING. The loud - * error above (plus any plugin-specific hint, e.g. the - * Npcap notice in native_plugin_get_symbols) tells the - * user what to fix. */ + /* Fail-safe: an enabled plugin that cannot load its + * .so (missing runtime dep, e.g. Npcap on Windows) + * is marked degraded and skipped, so the PLC can + * still reach RUNNING. */ log_error("[PLUGIN] enabled native plugin '%s' failed to load symbols " "- continuing without it (plugin unavailable)", configs[w].name); @@ -594,12 +512,9 @@ int plugin_driver_load_config(plugin_driver_t *driver, const char *config_file) return -1; } - /* plugin_driver_update_config now performs the full teardown + rebuild, - * including symbol loading for both Python and native plugins. The - * previous post-update_config loop here would re-call the *get_symbols - * functions on already-loaded slots — those allocate fresh bundles and - * overwrite the pointer, leaking the bundle that update_config just - * created. Just forward the return code. */ + /* update_config does the full teardown + rebuild including symbol + * loading, so just forward. Re-calling get_symbols here would leak + * the freshly-allocated bundles. */ return plugin_driver_update_config(driver, config_file); } @@ -620,11 +535,9 @@ int plugin_driver_init(plugin_driver_t *driver) local_gstate = PyGILState_Ensure(); } - // Initialize ALL plugins regardless of enabled flag. - // This allows features like EtherCAT slave scanning from the editor - // even when the plugin is not enabled for PLC runtime cycling. - // The init() contract: set up internal state, parse config, allocate - // resources. Do NOT start servers, threads, or read buffer values. + // Initialize ALL plugins regardless of enabled flag (needed for + // features like EtherCAT slave scanning from the editor). + // init() must only set up internal state; starting is in start_loop. for (int i = 0; i < driver->plugin_count; i++) { plugin_instance_t *plugin = &driver->plugins[i]; @@ -774,15 +687,9 @@ int plugin_driver_start(plugin_driver_t *driver) return 0; } - // Only manage Python GIL if we have Python plugins and Python is initialized. - // - // Acquire-then-save leaves this thread without the GIL, which is the point: - // the plugin threads started below need it. The saved state is deliberately - // NOT stored in main_tstate -- this runs on the PLC cycle thread, and - // main_tstate is what plugin_driver_destroy restores before Py_FinalizeEx(), - // which must be the MAIN thread's state. Overwriting it here meant a shutdown - // after a start restored a state belonging to a thread that no longer exists. - // plugin_driver_release_gil() owns that value. + // Release the GIL for plugin threads. Do NOT store the saved state + // in main_tstate (that is owned by plugin_driver_release_gil on the + // MAIN thread, and must survive Py_FinalizeEx). if (has_python_plugin && Py_IsInitialized()) { gstate = PyGILState_Ensure(); @@ -1001,12 +908,9 @@ void plugin_driver_destroy(plugin_driver_t *driver) if (python_initialized) { - /* Py_FinalizeEx() requires the GIL, and getting there with it released is - * what used to segfault the runtime on every graceful shutdown where the - * PLC had never run (Py_FinalizeEx -> PyImport_GetModule with no thread - * state). main_tstate is only non-NULL once the GIL has been saved by the - * main thread, so when it is NULL the right move is to KEEP the state - * PyGILState_Ensure() gave us above rather than dropping it. */ + /* Py_FinalizeEx needs the GIL. If main_tstate is NULL (the PLC + * never ran), keep the state PyGILState_Ensure gave us above + * instead of dropping it. */ if (main_tstate != NULL) { PyGILState_Release(local_gstate); @@ -1126,12 +1030,6 @@ void *generate_structured_args_with_driver(plugin_type_t type, plugin_driver_t * // guard against the value being smaller than their needed resolution. args->base_tick_ns = base_tick_ns; - // printf("[PLUGIN]: Runtime args initialized:\n"); - // printf("[PLUGIN]: buffer_size = %d\n", args->buffer_size); - // printf("[PLUGIN]: bits_per_buffer = %d\n", args->bits_per_buffer); - // printf("[PLUGIN]: bool_input = %p\n", (void *)args->bool_input); - // printf("[PLUGIN]: image_lock = %p\n", (void *)args->image_lock); - // Validate critical pointers if (!args->image_lock || !args->image_unlock) { @@ -1348,15 +1246,9 @@ int native_plugin_get_symbols(plugin_instance_t *plugin) return -1; } - /* Last metre before execution: a VPP plugin .so must match the hash - * scripts/compile.sh sealed when it built that .so on this device. The - * check binds what the loader executes to what this runtime's compile - * step produced, so an object dropped into build/vpp/ after the compile - * is refused. - * - * Built-in plugins from plugins.conf are produced by the runtime's own - * CMake build and are not sealed -- vpp_plugin_seal_required() scopes the - * check to objects that resolve inside build/vpp/. */ + /* A VPP .so must match the hash compile.sh sealed when it built + * that .so on this device, so a post-compile swap is refused. + * Scoped to build/vpp/; built-in plugins are not sealed. */ if (vpp_plugin_seal_required(plugin->config.path) && vpp_plugin_seal_verify(plugin->config.path) != 0) { @@ -1448,10 +1340,9 @@ int native_plugin_get_symbols(plugin_instance_t *plugin) // get_stats is fully optional — plugins that don't publish statistics // simply don't export it. No warning. - // Retain store (NODE-94), fully optional. A plugin exporting both becomes - // a candidate for this device's retain store; see - // plugin_driver_find_retain_store. No warning when absent — most plugins - // have nothing to do with retention. + // Optional retain store. A plugin exporting both save and load + // becomes a candidate; see plugin_driver_find_retain_store. No + // warning when absent — most plugins have no retention role. native_bundle->retain_save = (plugin_retain_save_func_t)dlsym(handle, "retain_save"); native_bundle->retain_load = (plugin_retain_load_func_t)dlsym(handle, "retain_load"); native_bundle->retain_flush = (plugin_retain_flush_func_t)dlsym(handle, "retain_flush"); @@ -1480,32 +1371,14 @@ void python_plugin_cycle(plugin_instance_t *plugin) // and call the cycle function } -// Call cycle_start for all active native plugins that have registered the hook -// This should be called at the beginning of each PLC scan cycle, before PLC logic execution -// Plugins opt-in by implementing cycle_start(); opt-out by not implementing it (NULL pointer) -// --------------------------------------------------------------------------- // Retain store -// --------------------------------------------------------------------------- - static bool plugin_provides_retain_store(const plugin_instance_t *p) { if (!p) return false; - - // A DISABLED plugin is not a store, even though its symbols resolved. - // - // Loading resolves symbols for every plugin in plugins.conf; only starting - // is gated on `enabled`. Without this check a disabled plugin is still - // picked as the store, so retain reports itself active, hands it the blob - // every scan, and gets nothing back on the next boot — the values are - // simply gone, with a log line at start saying retain is configured and - // working. Found on hardware: an upload rewrote plugins.conf, disabled the - // storage plugin, and retain went on claiming to work. + // Disabled plugins never serve as a store: they resolve symbols but + // do not run, so they would accept save() and lose everything. if (!p->config.enabled) return false; - - // BOTH halves required. A store that can save and not load is worse than - // none: it would accept values every scan and silently never give them - // back, which looks like working retention right up until the reboot that - // matters. + // Save without load is worse than none; require both halves. return p->native_plugin && p->native_plugin->retain_save && p->native_plugin->retain_load; } @@ -1655,20 +1528,6 @@ int plugin_driver_execute_command(plugin_driver_t *driver, const char *plugin_na return -1; } -// =================================================================== -// Plugin-contributed statistics aggregation -// =================================================================== -// -// Called from the STATS response path. Takes an already-formatted JSON -// response ending in "}\n" (or "}"), asks each native plugin that -// exports get_stats to produce a JSON object snippet, and splices them -// into a "plugin_stats" member before the closing brace. -// -// Per-plugin budget: PLUGIN_STATS_SLOT_BUDGET bytes. -// Combined budget: PLUGIN_STATS_TOTAL_BUDGET bytes. -// Output is best-effort: malformed plugin output (doesn't start with -// '{' and end with '}') is silently dropped, overflow truncates, and -// the core STATS response is always preserved. #define PLUGIN_STATS_SLOT_BUDGET 1024 #define PLUGIN_STATS_TOTAL_BUDGET 8192 diff --git a/core/src/drivers/plugin_driver.h b/core/src/drivers/plugin_driver.h index 0ee41bd9..e4bf1ff8 100644 --- a/core/src/drivers/plugin_driver.h +++ b/core/src/drivers/plugin_driver.h @@ -36,48 +36,12 @@ typedef int (*plugin_execute_command_func_t)(const char *command_json, char *res // Return 0 on success; any other value means "skip me this cycle." typedef int (*plugin_get_stats_func_t)(char *out, size_t out_size); -/* ---- Optional: retain-variable storage (NODE-94) ------------------------- - * - * A plugin that owns retention hardware exports these and becomes the device's - * retain store, displacing the runtime's built-in file store. Same names, same - * status meaning and the same contract text as baremetal's `openplc_retain.h`, - * and the two runtimes call them at the same points in the PLC lifecycle, so a - * vendor writes one shape twice rather than learning two interfaces for one job. - * - * start retain_load() once, before the first scan - * scan retain_save() every cycle, WHILE RUNNING ONLY - * stop retain_flush() once, as the program is unloaded - * - * The runtime MARSHALS and the plugin STORES: what arrives is an opaque blob, - * already validated on the way back in (magic, format, layout hash, crc32), so - * a backend needs no understanding of retained variables at all. - * - * SAVE IS THE DURABILITY PATH; FLUSH IS ONLY A HINT. Retention exists for the - * power cut nobody schedules, and a power cut does not call flush(). A plugin - * that commits solely in flush() therefore loses everything in exactly the case - * it was written for. Decide durability in save(). - * - * `retain_save` is called ONCE PER SCAN CYCLE, unconditionally, from the - * dispatcher's quiescent window, for as long as the PLC is running. The runtime - * does not diff and does not rate-limit — holding the bytes and committing on a - * schedule the medium can sustain is the plugin's job, and the reason the call - * exists at that cadence is so a plugin that CAN write every cycle (FRAM, - * battery-backed SRAM) is free to. It MUST return promptly and MUST NOT block: - * this runs inside the scan, so time spent here is time the PLC is not scanning. - * - * `retain_load` is handed the running program's identity — `md5_len` characters - * of lower-case hex, NOT guaranteed NUL-terminated, so compare with memcmp. - * THE PLUGIN DECIDES whether the bytes it holds still belong to this program: - * identity differs → discard them, log one line saying storage was cleared, and - * report empty (`*out_len = 0`), so every retained variable starts at its - * declared initial value. Do NOT persist the new identity here; hold it and - * commit it alongside the blob on the next `retain_save`, so a load never - * mutates storage and identity and bytes are written as one unit. - * - * Both save and load must be exported for the plugin to be used as the store; a - * plugin exporting only one is ignored, since a store that can save and not - * load is worse than none. Return 0 on success, non-zero otherwise. - */ +/* Optional retain storage. load() once pre-first-scan, save() every + * cycle while RUNNING (inside the scan, MUST NOT block), flush() at + * stop. load() memcmps program_md5 (not NUL-terminated); save AND + * load are both required. Protocol: on identity mismatch load reports + * *out_len=0; load MUST NOT persist the new identity — it is committed + * alongside the blob by the next save. */ typedef int (*plugin_retain_save_func_t)(const uint8_t *blob, uint16_t len); typedef int (*plugin_retain_load_func_t)(const char *program_md5, uint16_t md5_len, uint8_t *out, uint16_t cap, uint16_t *out_len); @@ -110,19 +74,13 @@ typedef struct plugin_instance_s plugin_funct_bundle_t *native_plugin; // pthread_t thread; int running; - /* Set after a successful init() call; cleared by cleanup. Tracked - * separately from `running` so a partial init failure (e.g., - * pthread_create on the cycle thread fails AFTER plugin_driver_init - * succeeded) can roll back only the plugins that actually got - * initialised, not those still untouched. */ + /* Set by init(); cleared by cleanup. Tracked apart from `running` + * so a partial init failure can roll back only the plugins that + * actually completed init. */ int initialized; - /* Set when an *enabled* plugin failed to load its symbols (e.g. a - * native .so whose runtime dependency is missing, such as the - * EtherCAT plugin without Npcap on Windows). A degraded plugin keeps - * its config slot but has a NULL native_plugin/python_plugin, so it - * is skipped by init/start/cycle. The runtime stays out of ERROR; - * commands routed to it return a clear "unavailable" response instead - * of crashing the whole runtime on boot. */ + /* An enabled plugin whose symbols failed to load (e.g. missing + * runtime dep). Slot kept but its *_plugin pointers are NULL; + * init/start/cycle skip it. Keeps the runtime out of ERROR. */ int degraded; plugin_config_t config; } plugin_instance_t; @@ -138,35 +96,22 @@ typedef struct plugin_driver_t *plugin_driver_create(void); int plugin_driver_load_config(plugin_driver_t *driver, const char *config_file); int plugin_driver_update_config(plugin_driver_t *driver, const char *config_file); -/** Append plugins from a secondary config file without tearing down the - * plugins already loaded by plugin_driver_update_config. Used to load - * VPP plugins from vpp_plugins.conf after built-ins from plugins.conf. - * Returns 0 on success, -1 if any enabled plugin fails to load its .so. */ +/** Append plugins from a secondary conf without tearing down already- + * loaded plugins. Returns 0 on success, -1 if any enabled plugin + * fails to load its .so. */ int plugin_driver_append_config(plugin_driver_t *driver, const char *config_file); int plugin_driver_init(plugin_driver_t *driver); -/* Mirror of plugin_driver_init: walks plugins[] in reverse order and calls - * the matching cleanup hook on every plugin whose `initialized` flag is - * set, then clears the flag. Used to roll back a partial init when a - * later step (e.g., spawning the cycle thread) fails — without this, a - * subsequent INIT cycle re-runs plugin init() on top of half-allocated - * state and duplicates threads/sockets. Safe to call when no plugins are - * initialised. Returns the count of plugins it cleaned up. */ +/* Roll back a partial init: walks plugins[] in reverse and calls + * cleanup on every plugin whose `initialized` flag is set. Returns + * the count cleaned up. */ int plugin_driver_cleanup_init(plugin_driver_t *driver); int plugin_driver_start(plugin_driver_t *driver); int plugin_driver_stop(plugin_driver_t *driver); void plugin_driver_destroy(plugin_driver_t *driver); -/* Release the Python GIL held by the calling thread after plugin loading, and - * remember the thread state so plugin_driver_destroy() can restore it before - * Py_FinalizeEx(). - * - * Call from the MAIN thread, once, after plugins are loaded. The runtime used to - * call PyEval_SaveThread() directly and drop the returned state on the floor, - * which left the driver with nothing to restore: shutdown then finalised the - * interpreter with no GIL held and segfaulted -- on every graceful shutdown of a - * runtime whose PLC had never started, safe mode included. Keeping the - * bookkeeping next to the code that consumes it is what makes that - * unrepresentable. No-op when Python was never initialised. */ +/* Release the Python GIL and remember the thread state so + * plugin_driver_destroy can restore it before Py_FinalizeEx. Call once + * from the MAIN thread after plugins are loaded. */ void plugin_driver_release_gil(void); // Cycle hook functions for native plugins (called during PLC scan cycle) @@ -177,12 +122,8 @@ void plugin_driver_cycle_end(plugin_driver_t *driver); /* ---- Retain store ------------------------------------------------------- */ -/* The plugin acting as this device's retain store, or NULL. - * - * The FIRST plugin that exports both retain_save and retain_load wins, and any - * others are logged and ignored. Two plugins writing the same retained values - * to different places would both appear to work and disagree on the next boot, - * which is a far worse failure than refusing the second. */ +/* The retain-store plugin (or NULL). First plugin exporting BOTH + * retain_save and retain_load wins; later ones are logged and ignored. */ plugin_instance_t *plugin_driver_find_retain_store(plugin_driver_t *driver); int plugin_driver_retain_save(plugin_instance_t *store, const uint8_t *blob, uint16_t len); @@ -194,11 +135,8 @@ int plugin_driver_retain_flush(plugin_instance_t *store); int plugin_driver_execute_command(plugin_driver_t *driver, const char *plugin_name, const char *command_json, char *response, size_t response_size); -// Splice plugin-contributed statistics into an already-formatted STATS -// response. Walks loaded native plugins, calls each get_stats, and -// inserts a "plugin_stats":{...} member before the closing `}` of the -// existing STATS JSON. A trailing newline in `buffer` is preserved. -// No-op if no plugin provides stats. Returns the new string length. +// Splice plugin get_stats JSON into an existing STATS response before +// the closing `}`. Trailing newline preserved. Returns new length. size_t plugin_driver_append_stats_json(plugin_driver_t *driver, char *buffer, size_t buffer_size); diff --git a/core/src/drivers/plugin_types.h b/core/src/drivers/plugin_types.h index c737ad13..87824893 100644 --- a/core/src/drivers/plugin_types.h +++ b/core/src/drivers/plugin_types.h @@ -207,20 +207,15 @@ typedef struct IEC_ULINT **lint_memory; IEC_BOOL *(*bool_memory)[8]; - /* Flush-on-lock image read API for thread-safe buffer access. - * - * image_lock() takes the runtime's image mutex and drains the journal so - * the holder sees every committed write; image_unlock() releases it. Writes - * never take this lock -- use journal_write_* (lock-free). Prefer the bulk - * pattern for reads: lock, copy the region to a local buffer, unlock, then - * do any slow work (network, conversion) on the buffer OUTSIDE the lock. */ + /* Flush-on-lock image read API. image_lock takes the mutex and + * drains the journal; image_unlock releases. Writes use + * journal_write_* (lock-free). Reads: lock, memcpy, unlock, + * then slow work OUTSIDE the lock. */ void (*image_lock)(void); void (*image_unlock)(void); - /* STruC++ debugger variable-access surface. - * Replaces the MatIEC-era flat-index API (get_var_list / - * get_var_size / get_var_count). Plugins like OPC-UA receive - * pre-resolved (arr, elem) tuples from the editor in their + /* STruC++ debugger variable-access surface. Plugins (e.g. OPC-UA) + * receive pre-resolved (arr, elem) tuples from the editor in their * per-plugin config and forward them through these thunks. */ plugin_debug_array_count_func_t debug_array_count; plugin_debug_elem_count_func_t debug_elem_count; @@ -249,6 +244,7 @@ typedef struct plugin_journal_write_dint_func_t journal_write_dint; plugin_journal_write_lint_func_t journal_write_lint; + /* Append-only below: compiled plugins bake in the offsets above. */ /* Async request to stop the whole PLC — see plugin_request_plc_stop_func_t. */ plugin_request_plc_stop_func_t request_plc_stop; @@ -257,14 +253,6 @@ typedef struct * symbols are not yet resolved (plugin must guard against zero). */ unsigned long long base_tick_ns; - /* --------------------------------------------------------------------- - * Run/stop control. Appended at the end of the struct so plugin binaries - * compiled against an earlier layout keep their field offsets. - * - * A plugin that ignores all three behaves exactly as before: the switch - * position stays at its RUN default, so every start path is unguarded. - * ------------------------------------------------------------------- */ - /* Async request to run — see plugin_request_plc_start_func_t. */ plugin_request_plc_start_func_t request_plc_start; diff --git a/core/src/drivers/plugins/native/s7comm/s7comm_plugin.cpp b/core/src/drivers/plugins/native/s7comm/s7comm_plugin.cpp index c402ff8d..0349e5d7 100644 --- a/core/src/drivers/plugins/native/s7comm/s7comm_plugin.cpp +++ b/core/src/drivers/plugins/native/s7comm/s7comm_plugin.cpp @@ -618,15 +618,6 @@ static void s7comm_event_callback(void *usrPtr, PSrvEvent PEvent, int Size) } } -/* - * ============================================================================= - * On-Demand Data Synchronization (via RWArea Callback) - * - * S7 client READs: Acquire OpenPLC mutex, copy to S7 buffer, release mutex - * S7 client WRITEs: Use journal writes (thread-safe, no mutex needed) - * ============================================================================= - */ - /** * @brief Map s7comm buffer type to journal buffer type * @@ -720,14 +711,6 @@ static int get_type_size(s7comm_buffer_type_t type) } } -/* - * ============================================================================= - * Read Functions: OpenPLC -> S7 Buffer (for S7 client READs) - * These functions copy data from OpenPLC image tables to the S7 buffer. - * Called with OpenPLC mutex held. - * ============================================================================= - */ - /** * @brief Read OpenPLC bool buffer to destination (mutex must be held) */ @@ -906,14 +889,6 @@ static void read_openplc_to_buffer(uint8_t *dest, int size, s7comm_buffer_type_t } } -/* - * ============================================================================= - * Write Functions: S7 Buffer -> OpenPLC via Journal (for S7 client WRITEs) - * These functions write data from S7 buffer to OpenPLC via journal. - * No mutex needed - journal writes are thread-safe. - * ============================================================================= - */ - /** * @brief Write bool buffer to OpenPLC via journal */ diff --git a/core/src/drivers/plugins/python/modbus_master/modbus_master_connection.py b/core/src/drivers/plugins/python/modbus_master/modbus_master_connection.py index ac21fb93..a79cd860 100644 --- a/core/src/drivers/plugins/python/modbus_master/modbus_master_connection.py +++ b/core/src/drivers/plugins/python/modbus_master/modbus_master_connection.py @@ -11,7 +11,6 @@ TransportType = Literal["tcp", "rtu"] ParityType = Literal["N", "E", "O"] - class ModbusConnectionManager: # pylint: disable=too-many-instance-attributes """Manages Modbus TCP and RTU connections with retry logic.""" diff --git a/core/src/drivers/plugins/python/modbus_master/modbus_master_memory.py b/core/src/drivers/plugins/python/modbus_master/modbus_master_memory.py index acfb49d4..7c8b2c3e 100644 --- a/core/src/drivers/plugins/python/modbus_master/modbus_master_memory.py +++ b/core/src/drivers/plugins/python/modbus_master/modbus_master_memory.py @@ -39,7 +39,6 @@ get_modbus_registers_count_for_iec_size, ) - def get_sba_access_details(iec_addr, is_write_op: bool = False) -> Optional[BufferAccessDetails]: """ Maps IECAddress to SafeBufferAccess method parameters. @@ -148,16 +147,6 @@ def get_sba_access_details(iec_addr, is_write_op: bool = False) -> Optional[Buff print(f"(FAIL) Error in get_sba_access_details: {e}") return None - -# ============================================================================= -# OPTIMIZED FUNCTIONS FOR MINIMAL MUTEX HOLD TIME -# ============================================================================= -# These functions separate data conversion from buffer access to minimize -# the time the mutex is held. Use these instead of the legacy functions -# when mutex hold time is critical. -# ============================================================================= - - def convert_modbus_data_to_iec_values( # pylint: disable=too-many-locals iec_addr, modbus_data: list, length: int ) -> Tuple[Optional[List], Optional[BufferAccessDetails]]: @@ -243,7 +232,6 @@ def convert_modbus_data_to_iec_values( # pylint: disable=too-many-locals print(f"(FAIL) Error in convert_modbus_data_to_iec_values: {e}") return None, None - def write_preconverted_iec_values( sba, converted_values: List[Tuple], details: BufferAccessDetails ) -> bool: @@ -327,7 +315,6 @@ def write_preconverted_iec_values( print(f"(FAIL) Error in write_preconverted_iec_values: {e}") return False - def read_raw_iec_values( # pylint: disable=too-many-locals sba, iec_addr, length: int ) -> Tuple[Optional[List], Optional[BufferAccessDetails], Optional[str]]: @@ -435,7 +422,6 @@ def read_raw_iec_values( # pylint: disable=too-many-locals print(f"(FAIL) Error in read_raw_iec_values: {e}") return None, None, None - def convert_raw_iec_to_modbus( raw_values: List, details: BufferAccessDetails, iec_size: str ) -> Optional[List]: @@ -486,15 +472,6 @@ def convert_raw_iec_to_modbus( print(f"(FAIL) Error in convert_raw_iec_to_modbus: {e}") return None - -# ============================================================================= -# LEGACY FUNCTIONS (kept for backward compatibility) -# ============================================================================= -# These functions perform conversion inside the mutex-protected section. -# For new code, prefer using the optimized functions above. -# ============================================================================= - - def update_iec_buffer_from_modbus_data( # pylint: disable=too-many-locals sba, iec_addr, modbus_data: list, length: int ): @@ -647,7 +624,6 @@ def update_iec_buffer_from_modbus_data( # pylint: disable=too-many-locals except Exception as e: print(f"(FAIL) Error updating IEC buffer: {e}") - def read_data_for_modbus_write( # pylint: disable=too-many-locals sba, iec_addr, length: int ) -> Optional[list]: diff --git a/core/src/drivers/plugins/python/modbus_master/modbus_master_plugin.py b/core/src/drivers/plugins/python/modbus_master/modbus_master_plugin.py index 460b550a..b8ddcbfa 100644 --- a/core/src/drivers/plugins/python/modbus_master/modbus_master_plugin.py +++ b/core/src/drivers/plugins/python/modbus_master/modbus_master_plugin.py @@ -71,7 +71,6 @@ slave_threads: List[threading.Thread] = [] # pylint: enable=invalid-name - def queue_zero_fill_on_failure(point: Any, read_results_to_update: List[Any]) -> bool: """ Queue a zeroed payload for a read point whose group asks for it. @@ -94,7 +93,6 @@ def queue_zero_fill_on_failure(point: Any, read_results_to_update: List[Any]) -> ) return True - class ModbusSlaveDevice(threading.Thread): """ Handles a single Modbus TCP device with its own connection. @@ -293,10 +291,9 @@ def run(self): # pylint: disable=too-many-locals f"for read updates: {lock_msg}" ) - # 2. WRITE OPERATIONS - Process only I/O points that are due for polling this cycle - # OPTIMIZATION: Batch all write preparations under a single mutex acquisition - # to minimize mutex hold time. Read all raw IEC values at once, then convert - # and write to Modbus outside the mutex. + # WRITES - process only points due this cycle. Batch all + # raw IEC reads under one image_lock; convert and send + # to Modbus outside the lock to minimise hold time. # Phase 1: Collect all write points that are due this cycle write_points_due = [] @@ -460,7 +457,6 @@ def stop(self): self.logger.info(f"[{self.name}] Stop signal received.") self._stop_event.set() - class ModbusBusHandler(threading.Thread): """ Handles multiple Modbus devices that share ONE connection, multiplexed by @@ -858,11 +854,9 @@ def stop(self): self.logger.info(f"[{self.name}] Stop signal received.") self._stop_event.set() - # Backward-compatible alias: the bus handler used to be RTU-only. ModbusRtuBusHandler = ModbusBusHandler - def group_rtu_devices_by_bus(devices: List[Any]) -> dict: """ Group RTU devices by serial port configuration (forming unique "buses"). @@ -911,7 +905,6 @@ def group_rtu_devices_by_bus(devices: List[Any]) -> dict: return buses - def group_tcp_devices_by_endpoint(devices: List[Any]) -> dict: """ Group TCP devices by (host, port) endpoint. @@ -957,7 +950,6 @@ def group_tcp_devices_by_endpoint(devices: List[Any]) -> dict: return endpoints - def init(args_capsule): """ Initialize the Modbus Master plugin. @@ -987,7 +979,6 @@ def init(args_capsule): traceback.print_exc() return False - def start_loop(): """ Start the main loop for all configured Modbus devices. @@ -1037,10 +1028,9 @@ def start_loop(): logger.info(f"Found {len(tcp_devices)} TCP device(s) and {len(rtu_devices)} RTU device(s)") - # Group TCP devices by (host, port). A lone device on an endpoint keeps - # the simple one-thread-per-device path; multiple devices on one endpoint - # (a Modbus gateway / TCP-to-RTU converter) share a single TCP connection - # and are multiplexed by slave/unit ID via ModbusBusHandler. + # Group TCP devices by (host, port). One device → one thread; + # multiple on one endpoint (a Modbus gateway) share a TCP + # connection, multiplexed by unit ID via ModbusBusHandler. if tcp_devices: tcp_endpoints = group_tcp_devices_by_endpoint(tcp_devices) logger.info(f"TCP devices grouped into {len(tcp_endpoints)} endpoint(s)") @@ -1115,7 +1105,6 @@ def start_loop(): traceback.print_exc() return False - def stop_loop(): """ Stop the main loop and all running device threads. @@ -1167,7 +1156,6 @@ def stop_loop(): traceback.print_exc() return False - def cleanup(): """ Clean up resources before plugin unload. @@ -1200,7 +1188,6 @@ def cleanup(): traceback.print_exc() return False - if __name__ == "__main__": # Test mode for development purposes. # This allows running the plugin standalone for testing. diff --git a/core/src/drivers/plugins/python/modbus_master/modbus_master_types.py b/core/src/drivers/plugins/python/modbus_master/modbus_master_types.py index dcd6b285..f9fb38f5 100644 --- a/core/src/drivers/plugins/python/modbus_master/modbus_master_types.py +++ b/core/src/drivers/plugins/python/modbus_master/modbus_master_types.py @@ -6,7 +6,6 @@ from dataclasses import dataclass from typing import Optional, Dict, Any, List - @dataclass class ModbusConnectionConfig: """Configuration for Modbus TCP connection.""" @@ -14,7 +13,6 @@ class ModbusConnectionConfig: port: int timeout_ms: int - @dataclass class ModbusIOPoint: """Represents a Modbus I/O point configuration.""" @@ -25,7 +23,6 @@ class ModbusIOPoint: iec_location: Any # IECAddress object cycle_time_ms: int - @dataclass class ModbusDeviceConfig: """Configuration for a Modbus slave device.""" @@ -35,7 +32,6 @@ class ModbusDeviceConfig: timeout_ms: int io_points: List[ModbusIOPoint] - @dataclass class BufferAccessDetails: """Details for SafeBufferAccess operations.""" diff --git a/core/src/drivers/plugins/python/modbus_master/modbus_master_utils.py b/core/src/drivers/plugins/python/modbus_master/modbus_master_utils.py index 10566d4a..2fd52947 100644 --- a/core/src/drivers/plugins/python/modbus_master/modbus_master_utils.py +++ b/core/src/drivers/plugins/python/modbus_master/modbus_master_utils.py @@ -6,7 +6,6 @@ import math from typing import List, Dict, Any - def gcd(a: int, b: int) -> int: """ Calculate the Greatest Common Divisor of two numbers using Euclidean algorithm. @@ -15,7 +14,6 @@ def gcd(a: int, b: int) -> int: a, b = b, a % b return a - def calculate_gcd_of_cycle_times(io_points: List[Any]) -> int: """ Calculate the GCD of all cycle_time_ms values from I/O points. @@ -36,7 +34,6 @@ def calculate_gcd_of_cycle_times(io_points: List[Any]) -> int: return result - def get_batch_read_requests_from_io_points(io_points: List[Any]) -> Dict[int, List[Any]]: """ Groups I/O points by Modbus read function code (1,2,3,4) and creates @@ -52,7 +49,6 @@ def get_batch_read_requests_from_io_points(io_points: List[Any]) -> Dict[int, Li read_requests[fc].append(point) return read_requests - def get_batch_write_requests_from_io_points(io_points: List[Any]) -> Dict[int, List[Any]]: """ Groups I/O points by Modbus write function code (5,6,15,16) and creates @@ -68,7 +64,6 @@ def get_batch_write_requests_from_io_points(io_points: List[Any]) -> Dict[int, L write_requests[fc].append(point) return write_requests - def get_modbus_registers_count_for_iec_size(iec_size: str) -> int: """ Returns how many 16-bit Modbus registers are needed for an IEC data type. @@ -92,7 +87,6 @@ def get_modbus_registers_count_for_iec_size(iec_size: str) -> int: else: return 1 # Default fallback - def convert_modbus_registers_to_iec_value(registers: List[int], iec_size: str, use_big_endian: bool = False): """ Converts Modbus register values to IEC data type value. @@ -130,7 +124,6 @@ def convert_modbus_registers_to_iec_value(registers: List[int], iec_size: str, u else: raise ValueError(f"Unsupported IEC size for register conversion: {iec_size}") - def convert_iec_value_to_modbus_registers(value: int, iec_size: str, use_big_endian: bool = False) -> List[int]: """ Converts IEC data type value to Modbus register values. @@ -174,7 +167,6 @@ def convert_iec_value_to_modbus_registers(value: int, iec_size: str, use_big_end else: raise ValueError(f"Unsupported IEC size for register conversion: {iec_size}") - def parse_modbus_offset(offset_str: str) -> int: """ Parse Modbus offset string supporting decimal and hexadecimal formats. @@ -206,7 +198,6 @@ def parse_modbus_offset(offset_str: str) -> int: return address - def get_read_count_for_io_point(point: Any) -> int: """ Returns how many Modbus registers/coils a read of `point` asks for. @@ -222,7 +213,6 @@ def get_read_count_for_io_point(point: Any) -> int: return point.length * registers_per_element return point.length - def get_zero_payload_for_io_point(point: Any) -> List[Any]: """ Returns a zeroed payload shaped like the response a successful read of diff --git a/core/src/drivers/plugins/python/modbus_slave/conftest.py b/core/src/drivers/plugins/python/modbus_slave/conftest.py index bf74de91..2800b69c 100644 --- a/core/src/drivers/plugins/python/modbus_slave/conftest.py +++ b/core/src/drivers/plugins/python/modbus_slave/conftest.py @@ -8,7 +8,6 @@ MAX_BITS = 8 # matches OpenPLC bit grouping MAX_REGS = 1 # word-aligned registers - class AdvancedObservingSBA: """ Fully functional SafeBufferAccess mock: @@ -118,8 +117,6 @@ def write_uint16_output(self, idx, value, thread_safe=True): read_int_input = read_uint16_input write_int_output = write_uint16_output - - # ====================================================================== # Fixtures # ====================================================================== @@ -138,7 +135,6 @@ def advanced_sba(runtime_args, monkeypatch): # <-- Added monkeypatch return sba - @pytest.fixture def runtime_args(): """Fake runtime args with a mock PLC buffer and proper lock.""" diff --git a/core/src/drivers/plugins/python/modbus_slave/simple_modbus.py b/core/src/drivers/plugins/python/modbus_slave/simple_modbus.py index 69680af5..2e5e8dd9 100644 --- a/core/src/drivers/plugins/python/modbus_slave/simple_modbus.py +++ b/core/src/drivers/plugins/python/modbus_slave/simple_modbus.py @@ -1,18 +1,9 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® +# pylint disables kept so pymodbus-mandated names, segmented data-block +# constructors and the shared-module import order do not warn here. # pylint: disable=C0103,C0301,C0302,C0413,W0107,W0602,W0621,C0415,R0913,R0914,R0917 -# C0103: Method/variable naming (getValues/setValues required by pymodbus API) -# C0301: Line too long (some lines exceed 100 chars) -# C0302: Too many lines in module (complex Modbus implementation) -# C0413: Import position (shared module import must be after sys.path modification) -# W0107: Unnecessary pass (used for read-only setValues methods) -# W0602: Global variable not assigned (threading.Event uses methods, not reassignment) -# W0621: Redefining name from outer scope (runtime_args parameter shadows global) -# C0415: Import outside toplevel (traceback imported in exception handlers) -# R0913: Too many arguments (required for segmented data block configuration) -# R0914: Too many local variables (complex address segmentation logic) -# R0917: Too many positional arguments (required for segmented data block configuration) import asyncio import os @@ -61,7 +52,6 @@ safe_extract_runtime_args_from_capsule, ) - class OpenPLCDeviceContext(ModbusDeviceContext): """ Custom Modbus device context that correctly handles FC5/FC6 response echo. @@ -148,7 +138,6 @@ def getValues(self, func_code, address, count=1): return super().getValues(func_code, address, count) - class OpenPLCCoilsDataBlock(ModbusSparseDataBlock): """Custom Modbus coils data block that mirrors OpenPLC bool_output using SafeBufferAccess""" @@ -239,7 +228,6 @@ def setValues(self, address, values): if logger: logger.error(f"Error setting coil {coil_addr}: {error_msg}") - class OpenPLCDiscreteInputsDataBlock(ModbusSparseDataBlock): """Custom Modbus discrete inputs data block that mirrors OpenPLC bool_input.""" @@ -303,7 +291,6 @@ def setValues(self, address, values): """Discrete inputs are read-only, this method should not be called""" pass # Silently ignore writes to read-only inputs - class OpenPLCInputRegistersDataBlock(ModbusSparseDataBlock): """Custom Modbus input registers data block that mirrors OpenPLC analog inputs.""" @@ -363,7 +350,6 @@ def setValues(self, address, values): """Input registers are read-only, this method should not be called""" pass # Silently ignore writes to read-only registers - class OpenPLCHoldingRegistersDataBlock(ModbusSparseDataBlock): """Custom Modbus holding registers data block that mirrors OpenPLC analog outputs.""" @@ -443,7 +429,6 @@ def setValues(self, address, values): if logger: logger.error(f"Error setting holding register {reg_addr}: {error_msg}") - class OpenPLCSegmentedCoilsDataBlock(ModbusSparseDataBlock): """ Segmented Modbus coils data block supporting both %QX (bool_output) and %MX (bool_memory). @@ -580,7 +565,6 @@ def setValues(self, address, values): if logger: logger.error(f"Error setting coil %MX{mx_addr}: {error_msg}") - class OpenPLCSegmentedHoldingRegistersDataBlock(ModbusSparseDataBlock): """ Segmented Modbus holding registers data block supporting %QW, %MW, %MD, and %ML. @@ -883,7 +867,6 @@ def setValues(self, address, values): finally: self.safe_buffer_access.release_mutex() - def parse_buffer_mapping_config(config_map): """ Parse buffer_mapping configuration from JSON config. @@ -972,7 +955,6 @@ def parse_buffer_mapping_config(config_map): "word_order": "high_word_first", } - # Global variables for plugin lifecycle server_task = None server_context = None @@ -989,7 +971,6 @@ def parse_buffer_mapping_config(config_map): RETRY_DELAY_BASE = 2.0 # Initial delay between restart attempts (seconds) RETRY_DELAY_MAX = 30.0 # Maximum delay between restart attempts (seconds) - def init(args_capsule): """Initialize the Modbus plugin""" global runtime_args, logger @@ -1027,7 +1008,6 @@ def init(args_capsule): traceback.print_exc() return False - def start_loop(): """Start the Modbus server with automatic restart on failure.""" global server_task, running, server_loop, server_started_event, server_error @@ -1223,13 +1203,11 @@ async def server_runner(): logger.error(f"Timeout waiting for server to start on {gIp}:{gPort}") return False - def _cancel_all_tasks(loop): """Cancel all running tasks on the event loop.""" for task in asyncio.all_tasks(loop): task.cancel() - def stop_loop(): """Stop the Modbus server gracefully. @@ -1272,7 +1250,6 @@ def stop_loop(): logger.info("Server stopped") return True - def cleanup(): """Cleanup plugin resources""" global server_context, runtime_args @@ -1283,7 +1260,6 @@ def cleanup(): logger.info("Plugin cleaned up") return True - async def main(): """Standalone server for testing""" # Create a proper mock runtime args that inherits from PluginRuntimeArgs @@ -1333,6 +1309,5 @@ def __str__(self): else: print("Failed to initialize plugin") - if __name__ == "__main__": asyncio.run(main()) diff --git a/core/src/drivers/plugins/python/modbus_slave/test_inputs.py b/core/src/drivers/plugins/python/modbus_slave/test_inputs.py index 414c21a3..c3f4f558 100644 --- a/core/src/drivers/plugins/python/modbus_slave/test_inputs.py +++ b/core/src/drivers/plugins/python/modbus_slave/test_inputs.py @@ -4,7 +4,6 @@ # tests/test_discrete_inputs.py import simple_modbus - def test_inputs_basic(advanced_sba, runtime_args): # <-- Fixed: Added runtime_args # The advanced_sba fixture has already patched SafeBufferAccess. # We must pass the *real* runtime_args object to the block. @@ -16,7 +15,6 @@ def test_inputs_basic(advanced_sba, runtime_args): # <-- Fixed: Added runtime_a values = block.getValues(5, 1) # modbus address 5 -> index 4 assert values == [1] - def test_inputs_invalid_range_non_blocking(advanced_sba, runtime_args): # <-- Fixed: Added runtime_args # Pass the real runtime_args object block = simple_modbus.OpenPLCDiscreteInputsDataBlock(runtime_args=runtime_args, num_inputs=4) diff --git a/core/src/drivers/plugins/python/modbus_slave/test_modbus_slave.py b/core/src/drivers/plugins/python/modbus_slave/test_modbus_slave.py index 8209fad0..e995efdd 100644 --- a/core/src/drivers/plugins/python/modbus_slave/test_modbus_slave.py +++ b/core/src/drivers/plugins/python/modbus_slave/test_modbus_slave.py @@ -36,17 +36,11 @@ def runtime_args(): return ra - def assert_block_zeroed(block, size): """Ensure the ModbusSparseDataBlock-like block returns zeros for fresh region.""" assert isinstance(block, ModbusSparseDataBlock) assert block.getValues(0, size) == [0] * size - -# ----------------------------------------------------------------------- -# Fake SafeBufferAccess used to observe locking behavior. -# We patch simple_modbus.SafeBufferAccess to return this object inside blocks -# ----------------------------------------------------------------------- class ObservingSafeBufferAccess: """ Test double for SafeBufferAccess that matches the REAL method signatures used @@ -130,8 +124,6 @@ def read_int_output(self, index, thread_safe=True): return (0, "Invalid buffer index") return (int(self.args.analog_output[index]) & 0xFFFF, "Success") - - # ----------------------------------------------------------------------- # Data Block tests (use ObservingSafeBufferAccess patched in) # ----------------------------------------------------------------------- @@ -157,7 +149,6 @@ def test_coils_read_write_and_locking(runtime_args): assert sba.acquire_count >= 1 assert sba.release_count >= 1 - def test_coils_invalid_ranges_return_zero(runtime_args): with patch("simple_modbus.SafeBufferAccess", new=ObservingSafeBufferAccess): block = simple_modbus.OpenPLCCoilsDataBlock(runtime_args, num_coils=8) @@ -170,7 +161,6 @@ def test_coils_invalid_ranges_return_zero(runtime_args): block.setValues(1000, [1, 1]) assert runtime_args.bool_output.count(1) == 0 - def test_discrete_inputs_behavior(runtime_args): with patch("simple_modbus.SafeBufferAccess", new=ObservingSafeBufferAccess): blk = simple_modbus.OpenPLCDiscreteInputsDataBlock(runtime_args, num_inputs=8) @@ -181,7 +171,6 @@ def test_discrete_inputs_behavior(runtime_args): val = blk.getValues(3, 1) assert val == [1] - def test_holding_registers_masking(runtime_args): with patch("simple_modbus.SafeBufferAccess", new=ObservingSafeBufferAccess): blk = simple_modbus.OpenPLCHoldingRegistersDataBlock(runtime_args, num_registers=8) @@ -191,7 +180,6 @@ def test_holding_registers_masking(runtime_args): stored = blk.getValues(1, 1)[0] assert stored == (70000 & 0xFFFF) - def test_input_registers_out_of_range(runtime_args): with patch("simple_modbus.SafeBufferAccess", new=ObservingSafeBufferAccess): blk = simple_modbus.OpenPLCInputRegistersDataBlock(runtime_args, num_registers=4) @@ -201,7 +189,6 @@ def test_input_registers_out_of_range(runtime_args): got = blk.getValues(10, 3) assert got == [0, 0, 0] - # ----------------------------------------------------------------------- # SafeBufferAccess concurrency test (basic) # Verify lock prevents race when multiple threads write/read @@ -231,7 +218,6 @@ def writer(idx, value, count=1000): expected = (1 + 2 + 3 + 4) * 200 & 0xFFFF assert runtime_args.analog_output[0] == expected - # ----------------------------------------------------------------------- # Verify that blocks do not raise on odd inputs (robustness) # ----------------------------------------------------------------------- diff --git a/core/src/drivers/plugins/python/modbus_slave/test_openplc_input_registers_datablock.py b/core/src/drivers/plugins/python/modbus_slave/test_openplc_input_registers_datablock.py index 9d9b6123..78c93a98 100644 --- a/core/src/drivers/plugins/python/modbus_slave/test_openplc_input_registers_datablock.py +++ b/core/src/drivers/plugins/python/modbus_slave/test_openplc_input_registers_datablock.py @@ -9,7 +9,6 @@ OpenPLCInputRegistersDataBlock ) - # ----------------------------- # Advanced Observing SBA mock # ----------------------------- @@ -80,7 +79,6 @@ def set_value(self, index, value): if 0 <= index < self.length: self._buf[index] = int(value) & 0xFFFF - # ----------------------------- # Fixtures # ----------------------------- @@ -97,7 +95,6 @@ def validate_pointers(self): return True, "" return FakeRuntimeArgs() - @pytest.fixture def runtime_args_invalid(): class BadRuntimeArgs: @@ -105,7 +102,6 @@ def validate_pointers(self): return False, "invalid" return BadRuntimeArgs() - # ----------------------------- # Tests # ----------------------------- @@ -121,7 +117,6 @@ def test_datablock_initialization(runtime_args): assert hasattr(db, "safe_buffer_access") assert db.safe_buffer_access.is_valid is True - def test_datablock_invalid_sba(runtime_args_invalid, capfd): # constructing with invalid runtime args should produce a warning and mark SBA invalid db = OpenPLCInputRegistersDataBlock(runtime_args_invalid, num_registers=4) @@ -130,7 +125,6 @@ def test_datablock_invalid_sba(runtime_args_invalid, capfd): assert "Warning" in out assert db.safe_buffer_access.is_valid is False - def test_datablock_read_from_sba(runtime_args): # Patch SafeBufferAccess to use our advanced mock with values [10,20,30,40,...] with patch("core.src.drivers.plugins.python.modbus_slave.simple_modbus.SafeBufferAccess", @@ -152,7 +146,6 @@ def test_datablock_read_from_sba(runtime_args): read_indices = [e[1] for e in read_events] assert read_indices == [0, 1, 2, 3] - def test_datablock_read_out_of_range(runtime_args): with patch( "core.src.drivers.plugins.python.modbus_slave.simple_modbus.SafeBufferAccess", @@ -169,12 +162,10 @@ def test_datablock_read_out_of_range(runtime_args): # it correctly returns 0 without calling the SBA mock. # REMOVED ASSERTION: assert any(e[0] == "read_oor" and e[1] == 4 for e in sba.trace) - def test_read_zero_length(runtime_args): db = OpenPLCInputRegistersDataBlock(runtime_args, num_registers=4) assert db.getValues(1, 0) == [] - def test_read_negative_index(runtime_args): with patch( "core.src.drivers.plugins.python.modbus_slave.simple_modbus.SafeBufferAccess", @@ -186,7 +177,6 @@ def test_read_negative_index(runtime_args): vals = db.getValues(-5, 2) assert vals == [0, 0] - def test_read_past_modbus_block_size(runtime_args): with patch( "core.src.drivers.plugins.python.modbus_slave.simple_modbus.SafeBufferAccess", @@ -197,7 +187,6 @@ def test_read_past_modbus_block_size(runtime_args): vals = db.getValues(10, 3) assert vals == [0, 0, 0] - def test_overlapping_reads_consistent(runtime_args): with patch( "core.src.drivers.plugins.python.modbus_slave.simple_modbus.SafeBufferAccess", @@ -218,7 +207,6 @@ def test_overlapping_reads_consistent(runtime_args): assert read_indices.count(1) >= 2 assert read_indices.count(2) >= 2 - def test_sba_invalid_returns_zero(runtime_args): # create an SBA subclass that reports invalid class AlwaysInvalid(AdvancedObservingSBA): diff --git a/core/src/drivers/plugins/python/opcua/address_space.py b/core/src/drivers/plugins/python/opcua/address_space.py index 4feeb2f2..39d112d7 100644 --- a/core/src/drivers/plugins/python/opcua/address_space.py +++ b/core/src/drivers/plugins/python/opcua/address_space.py @@ -43,7 +43,6 @@ VariablePermissions, ) - def _type_default(datatype: str) -> Any: """Per-type seed value for newly-created OPC-UA nodes. Replaces the removed `initial_value` config field — the first sync cycle @@ -55,7 +54,6 @@ def _type_default(datatype: str) -> Any: ByteString, so the seed was the wrong Python type for a whole tick.""" return default_for_opcua(datatype) - class AddressSpaceBuilder: """ Builds OPC-UA address space from configuration. diff --git a/core/src/drivers/plugins/python/opcua/callbacks.py b/core/src/drivers/plugins/python/opcua/callbacks.py index cb5a3f70..185bee91 100644 --- a/core/src/drivers/plugins/python/opcua/callbacks.py +++ b/core/src/drivers/plugins/python/opcua/callbacks.py @@ -32,7 +32,6 @@ from shared.plugin_config_decode.opcua_config_model import VariablePermissions - class PermissionCallbackHandler: """ Handles OPC-UA read/write permission callbacks. diff --git a/core/src/drivers/plugins/python/opcua/config.py b/core/src/drivers/plugins/python/opcua/config.py index 3cb1a5e6..732c7dce 100644 --- a/core/src/drivers/plugins/python/opcua/config.py +++ b/core/src/drivers/plugins/python/opcua/config.py @@ -39,7 +39,6 @@ OPCUA_CONFIG_MIN_FORMAT_VERSION, ) - def load_config(config_path: str) -> Optional[OpcuaConfig]: """ Load OPC UA configuration from JSON file. @@ -87,7 +86,6 @@ def load_config(config_path: str) -> Optional[OpcuaConfig]: log_error(f"Failed to load configuration: {e}") return None - def load_config_from_dict(raw_config: dict) -> Optional[OpcuaConfig]: """ Load OPC UA configuration from a dictionary. @@ -116,7 +114,6 @@ def load_config_from_dict(raw_config: dict) -> Optional[OpcuaConfig]: log_error(f"Failed to parse configuration: {e}") return None - def _normalize_config(raw_config: Any) -> dict: """ Normalize configuration to single-server format. @@ -142,7 +139,6 @@ def _normalize_config(raw_config: Any) -> dict: # Already in new format return raw_config - def get_default_config() -> OpcuaConfig: """ Get default configuration for development/testing. diff --git a/core/src/drivers/plugins/python/opcua/opcua_endpoints_config.py b/core/src/drivers/plugins/python/opcua/opcua_endpoints_config.py index 11f73828..146b95b8 100644 --- a/core/src/drivers/plugins/python/opcua/opcua_endpoints_config.py +++ b/core/src/drivers/plugins/python/opcua/opcua_endpoints_config.py @@ -15,13 +15,11 @@ except ImportError: PSUTIL_AVAILABLE = False - def _is_docker_interface(interface_name: str) -> bool: """Check if interface name looks like a Docker/container internal interface.""" docker_prefixes = ('docker', 'br-', 'veth', 'cni', 'flannel', 'cali', 'weave') return interface_name.lower().startswith(docker_prefixes) - def _is_docker_ip(ip: str) -> bool: """ Check if IP is in a Docker internal network range. @@ -41,7 +39,6 @@ def _is_docker_ip(ip: str) -> bool: pass return False - def _get_ips_from_psutil() -> List[str]: """Get non-loopback, non-Docker IPs using psutil (preferred method).""" if not PSUTIL_AVAILABLE: @@ -65,7 +62,6 @@ def _get_ips_from_psutil() -> List[str]: except Exception: return [] - def _get_ips_from_socket() -> List[str]: """ Get non-loopback, non-Docker IPs using socket (fallback, no network access required). @@ -101,7 +97,6 @@ def _get_ips_from_socket() -> List[str]: return non_loopback_ips - def _get_ip_from_external_connection() -> Optional[str]: """ Get IP by connecting to external address (last resort, requires network). @@ -121,7 +116,6 @@ def _get_ip_from_external_connection() -> Optional[str]: pass return None - def get_local_ip() -> Optional[str]: """ Get the local IP address of the machine. @@ -147,7 +141,6 @@ def get_local_ip() -> Optional[str]: # Last resort: external connection (requires network) return _get_ip_from_external_connection() - def get_available_hostnames() -> List[str]: """Get list of available hostnames/IPs for the server.""" hostnames = ["localhost", "127.0.0.1"] @@ -173,7 +166,6 @@ def get_available_hostnames() -> List[str]: return hostnames - def normalize_endpoint_url(endpoint_url: str) -> str: """ Normalize endpoint URL for better client compatibility. @@ -206,7 +198,6 @@ def normalize_endpoint_url(endpoint_url: str) -> str: return endpoint_url - def create_multiple_endpoints(base_endpoint: str) -> List[str]: """Create multiple endpoint variations for better connectivity.""" parsed = urlparse(base_endpoint) @@ -221,7 +212,6 @@ def create_multiple_endpoints(base_endpoint: str) -> List[str]: return endpoints - def suggest_client_endpoints(server_endpoint: str) -> Dict[str, str]: """Suggest different endpoint URLs for different client scenarios.""" parsed = urlparse(server_endpoint) @@ -233,7 +223,6 @@ def suggest_client_endpoints(server_endpoint: str) -> Dict[str, str]: "network_ip": f"opc.tcp://{get_local_ip()}:{parsed.port}{parsed.path}" if get_local_ip() else None } - def validate_endpoint_format(endpoint_url: str) -> bool: """Validate if endpoint URL has correct OPC-UA format.""" try: diff --git a/core/src/drivers/plugins/python/opcua/opcua_logging.py b/core/src/drivers/plugins/python/opcua/opcua_logging.py index e1d015ab..e191822e 100644 --- a/core/src/drivers/plugins/python/opcua/opcua_logging.py +++ b/core/src/drivers/plugins/python/opcua/opcua_logging.py @@ -12,7 +12,6 @@ from typing import Optional, Callable import sys - class OpcuaLogger: """ Singleton logger for OPC UA plugin. @@ -109,28 +108,23 @@ def debug(self, message: str) -> None: timestamp = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M:%S") print(f"[{timestamp}] [DEBUG] [OPCUA] {message}", file=sys.stdout) - # Module-level convenience functions def get_logger() -> OpcuaLogger: """Get the singleton logger instance.""" return OpcuaLogger.get_instance() - def log_info(message: str) -> None: """Log an informational message.""" get_logger().info(message) - def log_warn(message: str) -> None: """Log a warning message.""" get_logger().warn(message) - def log_error(message: str) -> None: """Log an error message.""" get_logger().error(message) - def log_debug(message: str) -> None: """Log a debug message.""" get_logger().debug(message) diff --git a/core/src/drivers/plugins/python/opcua/opcua_memory.py b/core/src/drivers/plugins/python/opcua/opcua_memory.py index 00021571..e71e78be 100644 --- a/core/src/drivers/plugins/python/opcua/opcua_memory.py +++ b/core/src/drivers/plugins/python/opcua/opcua_memory.py @@ -42,7 +42,6 @@ from opcua_types import VariableMetadata from opcua_logging import log_debug, log_error, log_warn - # TIME-related datatypes are encoded as 8-byte signed integers in # nanoseconds (matching strucpp's TIME_t / DATE_t / TOD_t / DT_t). TIME_DATATYPES = frozenset(["TIME", "DATE", "TOD", "DT"]) @@ -55,45 +54,26 @@ # Status code from debug_dispatch.hpp STATUS_OK = 0x7E -# STRING / WSTRING are variable-length and share one wire format with -# strucpp's debug surface: a single count byte, then the payload. -# -# STRING [count][count bytes] padded to 127 bytes -# WSTRING [count][count * 2 bytes LE] padded to 253 bytes -# -# `count` is in BYTES for STRING and in UTF-16 CODE UNITS for WSTRING, and is -# capped at DEBUG_STRING_CAP on both sides -- strucpp's `validate_payload` -# REFUSES a longer write outright rather than truncating it, so the truncation -# has to happen here. -# -# BYTES, not characters: `IECString` stores `char data_[MaxLen + 1]` with -# `length_` counting bytes, and `_truncate_utf8` below spends its whole body on -# that fact. The two comments used to contradict each other, and the difference -# is user-visible -- 126 bytes is ~63 two-byte accented characters, or ~31 -# four-byte emoji, not 126 of either. +# STRING/WSTRING wire: [count][payload]. `count` is BYTES (STRING) or +# UTF-16 CODE UNITS (WSTRING), capped at DEBUG_STRING_CAP. strucpp's +# validate_payload refuses over-cap; truncation happens here. STRING_DATATYPES = frozenset(["STRING", "WSTRING"]) DEBUG_STRING_CAP = 126 - -# Warnings raised from the READ path, which asyncua calls once per variable per -# client Read. A leaf that is persistently malformed is not a new event every -# poll -- at a one-second poll and a handful of clients it is a log that scrolls -# its own cause off the screen. Say it once per distinct problem and stay quiet -# after that; the condition is a property of the program, not of the poll. +# Warnings raised from the READ path (called once per variable per +# client Read). Say each distinct problem once; it is a property of +# the program, not of the poll. _warned: set = set() - def _warn_once(key: str, message: str) -> None: if key in _warned: return _warned.add(key) log_warn(f"{message} (further identical warnings suppressed)") - def _is_string(datatype: str) -> bool: return (datatype or "").upper() in STRING_DATATYPES - def _decode_string(datatype: str, buf: Any, n: int) -> Optional[Any]: """Decode strucpp's [count][payload] wire form. @@ -123,14 +103,11 @@ def _decode_string(datatype: str, buf: Any, n: int) -> Optional[Any]: raw = bytes(bytearray(buf[1:1 + payload_len])) if wide: return raw - # UTF-8, because that is what the rest of the system already agrees on: - # the editor's debugger decodes this same wire form with the `len8-utf8` - # codec (`variable-sizes.ts`). `errors="replace"` rather than strict so a - # truncated multi-byte sequence degrades to one replacement character - # instead of taking down the read. + # UTF-8 (matches the editor's `len8-utf8` codec). + # errors="replace" so a truncated multi-byte sequence degrades to + # a replacement character instead of failing the read. return raw.decode("utf-8", errors="replace") - def _encode_string(datatype: str, value: Any) -> Optional[bytes]: """Encode a Python value into strucpp's [count][payload] wire form.""" wide = datatype.upper() == "WSTRING" @@ -146,12 +123,8 @@ def _encode_string(datatype: str, value: Any) -> Optional[bytes]: log_warn("WSTRING payload has an odd byte count; dropping the trailing byte") raw = raw[:-1] count = min(len(raw) // 2, DEBUG_STRING_CAP) - # Do not cut between the halves of a surrogate pair. The STRING path - # goes to real trouble not to split a UTF-8 sequence (_truncate_utf8); - # the same care is owed here, because a lone high surrogate is not a - # shorter string, it is an undecodable one -- `bytes.decode('utf-16-le')` - # raises on it. Astral characters (emoji, most CJK extensions) are the - # common case. + # Do not cut a UTF-16 surrogate pair in half: a lone high surrogate + # is undecodable and raises in bytes.decode('utf-16-le'). if count > 0: last = int.from_bytes(raw[(count - 1) * 2 : count * 2], "little") if 0xD800 <= last <= 0xDBFF: # high surrogate with its pair cut off @@ -166,7 +139,6 @@ def _encode_string(datatype: str, value: Any) -> Optional[bytes]: count = len(payload) return bytes([count]) + payload - def _truncate_utf8(raw: bytes, limit: int) -> bytes: """Cut `raw` to at most `limit` bytes without splitting a character. @@ -181,7 +153,6 @@ def _truncate_utf8(raw: bytes, limit: int) -> bytes: end -= 1 return raw[:end] - def _ctype_for(datatype: str) -> Optional[Any]: """Map an IEC type name to the ctypes scalar that owns its bytes on disk. Returns None for variable-length types (STRING/WSTRING) @@ -213,15 +184,12 @@ def _ctype_for(datatype: str) -> Optional[Any]: # strucpp encodes time-family types as int64 nanoseconds (TIME_t). return ctypes.c_int64 if t in STRING_DATATYPES: - # Variable-length: no fixed-width ctype owns these. They ARE readable - # and writable -- strucpp wires read_string / write_string / - # read_wstring / write_wstring into type_ops[] at tags 19/20 -- through - # _decode_string / _encode_string instead. Callers must therefore pair - # a None from here with an _is_string() check rather than giving up. + # Variable-length: no fixed-width ctype. Read/write go through + # _decode_string / _encode_string; callers must pair None with + # an _is_string() check rather than giving up. return None return None - def debug_read_value(args: Any, arr: int, elem: int, datatype: str) -> Optional[Any]: """Read a single PLC variable through args.debug_read and decode it into a Python value matching the IEC datatype. @@ -255,7 +223,6 @@ def debug_read_value(args: Any, arr: int, elem: int, datatype: str) -> Optional[ typed = ctypes.cast(buf, ctypes.POINTER(ctype)).contents return typed.value - def debug_write_value(args: Any, arr: int, elem: int, datatype: str, value: Any) -> bool: """Soft-write a Python value to a PLC variable through args.debug_write. Encodes the value per the IEC datatype, then @@ -294,7 +261,6 @@ def debug_write_value(args: Any, arr: int, elem: int, datatype: str, value: Any) return False return status == STATUS_OK - def debug_force_value(args: Any, arr: int, elem: int, datatype: str, value: Any) -> bool: """Force-write a value (debug_set with forcing=True). Pins the variable until explicitly unforced. Distinct from debug_write @@ -305,10 +271,7 @@ def debug_force_value(args: Any, arr: int, elem: int, datatype: str, value: Any) if ctype is None and not _is_string(datatype): return False - # Strings force through the same wire form as a write. Read and write each - # grew a string path and force did not, so forcing a STRING answered False - # with no reason given -- the same silence that hid the read bug, kept - # alive in the one operation nobody calls yet. + # Strings force through the same wire form as a write. if _is_string(datatype): encoded_str = _encode_string(datatype, value) if encoded_str is None: @@ -336,7 +299,6 @@ def debug_force_value(args: Any, arr: int, elem: int, datatype: str, value: Any) return False return status == STATUS_OK - def debug_unforce(args: Any, arr: int, elem: int) -> bool: """Release a force on a variable (debug_set with forcing=False). The bytes/len arguments are ignored by the runtime's unforce path @@ -355,7 +317,6 @@ def debug_unforce(args: Any, arr: int, elem: int) -> bool: return False return status == STATUS_OK - def initialize_variable_cache( args: Any, addrs: Iterable[Tuple[int, int]], @@ -393,7 +354,6 @@ def initialize_variable_cache( log_debug(f"Cached size+type metadata for {len(cache)} variables") return cache - def time_to_timespec(value_ns: int) -> Tuple[int, int]: """Split an int64 nanosecond value into (tv_sec, tv_nsec) for callers that want to expose TIME-family values as their CODESYS @@ -406,7 +366,6 @@ def time_to_timespec(value_ns: int) -> Tuple[int, int]: nsec = value_ns % 1_000_000_000 return int(sec), int(nsec) - def timespec_to_time(tv_sec: int, tv_nsec: int) -> int: """Compose (tv_sec, tv_nsec) back into an int64 nanosecond value.""" return int(tv_sec) * 1_000_000_000 + int(tv_nsec) diff --git a/core/src/drivers/plugins/python/opcua/opcua_security.py b/core/src/drivers/plugins/python/opcua/opcua_security.py index a4badb32..e34a71ea 100644 --- a/core/src/drivers/plugins/python/opcua/opcua_security.py +++ b/core/src/drivers/plugins/python/opcua/opcua_security.py @@ -50,13 +50,11 @@ except ImportError: from opcua_logging import log_debug, log_error, log_info, log_warn - # ioctl constants for network interface enumeration (Linux) _SIOCGIFCONF = 0x8912 # ioctl request code to get interface configuration _SIZEOF_IFREQ = 40 # sizeof(struct ifreq) on 64-bit Linux _MAX_INTERFACES = 128 # Maximum number of network interfaces to query - def get_local_ip_addresses() -> Set[str]: """ Get all local IP addresses of the machine. @@ -134,7 +132,6 @@ def get_local_ip_addresses() -> Set[str]: return ip_addresses - def generate_certificate_with_sans( cert_path: Path, key_path: Path, @@ -288,7 +285,6 @@ def generate_certificate_with_sans( log_error(f"Failed to generate certificate: {e}") return False - class OpenPLCRoleRuleset(PermissionRuleset): """ Custom permission ruleset for OpenPLC OPC-UA server. @@ -366,7 +362,6 @@ def check_validity(self, user, action_type_id, body): return True return False - class OpcuaSecurityManager: """Manages OPC-UA security configuration and certificates.""" diff --git a/core/src/drivers/plugins/python/opcua/opcua_types.py b/core/src/drivers/plugins/python/opcua/opcua_types.py index e938b46b..990b2b7e 100644 --- a/core/src/drivers/plugins/python/opcua/opcua_types.py +++ b/core/src/drivers/plugins/python/opcua/opcua_types.py @@ -7,7 +7,6 @@ from typing import Any, Optional, Tuple from asyncua.common.node import Node - @dataclass class VariableNode: """Represents an OPC-UA node mapped to a PLC debug variable. @@ -25,7 +24,6 @@ class VariableNode: array_index: Optional[int] = None # 0..length-1 within the array array_length: Optional[int] = None # Length of array (for array nodes only) - @dataclass class VariableMetadata: """Metadata cache for direct memory access via debug_read/debug_write.""" diff --git a/core/src/drivers/plugins/python/opcua/opcua_utils.py b/core/src/drivers/plugins/python/opcua/opcua_utils.py index 062a7c2a..cf9a88a0 100644 --- a/core/src/drivers/plugins/python/opcua/opcua_utils.py +++ b/core/src/drivers/plugins/python/opcua/opcua_utils.py @@ -20,31 +20,9 @@ except ImportError: from opcua_logging import log_info, log_warn, log_error - # TIME-related datatypes that use IEC_TIMESPEC structure TIME_DATATYPES = frozenset(["TIME", "DATE", "TOD", "DT"]) - -# --------------------------------------------------------------------------- -# Per-type defaults — TWO tables, deliberately, and only two. -# -# There were four, in four files, and they disagreed: `address_space` seeded a -# WSTRING with `""` while everywhere else used `b""`, which is the wrong Python -# type for a node this plugin maps to a ByteString. That is what duplication -# costs — the copies drift, and the one that drifts is the one nobody reads. -# -# Two remain because there are genuinely two directions, not because nobody -# merged them: -# -# default_for_opcua() what a CLIENT should see (BOOL -> False) -# default_for_plc() what the PLC side encodes (BOOL -> 0) -# -# A default is a fallback, never an answer. A caller that substitutes one is -# telling the client something it does not know, so it must also mark the value -# Bad — see the read callback and `_push_array_node` in synchronization.py. -# --------------------------------------------------------------------------- - - def default_for_opcua(datatype: str) -> Any: """The OPC-UA-side representation of "nothing to report" for a type.""" t = (datatype or "").upper() @@ -58,7 +36,6 @@ def default_for_opcua(datatype: str) -> Any: return b"" # WSTRING is served as a ByteString, so bytes return 0 - def default_for_plc(datatype: str) -> Any: """The PLC-side representation, as the write path would encode it.""" t = (datatype or "").upper() @@ -75,7 +52,6 @@ def default_for_plc(datatype: str) -> Any: return (0, 0) return 0 - def map_plc_to_opcua_type(plc_type: str) -> ua.VariantType: """Map plc datatype to OPC-UA VariantType.""" type_mapping = { @@ -103,18 +79,10 @@ def map_plc_to_opcua_type(plc_type: str) -> ua.VariantType: "LREAL": ua.VariantType.Double, # IEC 61131-3 LREAL = 64-bit float # String type "STRING": ua.VariantType.String, - # WSTRING is UTF-16LE code units, carried as an opaque ByteString - # rather than a UA String. Transcoding to UTF-8 would need a scratch - # buffer the size of the string and is lossy for lone surrogates, so - # the client is given the bytes and the encoding is documented on the - # node. Same choice the baremetal runtime makes, so a project behaves - # the same on both targets. - # - # Leaving this out did NOT merely lose the mapping: the `.get` default - # below is `VariantType.Variant`, so asyncua tried to serialise a - # nested Variant around an int and died encoding the RESPONSE - # ("'int' object has no attribute 'VariantType'"), which the client saw - # as BadInternalError and which took the whole response with it. + # WSTRING carried as opaque ByteString (UTF-16LE units) rather than + # UA String: transcoding to UTF-8 needs a scratch buffer and is + # lossy for lone surrogates. Must be mapped: the `.get` default + # below is `VariantType.Variant`, which asyncua cannot serialise. "WSTRING": ua.VariantType.ByteString, # TIME-related types "TIME": ua.VariantType.Int64, # Duration in milliseconds @@ -125,7 +93,6 @@ def map_plc_to_opcua_type(plc_type: str) -> ua.VariantType: mapped_type = type_mapping.get(plc_type.upper(), ua.VariantType.Variant) return mapped_type - def timespec_to_milliseconds(tv_sec: int, tv_nsec: int) -> int: """ Convert IEC_TIMESPEC (tv_sec, tv_nsec) to milliseconds. @@ -139,7 +106,6 @@ def timespec_to_milliseconds(tv_sec: int, tv_nsec: int) -> int: """ return (tv_sec * 1000) + (tv_nsec // 1_000_000) - def milliseconds_to_timespec(ms: int) -> tuple[int, int]: """ Convert milliseconds to IEC_TIMESPEC format (tv_sec, tv_nsec). @@ -154,7 +120,6 @@ def milliseconds_to_timespec(ms: int) -> tuple[int, int]: tv_nsec = (ms % 1000) * 1_000_000 return (tv_sec, tv_nsec) - def convert_value_for_opcua(datatype: str, value: Any) -> Any: """Convert PLC debug variable value to OPC-UA compatible format.""" # The debug utils return raw integer values based on variable size @@ -207,11 +172,9 @@ def convert_value_for_opcua(datatype: str, value: Any) -> Any: return ctypes.c_uint64(clamped_value).value elif datatype.upper() in ["FLOAT", "REAL", "LREAL"]: - # debug_read_value casts the raw bytes to c_float / c_double, so - # what arrives here is the number itself, never its bit pattern. - # The MatIEC-era debug protocol did hand over raw integers, and the - # struct.unpack that decoded them outlived it -- on the STruC++ - # debug surface it would turn an integer-valued REAL of 1 into + # debug_read_value already casts to c_float/c_double, so this + # receives the number itself, not its bit pattern. A legacy + # struct.unpack on the integer bits would turn REAL 1 into # 1.4e-45. Mirror of the write-side fix below. return float(value) @@ -312,7 +275,6 @@ def convert_value_for_opcua(datatype: str, value: Any) -> Any: log_warn(f"Failed to convert value {value} to OPC-UA format for {datatype}: {e}") return default_for_opcua(datatype) - def convert_value_for_plc(datatype: str, value: Any) -> Any: """Convert OPC-UA value to PLC debug variable format.""" # Handle different OPC-UA value types more robustly @@ -366,11 +328,9 @@ def convert_value_for_plc(datatype: str, value: Any) -> Any: return ctypes.c_uint64(clamped_value).value elif datatype.upper() in ["FLOAT", "REAL", "LREAL"]: - # debug_write_value packs this through c_float / c_double, so hand - # it the number. Packing the bit pattern here -- correct back when - # the debug protocol exchanged raw integers -- meant a client - # writing 42.0 stored 1109917696.0 in the PLC (forum thread - # "Error changing values via OPC UA"). + # debug_write_value packs through c_float/c_double, so hand + # the number. Packing the integer bit-pattern would store + # 1109917696.0 for a client write of 42.0. return float(value) elif datatype.upper() == "STRING": @@ -439,7 +399,6 @@ def convert_value_for_plc(datatype: str, value: Any) -> Any: log_warn(f"Failed to convert value {value} to {datatype}, using default: {e}") return default_for_plc(datatype) - def infer_var_type(size: int) -> str: """ Infer variable type from size. diff --git a/core/src/drivers/plugins/python/opcua/plugin.py b/core/src/drivers/plugins/python/opcua/plugin.py index 8ab4901a..1e38eedb 100644 --- a/core/src/drivers/plugins/python/opcua/plugin.py +++ b/core/src/drivers/plugins/python/opcua/plugin.py @@ -49,7 +49,6 @@ from opcua_logging import get_logger, log_debug, log_error, log_info, log_warn from server import OpcuaServerManager - # Plugin state _runtime_args = None _buffer_accessor: Optional[SafeBufferAccess] = None @@ -59,7 +58,6 @@ _stop_event = threading.Event() _loop: Optional[asyncio.AbstractEventLoop] = None - class _PermissionDenialFilter(logging.Filter): """Quiet asyncua's "Error while processing message" traceback when the underlying cause is a UaError raised by our pre-read / @@ -98,12 +96,10 @@ def filter(self, record: logging.LogRecord) -> bool: return False # drop the record return True - def _install_asyncua_log_filter() -> None: """Install the permission-denial filter on asyncua's processor logger.""" logging.getLogger("asyncua.server.uaprocessor").addFilter(_PermissionDenialFilter()) - def init(args_capsule) -> bool: """ Initialize the OPC UA plugin. @@ -142,7 +138,6 @@ def init(args_capsule) -> bool: log_error(f"Initialization error: {e}") return False - def start_loop() -> bool: """ Start the OPC UA server. @@ -200,7 +195,6 @@ def start_loop() -> bool: log_error(f"Failed to start server: {e}") return False - def stop_loop() -> bool: """ Stop the OPC UA server. @@ -255,7 +249,6 @@ def stop_loop() -> bool: log_error(f"Error stopping server: {e}") return False - def cleanup() -> bool: """ Clean up plugin resources. @@ -287,7 +280,6 @@ def cleanup() -> bool: log_error(f"Cleanup error: {e}") return False - def _cancel_all_tasks(loop): """Cancel all running tasks on the event loop. @@ -296,7 +288,6 @@ def _cancel_all_tasks(loop): for task in asyncio.all_tasks(loop): task.cancel() - def _run_server_thread() -> None: """ Server thread main function. @@ -342,7 +333,6 @@ async def _monitor_stop(): finally: _loop = None - # For backwards compatibility, also export as module-level functions # that match the old plugin interface __all__ = ["init", "start_loop", "stop_loop", "cleanup"] diff --git a/core/src/drivers/plugins/python/opcua/server.py b/core/src/drivers/plugins/python/opcua/server.py index e60fba01..c55a6784 100644 --- a/core/src/drivers/plugins/python/opcua/server.py +++ b/core/src/drivers/plugins/python/opcua/server.py @@ -48,7 +48,6 @@ from shared import SafeBufferAccess from shared.plugin_config_decode.opcua_config_model import OpcuaConfig - class OpcuaServerManager: """ Manages the OPC-UA server lifecycle and coordinates components. diff --git a/core/src/drivers/plugins/python/opcua/synchronization.py b/core/src/drivers/plugins/python/opcua/synchronization.py index 857bf33e..23c7e697 100644 --- a/core/src/drivers/plugins/python/opcua/synchronization.py +++ b/core/src/drivers/plugins/python/opcua/synchronization.py @@ -4,10 +4,9 @@ """ OPC-UA ↔ PLC synchronization — request-driven. -Replaces the old unconditional bidirectional poll (which read every -writable node every cycle and wrote it back to the PLC — OpenPLC -bug #2: OPC-UA fighting the program for readwrite variables). The -model is now: +Replaces the old unconditional bidirectional poll, which read every +writable node every cycle and wrote it back to the PLC — making the +server fight the program for readwrite variables. The model is now: - READS (client → server): a per-node value_callback returns the LIVE PLC value via args.debug_read at read time. No staleness, no @@ -74,13 +73,11 @@ map_plc_to_opcua_type, ) - # Address tuple type alias for clarity. Addr = Tuple[int, int] _VALUE_ATTR = ua.AttributeIds.Value - class SynchronizationManager: """Request-driven OPC-UA ↔ PLC value bridge (see module docstring).""" @@ -188,14 +185,10 @@ def _register_value_hooks(self) -> None: "falling back to push-only sync") return - # A value_setter is installed on EVERY node — not only readwrite - # ones. Reason: write_attribute_value() clears value_callback when no - # setter is present (its else-branch), so our subscription push to a - # readonly node would otherwise kill that node's live-read callback. - # The setter forwards to the PLC only for readwrite nodes; on readonly - # nodes it is a no-op (client writes are already denied upstream by the - # PreWrite permission callback), but it keeps value_callback alive - # across pushes. + # Install value_setter on EVERY node. write_attribute_value() + # clears value_callback when no setter is present, so a push + # to a readonly node would kill its live-read. Readonly setter + # is a no-op; PreWrite already denies client writes. for (arr, elem), node in self.variable_nodes.items(): nodeid = node.node.nodeid try: @@ -250,12 +243,9 @@ def _make_read_callback( def callback(nodeid: Any, attr: Any) -> ua.DataValue: try: - # A read that did not reach the PLC is reported as BAD, not as - # a default stamped Good. Substituting a default and calling it - # Good gives the client no way to tell "the string is empty" - # from "this server cannot read strings" -- which is exactly - # how STRING reads went unnoticed: every one of them returned - # '' with a Good status while the PLC held a value. + # A failed read is reported BAD, not a default stamped Good: + # the client must be able to distinguish "value is empty" + # from "server cannot read it". failed = False if length > 0: values = [] @@ -351,33 +341,9 @@ def _write_one(self, addr: Addr, datatype: str, plc_value: Any) -> None: plc_value = int(tv_sec) * 1_000_000_000 + int(tv_nsec) ok = debug_write_value(self.args, addr[0], addr[1], datatype, plc_value) if not ok: - # KNOWN LIMITATION: the client is still told Good. - # - # asyncua's value_setter returns None -- there is no channel for a - # per-value StatusCode -- and `write_attribute_value` calls it - # unguarded before `return ua.StatusCode()`. Raising here would - # propagate out of the Write service, which has no try/except - # around its per-value loop, and fault the WHOLE request including - # the values that did write. A coarse fault is worse than a log for - # a multi-value write, so this stays a log until asyncua grows a - # way to report one value as bad. - # - # What still reaches here is NARROWER than a refused write, and it - # is worth being exact because the gap is silent. - # - # `plugin_debug_write` (plugin_driver.c) answers 0x7E as soon as the - # write is QUEUED, and `apply_global` (debug_write_journal.cpp) - # discards whatever `strucpp_debug_write` returns when the journal - # is later applied. So an out-of-bounds leaf, a CONSTANT/read-only - # leaf and an over-cap payload are all reported Good to the client - # AND never logged: the write is simply dropped at apply time with - # nobody watching. - # - # Only three things still produce False: no program loaded (0x81), - # the journal queue being full (0x82), and a Python-side encode - # failure. Closing the rest needs the applied status carried back - # out of the journal, which is a change to the journal contract - # rather than to this plugin -- DOPE-647. + # KNOWN LIMITATION: client is still told Good. asyncua's + # value_setter has no per-value StatusCode channel; raising + # would fault the whole multi-value Write. log_error(f"debug_write({addr[0]}, {addr[1]}) failed") # ----------------------------------------------------------------- @@ -490,18 +456,9 @@ async def _push_scalar_node(self, node: VariableNode, value: Any) -> None: async def _push_array_node( self, node: VariableNode, base_arr: int, base_elem: int ) -> None: - # An element that did not read is reported BAD for the whole array, - # exactly as the read callback reports a failed scalar. Substituting a - # default and pushing it Good is the bug this plugin was fixed for -- - # the subscriber cannot tell "the element is 0" from "this element - # could not be read", and an array kept its own copy of that mistake - # one function away from the callback that lost it. - # - # The status is per-DataValue, not per-element, so one bad element - # marks the push: OPC-UA has no way to say "element 3 is stale" on a - # plain array value. The scalar path above does not need this -- its - # caller skips the push entirely when the read fails, leaving the - # client's last good value in place. + # Any element that failed to read marks the whole array BAD — + # OPC-UA has no per-element status on a plain array value, and + # pushing default-filled Good would hide the failure. length = node.array_length or 0 values = [] failed = False diff --git a/core/src/drivers/plugins/python/opcua/tests/test_server_standalone.py b/core/src/drivers/plugins/python/opcua/tests/test_server_standalone.py index 55368aae..a3ad6852 100644 --- a/core/src/drivers/plugins/python/opcua/tests/test_server_standalone.py +++ b/core/src/drivers/plugins/python/opcua/tests/test_server_standalone.py @@ -28,7 +28,6 @@ from asyncua import Server, ua - class TestOpcuaServer: """ Standalone test server for subscription verification. @@ -264,7 +263,6 @@ async def run(self): finally: await self.stop() - async def main(): """Main entry point.""" print("=" * 60) @@ -279,7 +277,6 @@ async def main(): print("\nShutdown requested...") await server.stop() - if __name__ == "__main__": try: asyncio.run(main()) diff --git a/core/src/drivers/plugins/python/opcua/tests/test_subscription_client.py b/core/src/drivers/plugins/python/opcua/tests/test_subscription_client.py index 7ac8bbec..202a448b 100644 --- a/core/src/drivers/plugins/python/opcua/tests/test_subscription_client.py +++ b/core/src/drivers/plugins/python/opcua/tests/test_subscription_client.py @@ -20,7 +20,6 @@ from asyncua import Client, ua - class SubscriptionHandler: """ Handler for subscription notifications. @@ -62,7 +61,6 @@ def event_notification(self, event): """Called when an event is received.""" print(f"Event received: {event}") - async def test_subscriptions(endpoint_url: str): """ Test OPC-UA subscriptions. @@ -220,7 +218,6 @@ async def find_variables(node, depth=0, max_depth=3): if handler.last_notification_time: print(f"Last notification at: {handler.last_notification_time}") - async def main(): """Main entry point.""" # Default endpoint @@ -232,7 +229,6 @@ async def main(): await test_subscriptions(endpoint_url) - if __name__ == "__main__": try: asyncio.run(main()) diff --git a/core/src/drivers/plugins/python/opcua/user_manager.py b/core/src/drivers/plugins/python/opcua/user_manager.py index d39ef4c9..a96faf20 100644 --- a/core/src/drivers/plugins/python/opcua/user_manager.py +++ b/core/src/drivers/plugins/python/opcua/user_manager.py @@ -33,7 +33,6 @@ PBKDF2_HASH_NAME = "sha256" PBKDF2_SALT_LENGTH = 16 - def _pbkdf2_hash_password(password: str) -> str: """ Hash a password using PBKDF2-HMAC-SHA256. @@ -55,7 +54,6 @@ def _pbkdf2_hash_password(password: str) -> str: hash_b64 = base64.b64encode(hash_bytes).decode("ascii") return f"pbkdf2:{PBKDF2_HASH_NAME}:{PBKDF2_ITERATIONS}${salt_b64}${hash_b64}" - def _pbkdf2_verify_password(password: str, password_hash: str) -> bool: """ Verify a password against a PBKDF2 hash. @@ -100,7 +98,6 @@ def _pbkdf2_verify_password(password: str, password_hash: str) -> bool: except Exception: return False - def hash_password(password: str) -> str: """ Hash a password using the best available method. @@ -139,7 +136,6 @@ def hash_password(password: str) -> str: DEFAULT_LOCKOUT_DURATION_SECONDS = 300 # 5 minutes DEFAULT_ATTEMPT_WINDOW_SECONDS = 60 # 1 minute window for counting attempts - @dataclass class AuthAttemptTracker: """Tracks authentication attempts for rate limiting.""" @@ -148,7 +144,6 @@ class AuthAttemptTracker: first_attempt_time: float = 0.0 lockout_until: float = 0.0 - @dataclass class RateLimitConfig: """Configuration for rate limiting.""" @@ -157,7 +152,6 @@ class RateLimitConfig: lockout_duration_seconds: float = DEFAULT_LOCKOUT_DURATION_SECONDS attempt_window_seconds: float = DEFAULT_ATTEMPT_WINDOW_SECONDS - class RateLimiter: """ Rate limiter for authentication attempts. @@ -286,7 +280,6 @@ def cleanup_expired(self) -> int: return len(expired) - class OpenPLCUserManager(UserManager): """ Custom user manager for OpenPLC authentication. @@ -331,11 +324,8 @@ def __init__(self, config: OpcuaConfig, rate_limit_config: Optional[RateLimitCon # Initialize rate limiter for brute-force protection self.rate_limiter = RateLimiter(rate_limit_config) - # Build user dictionaries - # A password user with no hash cannot authenticate — `_validate_password` - # has no format to match and refuses it — so registering it only creates - # an account that looks configured and never works. Refuse it at load, - # where the reason can be said once, instead of once per failed login. + # Password user without a hash cannot authenticate; refuse it at + # load (once) rather than once per failed login. _credentialled = [] for user in config.users: if user.type != "password": @@ -360,14 +350,9 @@ def __init__(self, config: OpcuaConfig, rate_limit_config: Optional[RateLimitCon elif user.type == "certificate" and user.certificate_id: self._user_roles[f"cert:{user.certificate_id}"] = str(user.role) - # Anonymous role is a PER-PROFILE field, but anonymous authentication - # carries no endpoint identity into get_user(), so the lookup can only - # take the FIRST enabled profile that offers Anonymous - # (_find_profile_by_auth_method). The editor is where two Anonymous - # profiles should be prevented; this is the belt-and-suspenders: if more - # than one enabled profile offers Anonymous, warn at load (once, where - # the admin sees it) that list order decides the role, and name the one - # that wins. Behaviour is unchanged — the first profile is still used. + # Anonymous role is per-profile, but anonymous auth carries no + # endpoint identity, so the first enabled Anonymous profile wins + # (list order). Warn once if more than one is configured. anon_profiles = [ p for p in getattr(config.server, "security_profiles", []) if getattr(p, "enabled", False) and "Anonymous" in getattr(p, "auth_methods", []) @@ -588,14 +573,9 @@ def _authenticate_anonymous(self, profile: Any) -> tuple[Optional[User], Optiona log_warn("Anonymous authentication not allowed for this profile") return None, None - # Explicit, config-driven role (defaults to viewer). The value is - # validated at parse time (opcua_config_model.SecurityProfile), and - # normalized here through normalize_role() — the same strip+lowercase - # normalization callbacks applies to every other role — so casing or - # whitespace ("Engineer", " engineer ") cannot make an anonymous session - # silently degrade to viewer. Map to the asyncua role for - # operation-level checks; per-variable enforcement uses the OpenPLC role - # string. normalize_role always returns a ROLE_MAPPING key. + # Config-driven role (defaults to viewer), normalized here so + # casing/whitespace cannot silently degrade an anonymous session + # to viewer. normalize_role always returns a ROLE_MAPPING key. openplc_role = normalize_role(getattr(profile, "anonymous_role", "viewer") or "viewer") asyncua_role = self.ROLE_MAPPING[openplc_role] diff --git a/core/src/drivers/plugins/python/shared/batch_processor.py b/core/src/drivers/plugins/python/shared/batch_processor.py index cc2e19db..6033c596 100644 --- a/core/src/drivers/plugins/python/shared/batch_processor.py +++ b/core/src/drivers/plugins/python/shared/batch_processor.py @@ -26,7 +26,6 @@ from buffer_accessor import GenericBufferAccessor from mutex_manager import MutexManager - class BatchProcessor(IBatchProcessor): """ Processes batch operations for optimized buffer access. diff --git a/core/src/drivers/plugins/python/shared/buffer_accessor.py b/core/src/drivers/plugins/python/shared/buffer_accessor.py index be8dfcd3..cd15e409 100644 --- a/core/src/drivers/plugins/python/shared/buffer_accessor.py +++ b/core/src/drivers/plugins/python/shared/buffer_accessor.py @@ -28,7 +28,6 @@ from component_interfaces import IBufferAccessor from mutex_manager import MutexManager - class GenericBufferAccessor(IBufferAccessor): """ Generic buffer accessor that handles all buffer types uniformly. diff --git a/core/src/drivers/plugins/python/shared/buffer_types.py b/core/src/drivers/plugins/python/shared/buffer_types.py index f2ecd924..f7c94140 100644 --- a/core/src/drivers/plugins/python/shared/buffer_types.py +++ b/core/src/drivers/plugins/python/shared/buffer_types.py @@ -18,7 +18,6 @@ # Fall back to absolute imports (when testing standalone) from component_interfaces import IBufferType - class BoolBufferType(IBufferType): """Boolean buffer type (1-bit values accessed via bit indexing)""" @@ -42,7 +41,6 @@ def requires_bit_index(self) -> bool: def ctype_class(self) -> type: return ctypes.c_uint8 - class ByteBufferType(IBufferType): """Byte buffer type (8-bit unsigned integer)""" @@ -66,7 +64,6 @@ def requires_bit_index(self) -> bool: def ctype_class(self) -> type: return ctypes.c_uint8 - class IntBufferType(IBufferType): """Integer buffer type (16-bit unsigned integer)""" @@ -90,7 +87,6 @@ def requires_bit_index(self) -> bool: def ctype_class(self) -> type: return ctypes.c_uint16 - class DintBufferType(IBufferType): """Double integer buffer type (32-bit unsigned integer)""" @@ -114,7 +110,6 @@ def requires_bit_index(self) -> bool: def ctype_class(self) -> type: return ctypes.c_uint32 - class LintBufferType(IBufferType): """Long integer buffer type (64-bit unsigned integer)""" @@ -138,7 +133,6 @@ def requires_bit_index(self) -> bool: def ctype_class(self) -> type: return ctypes.c_uint64 - class BufferTypes: """ Singleton registry of all buffer types. @@ -221,11 +215,9 @@ def validate_buffer_exists(self, buffer_name: str) -> bool: """Check if a buffer name exists""" return buffer_name in self._buffer_mappings - # Singleton instance _buffer_types_instance = None # pylint: disable=C0103 - def get_buffer_types() -> BufferTypes: """Get the singleton BufferTypes instance""" global _buffer_types_instance diff --git a/core/src/drivers/plugins/python/shared/buffer_validator.py b/core/src/drivers/plugins/python/shared/buffer_validator.py index b3c46b18..f83345cf 100644 --- a/core/src/drivers/plugins/python/shared/buffer_validator.py +++ b/core/src/drivers/plugins/python/shared/buffer_validator.py @@ -19,7 +19,6 @@ from component_interfaces import IBufferValidator from buffer_types import get_buffer_types - class BufferValidator(IBufferValidator): """ Centralized validation for buffer operations. diff --git a/core/src/drivers/plugins/python/shared/component_interfaces.py b/core/src/drivers/plugins/python/shared/component_interfaces.py index 60fbd004..40faf6cc 100644 --- a/core/src/drivers/plugins/python/shared/component_interfaces.py +++ b/core/src/drivers/plugins/python/shared/component_interfaces.py @@ -12,7 +12,6 @@ from typing import List, Dict, Tuple, Any, Optional import ctypes - class IBufferType: """Interface for buffer type definitions""" @@ -46,7 +45,6 @@ def ctype_class(self) -> type: """Corresponding ctypes class""" pass - class IMutexManager: """Interface for mutex management operations""" @@ -65,7 +63,6 @@ def with_mutex(self, operation: callable) -> Any: """Execute operation within mutex context. Returns operation result.""" pass - class IBufferValidator: """Interface for buffer validation operations""" @@ -90,7 +87,6 @@ def validate_operation_params(self, buffer_type: str, buffer_idx: int, """Comprehensive parameter validation. Returns (is_valid, error_message)""" pass - class IBufferAccessor: """Interface for generic buffer access operations""" @@ -111,7 +107,6 @@ def get_buffer_pointer(self, buffer_type: str) -> Optional[ctypes.POINTER]: """Get the buffer pointer for a given type. Returns None if invalid.""" pass - class IBatchProcessor: """Interface for batch operations""" @@ -131,7 +126,6 @@ def process_mixed_operations(self, read_operations: List[Tuple], """Process mixed read/write operations. Returns (results_dict, error_message)""" pass - class IConfigHandler: """Interface for configuration file operations""" @@ -145,7 +139,6 @@ def get_config_as_map(self) -> Tuple[Dict, str]: """Parse config file as key-value map. Returns (config_dict, error_message)""" pass - class ISafeBufferAccess: """Main interface that maintains API compatibility""" diff --git a/core/src/drivers/plugins/python/shared/config_handler.py b/core/src/drivers/plugins/python/shared/config_handler.py index 0eace75e..056a8e4d 100644 --- a/core/src/drivers/plugins/python/shared/config_handler.py +++ b/core/src/drivers/plugins/python/shared/config_handler.py @@ -18,7 +18,6 @@ # Fall back to absolute imports (when testing standalone) from component_interfaces import IConfigHandler - class ConfigHandler(IConfigHandler): """ Handles plugin-specific configuration file operations. diff --git a/core/src/drivers/plugins/python/shared/mutex_manager.py b/core/src/drivers/plugins/python/shared/mutex_manager.py index 5aa52dcd..f72c1fba 100644 --- a/core/src/drivers/plugins/python/shared/mutex_manager.py +++ b/core/src/drivers/plugins/python/shared/mutex_manager.py @@ -22,7 +22,6 @@ # Fall back to absolute imports (when testing standalone) from component_interfaces import IMutexManager - class MutexManager(IMutexManager): """ Manages mutex operations for thread-safe buffer access. diff --git a/core/src/drivers/plugins/python/shared/plugin_config_decode/modbus_master_config_model.py b/core/src/drivers/plugins/python/shared/plugin_config_decode/modbus_master_config_model.py index b9b4cb1f..83945d53 100644 --- a/core/src/drivers/plugins/python/shared/plugin_config_decode/modbus_master_config_model.py +++ b/core/src/drivers/plugins/python/shared/plugin_config_decode/modbus_master_config_model.py @@ -239,13 +239,10 @@ def validate(self) -> None: tcp_devices = [d for d in self.devices if d.transport == "tcp"] rtu_devices = [d for d in self.devices if d.transport == "rtu"] - # Multiple TCP devices may share the same host:port — that is exactly how - # an Ethernet-to-Modbus gateway (TCP-to-RTU converter) is addressed: one IP, - # several serial slaves distinguished by their unit/slave ID. So we key TCP - # devices by (host, port, slave_id) and reject only TRUE duplicates (same - # endpoint AND slave ID), which would be ambiguous. The runtime routes each - # device's slave_id into the MBAP unit-ID byte and shares one TCP connection - # per host:port (see group_tcp_devices_by_endpoint / ModbusBusHandler). + # Multiple TCP devices may share host:port (Modbus gateway), so + # the key is (host, port, slave_id). Reject only TRUE duplicates + # — same endpoint AND slave ID. Runtime routes slave_id into the + # MBAP unit-ID byte and shares one TCP connection per endpoint. tcp_endpoints = [(device.host, device.port, device.slave_id) for device in tcp_devices] if len(tcp_endpoints) != len(set(tcp_endpoints)): raise ValueError( @@ -271,7 +268,6 @@ def validate(self) -> None: def __repr__(self) -> str: return f"{self.__class__.__name__}(devices={len(self.devices)})" - # What a read point does with its IEC buffer while communication is down. # The editor exposes this per I/O group ("Keep last value" / "Set to zero"); # these are the strings it writes into the plugin config. @@ -279,7 +275,6 @@ def __repr__(self) -> str: ERROR_HANDLING_SET_TO_ZERO = "set-to-zero" ERROR_HANDLING_MODES = (ERROR_HANDLING_KEEP_LAST, ERROR_HANDLING_SET_TO_ZERO) - class ModbusIoPointConfig: """ Model for a single Modbus I/O point configuration. diff --git a/core/src/drivers/plugins/python/shared/plugin_config_decode/opcua_config_model.py b/core/src/drivers/plugins/python/shared/plugin_config_decode/opcua_config_model.py index 38c8f7a7..6e997ef3 100644 --- a/core/src/drivers/plugins/python/shared/plugin_config_decode/opcua_config_model.py +++ b/core/src/drivers/plugins/python/shared/plugin_config_decode/opcua_config_model.py @@ -15,15 +15,10 @@ # Permission types for variables PermissionType = Literal["r", "w", "rw"] -# opcua.json contract version this runtime understands. v2 introduced -# compiler-canonical per-leaf `datatype` + `size` (sourced from the same -# STruC++ compile that builds the .so) so the runtime encodes the exact -# byte width instead of re-deriving it from a drift-prone stored datatype. -# A config without `format_version` (or below this) is an older editor's -# output: we refuse it gracefully (OPC-UA stays down; the rest of the PLC -# runs) rather than risk writing the wrong number of bytes to a variable. -# Mirror of OPCUA_CONFIG_FORMAT_VERSION in openplc-editor's -# generate-opcua-config.ts. +# Minimum opcua.json contract version. v2 introduced compiler-canonical +# per-leaf datatype+size. Older configs are refused gracefully (OPC-UA +# stays down; the rest of the PLC runs). Mirror of +# OPCUA_CONFIG_FORMAT_VERSION in generate-opcua-config.ts. OPCUA_CONFIG_MIN_FORMAT_VERSION = 2 # Valid datatypes for OPC-UA variables (IEC 61131-3 base types) @@ -51,7 +46,6 @@ # truth for every role consumer (config parse here, user_manager, callbacks). VALID_ROLES = frozenset(["viewer", "operator", "engineer"]) - def normalize_role(role: Any) -> str: """Normalize any role value to one of 'viewer' | 'operator' | 'engineer'. @@ -72,7 +66,6 @@ def normalize_role(role: Any) -> str: return "viewer" return "viewer" - @dataclass class SecurityProfile: """Configuration for a security profile/endpoint.""" @@ -81,10 +74,8 @@ class SecurityProfile: security_policy: str security_mode: str auth_methods: List[str] - # Role granted to Anonymous sessions on this profile. Explicit rather than - # inferred: an anonymous client has no identity, so what it may do is stated - # here. Absent (projects authored before this field) -> least-privilege - # 'viewer'. Only meaningful when 'Anonymous' is in auth_methods. + # Role for Anonymous sessions on this profile. Absent → 'viewer'. + # Only meaningful when 'Anonymous' is in auth_methods. anonymous_role: str = "viewer" @classmethod @@ -99,12 +90,9 @@ def from_dict(cls, data: Dict[str, Any]) -> 'SecurityProfile': except KeyError as e: raise ValueError(f"Missing required field in security profile: {e}") - # Optional; default to viewer for backward compatibility. Validate at - # parse time (like VALID_DATATYPES) so a typo surfaces once at server - # start — where the admin sees it — instead of a per-session log_warn. - # Case/whitespace are normalized so "Engineer" or " engineer " are - # accepted; anything not a known role is rejected rather than silently - # degraded, since that role decides what an unauthenticated client may do. + # Optional; default viewer. Validated at parse so a typo + # surfaces once at server start, not per session. Normalized + # case/whitespace; unknown value is rejected, never degraded. raw_role = data.get("anonymous_role") if raw_role is None or raw_role == "": anonymous_role = "viewer" @@ -452,11 +440,9 @@ class OpcuaConfig: @classmethod def from_dict(cls, data: Dict[str, Any]) -> 'OpcuaConfig': """Creates an OpcuaConfig instance from a dictionary.""" - # Contract gate FIRST, before parsing variables: an older editor's - # config omits per-leaf `size`, so reject it with a clear message - # instead of a confusing "missing size" KeyError. Raising here makes - # load_config() return None -> the OPC-UA server simply doesn't start - # while the rest of the runtime keeps running. + # Contract gate first (before parsing variables) so an older + # editor's config fails with a clear message instead of a + # confusing KeyError on missing `size`. format_version = data.get("format_version", 0) if not isinstance(format_version, int) or format_version < OPCUA_CONFIG_MIN_FORMAT_VERSION: raise ValueError( diff --git a/core/src/drivers/plugins/python/shared/plugin_config_decode/plugin_config_contact.py b/core/src/drivers/plugins/python/shared/plugin_config_decode/plugin_config_contact.py index 31c8da4f..28db5a69 100644 --- a/core/src/drivers/plugins/python/shared/plugin_config_decode/plugin_config_contact.py +++ b/core/src/drivers/plugins/python/shared/plugin_config_decode/plugin_config_contact.py @@ -12,7 +12,6 @@ class PluginConfigError(Exception): """Custom exception for plugin configuration errors.""" pass - class PluginConfigContract(ABC): """ Abstract base class for protocol-specific configurations. diff --git a/core/src/drivers/plugins/python/shared/plugin_config_decode/test_modbus_master_config_model.py b/core/src/drivers/plugins/python/shared/plugin_config_decode/test_modbus_master_config_model.py index cb08a78a..31278e4e 100644 --- a/core/src/drivers/plugins/python/shared/plugin_config_decode/test_modbus_master_config_model.py +++ b/core/src/drivers/plugins/python/shared/plugin_config_decode/test_modbus_master_config_model.py @@ -36,7 +36,6 @@ def test_parse_iec_address_invalid(): with pytest.raises(ValueError): parse_iec_address("%QZ0") # Invalid type - # --------------------------------------------------------------------- # TEST ModbusIoPointConfig # --------------------------------------------------------------------- @@ -64,7 +63,6 @@ def test_modbus_io_point_error_handling_defaults_to_keep_last(): ) assert point.error_handling == ERROR_HANDLING_KEEP_LAST - def test_modbus_io_point_parses_set_to_zero(): point = ModbusIoPointConfig.from_dict( { @@ -78,7 +76,6 @@ def test_modbus_io_point_parses_set_to_zero(): assert point.error_handling == ERROR_HANDLING_SET_TO_ZERO assert point.to_dict()["error_handling"] == "set-to-zero" - def test_modbus_io_point_unknown_error_handling_falls_back(): # A config from a newer editor must not take the whole device offline over # one unrecognised string. @@ -93,7 +90,6 @@ def test_modbus_io_point_unknown_error_handling_falls_back(): ) assert point.error_handling == ERROR_HANDLING_KEEP_LAST - def test_modbus_io_point_missing_field(): data = { "offset": "40001", @@ -103,7 +99,6 @@ def test_modbus_io_point_missing_field(): with pytest.raises(ValueError): ModbusIoPointConfig.from_dict(data) - # --------------------------------------------------------------------- # TEST ModbusDeviceConfig # --------------------------------------------------------------------- @@ -137,7 +132,6 @@ def test_device_invalid_fc(): with pytest.raises(ValueError): dev.validate() - # --------------------------------------------------------------------- # TEST ModbusMasterConfig # --------------------------------------------------------------------- diff --git a/core/src/drivers/plugins/python/shared/plugin_config_decode/test_plugin_config_models.py b/core/src/drivers/plugins/python/shared/plugin_config_decode/test_plugin_config_models.py index 90b4a363..0857d907 100644 --- a/core/src/drivers/plugins/python/shared/plugin_config_decode/test_plugin_config_models.py +++ b/core/src/drivers/plugins/python/shared/plugin_config_decode/test_plugin_config_models.py @@ -197,7 +197,6 @@ def test_modbus_io_point_config_from_dict(): return True - def test_modbus_config_error_handling(): """Test ModbusMasterConfig error handling with invalid files or data.""" print("\n--- Testing ModbusMasterConfig Error Handling ---") diff --git a/core/src/drivers/plugins/python/shared/plugin_logger.py b/core/src/drivers/plugins/python/shared/plugin_logger.py index 14fc5543..400a5e3e 100644 --- a/core/src/drivers/plugins/python/shared/plugin_logger.py +++ b/core/src/drivers/plugins/python/shared/plugin_logger.py @@ -32,7 +32,6 @@ from typing import Optional from .safe_logging_access import SafeLoggingAccess - class PluginLogger: """ Thread-safe logger for OpenPLC plugins that routes messages to the diff --git a/core/src/drivers/plugins/python/shared/plugin_runtime_args.py b/core/src/drivers/plugins/python/shared/plugin_runtime_args.py index 64f01fce..bb973eb6 100644 --- a/core/src/drivers/plugins/python/shared/plugin_runtime_args.py +++ b/core/src/drivers/plugins/python/shared/plugin_runtime_args.py @@ -14,7 +14,6 @@ # Import IEC type definitions from .iec_types import IEC_BOOL, IEC_BYTE, IEC_UDINT, IEC_UINT, IEC_ULINT - class PluginRuntimeArgs(ctypes.Structure): """ Python ctypes structure matching plugin_runtime_args_t from plugin_driver.h @@ -46,13 +45,10 @@ class PluginRuntimeArgs(ctypes.Structure): # Both are void (*)(void). ("image_lock", ctypes.CFUNCTYPE(None)), ("image_unlock", ctypes.CFUNCTYPE(None)), - # STruC++ debugger variable-access surface. Replaces the - # MatIEC-era flat-index API (get_var_list/get_var_size/ - # get_var_count). Variables are addressed by (arr, elem); the - # editor resolves user-selected variables against debug-map.json - # and writes the tuples into each plugin's per-plugin config. + # STruC++ debugger variable-access surface. Variables addressed + # by (arr, elem), resolved by the editor against debug-map.json. # debug_set toggles forcing; debug_write does a soft write that - # respects existing forces (the next scan cycle can overwrite). + # respects existing forces (program can overwrite next scan). ("debug_array_count", ctypes.CFUNCTYPE(ctypes.c_uint8)), ("debug_elem_count", ctypes.CFUNCTYPE(ctypes.c_uint16, ctypes.c_uint8)), ("debug_size", ctypes.CFUNCTYPE(ctypes.c_uint16, ctypes.c_uint8, ctypes.c_uint16)), @@ -85,11 +81,7 @@ class PluginRuntimeArgs(ctypes.Structure): ("journal_write_dint", ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_int, ctypes.c_int, ctypes.c_uint)), ("journal_write_lint", ctypes.CFUNCTYPE(ctypes.c_int, ctypes.c_int, ctypes.c_int, ctypes.c_ulonglong)), # Async request to stop the whole PLC: void (*)(const char *reason). - # - # This entry was missing while the C struct had the field, so every - # field after it was shifted by one pointer -- `base_tick_ns` was - # actually reading the request_plc_stop pointer. Keep this list in - # lockstep with plugin_types.h; the offsets are load-bearing. + # Keep in lockstep with plugin_types.h: field offsets are load-bearing. ("request_plc_stop", ctypes.CFUNCTYPE(None, ctypes.c_char_p)), # PLC base tick time in nanoseconds (mirrors C-side base_tick_ns). ("base_tick_ns", ctypes.c_ulonglong), diff --git a/core/src/drivers/plugins/python/shared/safe_buffer_access_refactored.py b/core/src/drivers/plugins/python/shared/safe_buffer_access_refactored.py index 8317c2a8..38e233c4 100644 --- a/core/src/drivers/plugins/python/shared/safe_buffer_access_refactored.py +++ b/core/src/drivers/plugins/python/shared/safe_buffer_access_refactored.py @@ -31,7 +31,6 @@ from config_handler import ConfigHandler from mutex_manager import MutexManager - class SafeBufferAccess(ISafeBufferAccess): """ Refactored SafeBufferAccess with modular architecture. @@ -298,19 +297,6 @@ def batch_mixed_operations( """Process mixed read and write operations in batch.""" return self.batch_processor.process_mixed_operations(read_operations, write_operations) - # ============================================================================ - # Variable access (debug_*) — moved out of SafeBufferAccess. - # - # The MatIEC-era flat-index API (get_var_list / get_var_size / - # get_var_count / get_var_value / get_var_*_batch) is gone. Plugins - # that need to read/write program variables call the runtime's - # debug_read / debug_write / debug_set / debug_size function - # pointers directly via runtime_args.* — see - # opcua/opcua_memory.py for the typed Python helpers - # (debug_read_value, debug_write_value, debug_force_value, - # debug_unforce, initialize_variable_cache). - # ============================================================================ - # ============================================================================ # Configuration Operations # ============================================================================ diff --git a/core/src/drivers/vpp_plugin_seal.c b/core/src/drivers/vpp_plugin_seal.c index 8a379bff..60814fc7 100644 --- a/core/src/drivers/vpp_plugin_seal.c +++ b/core/src/drivers/vpp_plugin_seal.c @@ -225,12 +225,10 @@ int vpp_plugin_seal_required(const char *path) return 0; } - /* Compare RESOLVED paths, not the literal string: the config may say - * ./build/vpp/libx.so, build/vpp/libx.so, or an absolute path, and a - * symlink anywhere in the chain would defeat a textual match. When the - * file does not exist yet realpath fails -- fall back to a textual test so - * a missing object is still treated as VPP (and therefore still refused - * below) instead of being waved through as a built-in. */ + /* Compare RESOLVED paths so symlinks and relative vs absolute forms + * cannot defeat the match. If realpath fails (file does not exist + * yet), fall back to textual so a missing object is still treated + * as VPP and refused below rather than waved through. */ char resolved_path[PATH_MAX]; char resolved_vpp[PATH_MAX]; if (realpath(path, resolved_path) && realpath(VPP_BUILD_SUBDIR, resolved_vpp)) diff --git a/core/src/lib/strucpp_abi.hpp b/core/src/lib/strucpp_abi.hpp index 6deab613..bf71ec18 100644 --- a/core/src/lib/strucpp_abi.hpp +++ b/core/src/lib/strucpp_abi.hpp @@ -1,22 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// strucpp_abi.hpp — runtime-side mirror of the strucpp ABI we walk. -// -// The runtime executable is built ONCE; the .so it loads at runtime -// carries the actual strucpp runtime headers (shipped with the user -// program upload, used by scripts/compile.sh to build the .so). The -// runtime itself does NOT vendor strucpp headers — only this minimal -// set of layout-compatible mirror declarations. -// -// CONTRACT: every type below MUST match the layout strucpp's vendored -// headers expose. The .so's vtables, struct offsets, and enum values -// are all assumed identical. ABI consistency between the runtime and -// the strucpp version a user .so was built against is maintained as -// part of the development cycle — not enforced here. When strucpp's -// ABI version bumps in a breaking way, update this file. -// -// Mirrored from strucpp v0.4.5 (iec_located.hpp + iec_std_lib.hpp). +// Runtime-side mirror of the strucpp ABI the loaded .so exposes. +// Every type below MUST match that layout: vtables, struct offsets, +// enum values. Mirrored from strucpp v0.4.5; update on an ABI bump. #ifndef OPENPLC_STRUCPP_ABI_HPP #define OPENPLC_STRUCPP_ABI_HPP @@ -53,23 +40,6 @@ struct LocatedVar { void *pointer; }; -// --------------------------------------------------------------------------- -// ProgramBase (mirror of strucpp::ProgramBase, iec_std_lib.hpp) -// -// Polymorphic base. The runtime calls ->run() through a pointer; the -// vtable resolves into the .so's address space (where the actual -// derived class lives). Any extra virtual methods strucpp adds AFTER -// run() are fine — the runtime only calls run() so it doesn't need -// them in the mirror, but we keep them to preserve the vtable slot -// indices. -// -// strucpp v0.4.5 ProgramBase virtuals, in order: -// 0: ~ProgramBase() -// 1: run() -// 2: getRetainVars() const -// 3: getRetainCount() const -// --------------------------------------------------------------------------- - struct RetainVarInfo; // opaque; we never dereference struct ProgramBase { @@ -77,13 +47,9 @@ struct ProgramBase { virtual void run() = 0; virtual const RetainVarInfo *getRetainVars() const { return nullptr; } virtual size_t getRetainCount() const { return 0; } - // RESERVED vtable slots 4,5 (formerly sync_in / sync_out). The shared-global - // model moved from runtime-orchestrated per-task copy-in/out to per-global - // mutexes owned by strucpp's GlobalVar, so the runtime no longer calls - // these and current strucpp no longer overrides them. They are KEPT as - // no-op base slots — never renumber the vtable, or every program built - // against an older ABI would mis-dispatch run()/located_range() on a newer - // runtime (and vice versa). + // RESERVED vtable slots 4,5 (formerly sync_in/sync_out). Kept as + // no-op base slots: renumbering the vtable would mis-dispatch + // run()/located_range() across ABI versions in either direction. virtual void sync_in() {} virtual void sync_out() {} virtual void located_range(uint32_t *offset, uint32_t *count) const { @@ -115,19 +81,6 @@ struct ResourceInstance { size_t task_count; }; -// --------------------------------------------------------------------------- -// ConfigurationInstance (mirror of strucpp::ConfigurationInstance, -// iec_std_lib.hpp) -// -// Polymorphic. The runtime obtains a ConfigurationInstance* via the -// shim's strucpp_get_config() and walks resources/tasks/programs by -// virtual dispatch. vtable slots, in order: -// 0: ~ConfigurationInstance() -// 1: get_name() const -// 2: get_resources() -// 3: get_resource_count() const -// --------------------------------------------------------------------------- - struct ConfigurationInstance { virtual ~ConfigurationInstance() = default; virtual const char *get_name() const = 0; diff --git a/core/src/plc_app/client_tcp_udp.c b/core/src/plc_app/client_tcp_udp.c index 1850f0e2..22010d6d 100644 --- a/core/src/plc_app/client_tcp_udp.c +++ b/core/src/plc_app/client_tcp_udp.c @@ -1,10 +1,8 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// This is the file for the network routines of the OpenPLC. It has procedures -// to create a socket and connect to a server. These functions are called by -// the TCP communication function blocks (TCP_CONNECT, TCP_SEND, TCP_RECEIVE, -// TCP_CLOSE) defined in communication.h. +// Socket/connect helpers for the TCP/UDP communication function blocks +// (TCP_CONNECT, TCP_SEND, TCP_RECEIVE, TCP_CLOSE) in communication.h. #include #include diff --git a/core/src/plc_app/debug_handler.c b/core/src/plc_app/debug_handler.c index ded42758..f0666477 100644 --- a/core/src/plc_app/debug_handler.c +++ b/core/src/plc_app/debug_handler.c @@ -1,19 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * debug_handler.c — STruC++ hierarchical debugger PDU handler. - * - * The Modbus-style function codes (0x41-0x45) are kept for wire compatibility - * with the editor and the Arduino runtime. The payload format uses - * (array_idx: u8, elem_idx: u16) addressing — the editor's debug-map.json - * carries the path → (arr, elem) mapping. - * - * Mirrors the dispatch logic in resources/sources/StrucppBaremetal/ModbusSlave.cpp - * from the editor repo. Linux supports larger PDUs than RTU/Arduino; the cap - * here is the runtime-side MAX_DEBUG_FRAME, not the conservative 1400-byte - * limit the Arduino sketch uses. - */ +/* STruC++ debugger PDU handler. Codes 0x41-0x45 are Modbus-style for + * editor/Arduino wire compatibility. Addressing is (arr:u8, elem:u16); + * the editor's debug-map.json carries path→(arr, elem). */ #include @@ -55,11 +45,9 @@ static inline bool debug_symbols_ready(void) ext_strucpp_debug_read != NULL; } -/* Defense-in-depth bounds check on the array index that arrived over the - * wire. The editor's STruC++ codegen validates `arr` inside its debug - * thunks, but a malformed .so could OOB-read its internal table. Runtime - * gate first: reject `arr >= array_count` here so the .so only sees - * indices it claimed to support. */ +/* Defense-in-depth bounds check on the wire `arr` index. The .so's own + * thunks validate, but a malformed .so could OOB-read. Reject here so + * the .so only sees indices it claimed to support. */ static inline bool debug_arr_in_range(uint8_t arr) { return arr < ext_strucpp_debug_array_count(); @@ -154,10 +142,9 @@ static void debugSetTrace(uint8_t *frame, size_t *frame_len, size_t length) return; } - /* Do NOT poke the IECVar from this socket thread — that races the IEC task - * workers (OpenPLC bug #3). Enqueue the force/unforce; the dispatcher - * applies it at the no-task-running window (race-free, ~1 scan later). The - * editor polls continuously, so the small latency is invisible. */ + /* Enqueue, do NOT poke the IECVar: a socket-thread write races IEC + * task workers. The dispatcher applies at the no-task-running window + * (~1 scan later). */ uint8_t op = (force != 0) ? (uint8_t)DBGW_OP_FORCE : (uint8_t)DBGW_OP_UNFORCE; const uint8_t *vp = (force != 0) ? val_ptr : NULL; uint16_t vl = (force != 0) ? val_len : 0; @@ -320,23 +307,10 @@ static void debugGetTraceList(uint8_t *frame, size_t *frame_len, size_t length) *frame_len = HDR + response_sz; } -/* FC 0x45 — DEBUG_GET_MD5 - * - * The trailer carries a runtime-driven endianness sentinel, not an echo of - * what the editor sent. The variable-data path is pure memcpy on both - * sides (the strucpp dispatcher does no byte-order adaptation), so wire - * bytes for force / read are always in target-native order. To let the - * editor know what "native" means here, this handler writes the literal - * value 0xDEAD via a native uint16_t store; the two bytes that land in the - * frame reflect the target's byte order: - * - * LE target → trailer = [0xAD, 0xDE] - * BE target → trailer = [0xDE, 0xAD] - * - * The editor uses that to decide whether to byte-swap variable data at its - * end. The probe bytes in the request are ignored — the trailer is a - * sentinel, not an echo. - */ +/* FC 0x45 — DEBUG_GET_MD5. Trailer is an endianness sentinel: a + * native uint16_t store of 0xDEAD so wire bytes reveal target byte + * order to the editor (LE → [0xAD,0xDE], BE → [0xDE,0xAD]). The + * variable-data path is memcpy in target-native order. */ static void debugGetMd5(uint8_t *frame, size_t *frame_len, size_t length) { if (length < 3 || ext_strucpp_program_md5 == NULL) diff --git a/core/src/plc_app/debug_write_journal.cpp b/core/src/plc_app/debug_write_journal.cpp index e7b37a00..e2e98905 100644 --- a/core/src/plc_app/debug_write_journal.cpp +++ b/core/src/plc_app/debug_write_journal.cpp @@ -1,32 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * debug_write_journal.cpp — see debug_write_journal.h. - * - * A mutex-protected queue of external write/force requests, drained by the - * dispatcher at the no-task-running window. Producers (debugger socket - * thread, OPC-UA plugin thread) only enqueue; the single dispatcher consumer - * applies them while no IEC worker is mid-scan and while it holds the image - * lock, so no per-variable locking is needed and the application is race-free. - * - * Routing, decided per entry via ext_strucpp_debug_locate: - * - * - GLOBAL / program-internal leaf (or an older .so without the classifier): - * applied straight to the IECVar through the strucpp debug exports - * (ext_strucpp_debug_write / _set). This is the OPC-UA global-corruption / - * bug #3 case — the write that used to race the workers now lands here, - * serialized. - * - * - LOCATED variable (%I/%Q/%M): routed through the image journal and the - * forced-slot bitmap, because on v4 the image is decoupled from the IECVar - * (copy_in/copy_out) and a direct IECVar poke would be clobbered by the - * next copy_in. A WRITE becomes a journal_write_* (transient — program - * logic / the driver own the slot); a FORCE seeds + pins the slot via - * journal_force_set (drop-on-write keeps it pinned all cycle) AND forces - * the IECVar so the program's own view is forced too; UNFORCE reverses - * both. - */ +/* Queue of external write/force requests; dispatcher drains at the + * no-task-running window under image_lock. ext_strucpp_debug_locate + * routes each: GLOBAL → IECVar; LOCATED → journal_write_* / force. */ #include "debug_write_journal.h" #include "image_tables.h" /* ext_strucpp_debug_set / _write / _locate */ diff --git a/core/src/plc_app/debug_write_journal.h b/core/src/plc_app/debug_write_journal.h index 900b250a..19a34304 100644 --- a/core/src/plc_app/debug_write_journal.h +++ b/core/src/plc_app/debug_write_journal.h @@ -1,29 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * debug_write_journal.h — serialized external variable write/force path. - * - * The editor debugger and the OPC-UA plugin both need to write or force PLC - * variables addressed by the strucpp debug (arr, elem) namespace. Doing that - * straight from their own threads (the unix-socket thread, the OPC-UA asyncio - * thread) pokes the IECVar concurrently with the IEC task workers — a data - * race (OpenPLC bug #3) and the mechanism behind the OPC-UA global-corruption - * bug. - * - * This journal makes every external write/force an ENQUEUE from any thread - * (runtime_external_write), drained exactly once per cycle by the dispatcher - * at the no-task-running window. Because the dispatcher is the only consumer - * and it drains while no worker is mid-scan (and, for v4, while it owns the - * image double-buffer), the application is race-free without per-variable - * locking. The queue is mutex-protected (low rate — external writes are - * exceptional), and the drain has a lock-free skip-if-empty fast path so an - * idle cycle pays nothing. - * - * Routing of located variables (through the image journal + forced-slot - * bitmap) is layered on top of this; globals/program-internal leaves apply - * straight to the IECVar via the strucpp debug exports. - */ +/* Serialized external write/force path. Writers ENQUEUE from any + * thread via runtime_external_write; the dispatcher drains at the + * no-task-running window, so IECVars are never touched concurrently. */ #ifndef DEBUG_WRITE_JOURNAL_H #define DEBUG_WRITE_JOURNAL_H @@ -40,13 +20,10 @@ typedef enum { DBGW_OP_UNFORCE = 2 /* unforce — release a pinned variable */ } debug_write_op_t; -/* - * Enqueue an external write/force/unforce of the debug leaf (arr, elem). - * Safe to call from any thread (debugger socket thread, OPC-UA plugin - * thread). `bytes`/`len` carry the value payload for WRITE/FORCE (ignored - * for UNFORCE). Returns 0 on success, -1 if the queue is full (dropped + - * logged). The write is applied at the next dispatcher drain. - */ +/* Enqueue an external write/force/unforce of debug leaf (arr, elem). + * Thread-safe. bytes/len are the payload for WRITE/FORCE (ignored for + * UNFORCE). Returns 0, or -1 if full (dropped + logged). Applied at + * the next dispatcher drain. */ int runtime_external_write(uint8_t arr, uint16_t elem, uint8_t op, const uint8_t *bytes, uint16_t len); diff --git a/core/src/plc_app/image_tables.cpp b/core/src/plc_app/image_tables.cpp index d26427e9..940bcc09 100644 --- a/core/src/plc_app/image_tables.cpp +++ b/core/src/plc_app/image_tables.cpp @@ -1,12 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// image_tables.cpp -// -// Resolves the strucpp .so's exported symbols (configuration accessor, -// locks setter, debug PDU helpers) and walks strucpp::locatedVars[] to -// bind image-table buffer pointers. Plugins read/write through the -// buffer pointers directly under the image-tables mutex. +// Resolves the strucpp .so's exported symbols and binds image-table +// buffer pointers by walking strucpp::locatedVars[]. Plugins read/write +// through the buffer pointers under the image-tables mutex. #include #include @@ -19,11 +16,9 @@ extern "C" { #include "located_globals.h" } -// Layout-compatible mirror of the strucpp ABI. The runtime executable -// is built once and walks ConfigurationInstance / LocatedVar through -// these mirrors; the actual strucpp runtime headers ship with the user -// program upload (under core/generated/strucpp_runtime/include/) and -// are consumed only by scripts/compile.sh when building the .so. +// Layout-compatible mirror of the strucpp ABI. The real headers ship +// with each user program upload; the runtime walks ConfigurationInstance +// and LocatedVar through these mirrors. #include "../lib/strucpp_abi.hpp" #include "image_tables.h" @@ -55,17 +50,11 @@ IEC_UDINT *dint_memory[BUFFER_SIZE]; IEC_ULINT *lint_memory[BUFFER_SIZE]; IEC_BOOL *bool_memory[BUFFER_SIZE][8]; -// --------------------------------------------------------------------------- -// strucpp shim: per-project located-variable descriptor accessors -// (declared as C-linkage in the .so via runtime_v4_entry.cpp). -// --------------------------------------------------------------------------- namespace { using GetLocatedVarsFn = const strucpp::LocatedVar *(*)(void); using GetLocatedCountFn = uint32_t (*)(void); - // Located CONFIGURATION VAR_GLOBALs. strucpp emits these accessors beside - // the array in the generated configuration TU (not in our shim), so an older - // program simply does not export them -- see the OPTIONAL note in - // image_tables.h and the fallback in image_tables_bind_located_vars(). + // Located CONFIGURATION VAR_GLOBALs. Optional: older programs do + // not emit the accessors (see image_tables_bind_located_vars). using GetLocatedGlobalsFn = void *const *(*)(void); using GetLocatedGlobalCountFn = uint32_t (*)(void); @@ -113,21 +102,10 @@ namespace { uint64_t *g_located_snapshot = nullptr; uint32_t g_located_count = 0; - // Indices of the locatedVars[] entries that are CONFIGURATION VAR_GLOBAL - // ... AT. No program's located_range() covers them, so the per-task - // copy-in/out never touches them; the dispatcher copies them at the - // quiescent frame boundary instead (image_tables_copy_config_globals_in/out). - // - // Built once at program load by joining locatedVars[].pointer against the - // .so's locatedGlobals[] on POINTER IDENTITY. strucpp states which storage - // belongs to a configuration global; we never infer it. - // - // This deliberately replaces an earlier "[offset, count) tail not covered by - // any program range" slice. That assumed strucpp emitted program-local - // entries before config globals; the real order is the reverse, so the - // computed count collapsed to zero as soon as any POU declared a located - // variable and every located global silently stopped being serviced. Do not - // reintroduce any rule based on an entry's position in the array. + // Indices of VAR_GLOBAL ... AT entries in locatedVars[], built at + // load by joining against the .so's locatedGlobals[] on POINTER + // IDENTITY. Not covered by any program's located_range(); the + // dispatcher copies them at the quiescent frame boundary. uint32_t *g_located_globals_idx = nullptr; uint32_t g_located_globals_n = 0; @@ -135,12 +113,9 @@ namespace { { pthread_mutexattr_t attr; if (pthread_mutexattr_init(&attr) != 0) return -1; - // Priority inheritance is a POSIX optional feature. MSYS2/Cygwin - // pthread on Windows doesn't ship it (PTHREAD_PRIO_INHERIT is - // undefined, pthread_mutexattr_setprotocol is unavailable). - // Windows has no real-time scheduling anyway, so the PI protocol - // would be a no-op even if it linked — fall back to a plain - // recursive mutex. + // Priority inheritance is POSIX-optional; MSYS2/Cygwin lacks + // it. Windows has no real-time scheduling, so fall back to a + // plain recursive mutex. #if !defined(__CYGWIN__) && !defined(__MSYS__) && \ defined(_POSIX_THREAD_PRIO_INHERIT) && _POSIX_THREAD_PRIO_INHERIT > 0 pthread_mutexattr_setprotocol(&attr, PTHREAD_PRIO_INHERIT); @@ -151,13 +126,8 @@ namespace { return rc; } - /* Optional lookups go through the QUIET variant. plugin_manager_get_symbol - * reports every miss as "dlsym error", which is right for a symbol the - * runtime cannot work without and wrong for one it can: a program built by - * an editor older than a feature is correct, runs correctly, and used to - * announce itself with a burst of errors describing a healthy device as a - * broken one. Where absence is worth mentioning, the owning subsystem says - * so in its own words (see the located-globals warning below). */ + /* Optional lookups use the quiet variant so an older .so missing a + * newer symbol does not spam dlsym errors. */ void *resolve(PluginManager *pm, const char *name, bool required) { if (!required) @@ -178,23 +148,9 @@ extern "C" pthread_mutex_t *image_tables_mutex(void) return &g_image_tables_mutex; } -// Flush-on-lock read lock. This is the canonical entry for any consumer that -// needs a coherent view of the image to READ it (plugins reading %Q, the IEC -// task copy-in, etc.). It acquires the image mutex and then drains the journal -// so the holder sees every committed write. -// -// Usage mirrors the original BufferAccessor contract: -// - Individual read: image_lock(); v = ; image_unlock(); -// - Bulk read (preferred): image_lock(); ; -// image_unlock(); ; -// i.e. do the slow part (network, conversion) OUTSIDE the lock. -// -// Writes do NOT take this lock -- they go through journal_write_* (lock-free) -// and are applied by the drain here (or by the fastest task's drain). -// -// The mutex is recursive PI, so a consumer already holding it (e.g. the fastest -// task running plugin cycle hooks) can re-enter safely. The drain skips its -// bank flip when nothing is pending, so locking every cycle to read is cheap. +// Flush-on-lock read lock: takes the image mutex, drains the journal. +// Writes bypass this and go through journal_write_*. Mutex is recursive +// PI so a holder (e.g. plugin cycle hook) can re-enter. extern "C" void image_lock(void) { pthread_mutex_lock(&g_image_tables_mutex); @@ -254,12 +210,9 @@ extern "C" int symbols_init(PluginManager *pm) *(void **)&ext_strucpp_get_located_vars = resolve(pm, "strucpp_get_located_vars", true); *(void **)&ext_strucpp_get_located_var_count = resolve(pm, "strucpp_get_located_var_count", true); - /* Located CONFIGURATION VAR_GLOBALs — OPTIONAL. strucpp emits these - * accessors beside locatedGlobals[] in the generated configuration TU, so a - * program exported by an editor that predates them simply does not have the - * symbols. When absent the runtime cannot tell which located entries are - * config-scope, so it services none of them (the program still runs, and - * POU-local located I/O is unaffected) and warns once at load. */ + /* Config VAR_GLOBAL accessors are optional: older programs do not + * emit them; the runtime then skips config-scope servicing but the + * program (and POU-local located I/O) still runs. */ *(void **)&ext_strucpp_get_located_globals = resolve(pm, "strucpp_get_located_globals", false); *(void **)&ext_strucpp_get_located_global_count = @@ -294,12 +247,8 @@ extern "C" int symbols_init(PluginManager *pm) return -1; } - // The runtime compiles every program's .so itself, always with - // -DSTRUCPP_THREADED, so the only execution model is the threaded - // process-image one: per-task located copy-in/out for program-local - // `VAR AT`, dispatcher-boundary copy for config-scope located globals, and - // per-global mutexes (strucpp GlobalVar) for shared-global access. There - // is no legacy shared-image path and nothing to detect. + // Only execution model: threaded process-image (every .so is built + // with -DSTRUCPP_THREADED). log_info("[strucpp] execution model: threaded process-image"); if (!g_locks_initialized) @@ -356,22 +305,15 @@ void image_tables_bind_located_vars(void) uint32_t lv_count = ext_strucpp_get_located_var_count(); - // The runtime OWNS the image (temp_* backing buffers, installed by - // image_tables_fill_null_pointers) and copies image<->program storage per - // task. So we deliberately do NOT alias image slots to the .so located-var - // members here; we only size the dirty-diff snapshot buffer. + // Runtime owns the image; image<->program storage is copied per + // task. Do NOT alias image slots to .so located-var members here. g_located_count = lv_count; free(g_located_snapshot); g_located_snapshot = (uint64_t *)calloc(lv_count ? lv_count : 1, sizeof(uint64_t)); - // Build the config-scope index list. Authority is the .so's - // locatedGlobals[]: strucpp records there the canonical storage pointer of - // every located CONFIGURATION VAR_GLOBAL — the same raw_ptr() value it writes - // into locatedVars[].pointer — so an entry is config-scope exactly when its - // pointer appears in that array. Pointer identity, no layout or ordering - // assumption, and distinct objects have distinct addresses so there are no - // false positives. + // Build the config-scope index list by pointer identity between + // locatedVars[].pointer and locatedGlobals[]. No layout assumption. free(g_located_globals_idx); g_located_globals_idx = nullptr; g_located_globals_n = 0; @@ -467,18 +409,6 @@ void image_tables_bind_located_vars(void) lv_count, lv_count - g_located_globals_n, g_located_globals_n); } -// --------------------------------------------------------------------------- -// Threaded process-image copy-in / copy-out. -// -// In threaded mode the image (bool_input[] ... lint_memory[], backed by the -// temp_* buffers) is runtime-owned and decoupled from the program's located -// storage (the .so IECVar members, reachable via locatedVars[i].pointer). At a -// task boundary the runtime copies the task's located slice IN (image -> -// member) before run(), and commits CHANGED outputs OUT (member -> journal -> -// image) after. The journal makes the commit race-free vs other tasks/plugins; -// the snapshot makes it dirty (a task that only reads a shared output never -// clobbers a concurrent writer). -// --------------------------------------------------------------------------- namespace { uint64_t threaded_member_read(const strucpp::LocatedVar &v) @@ -612,20 +542,9 @@ extern "C" void image_tables_threaded_copy_out(uint32_t offset, uint32_t count) for (uint32_t k = offset; k < end; ++k) copy_out_one(lv, k); } -// Config-scope located globals (CONFIGURATION VAR_GLOBAL ... AT). No program's -// located_range() covers these, so the per-task copy-in/out never reaches them. -// The dispatcher calls these at the quiescent frame boundary -// (g_tasks_running == 0) so there is no concurrent task access to the shared -// canonical storage — the copy is safe WITHOUT the per-global mutex (quiescence -// is the synchronization). copy_in primes the canonical globals from the image -// (inputs get fresh hardware values); copy_out journals changed output/memory -// globals back to the image (drained by the dispatcher). -// -// The entries are an explicit index list built at load by -// image_tables_bind_located_vars() from the .so's locatedGlobals[]; they are NOT -// a contiguous slice, so these walk the list rather than calling the range-based -// helpers above. No-ops when the program has no located globals, or when it -// predates locatedGlobals[] (a warning is logged once at load). +// Config-scope VAR_GLOBAL AT. Not covered by any program's +// located_range(), so copies are driven here at the dispatcher's +// quiescent frame boundary (safe without the per-global mutex). extern "C" void image_tables_copy_config_globals_in(void) { if (!g_located_globals_n || !g_located_globals_idx) return; diff --git a/core/src/plc_app/image_tables.h b/core/src/plc_app/image_tables.h index 1df8741f..d52ced0c 100644 --- a/core/src/plc_app/image_tables.h +++ b/core/src/plc_app/image_tables.h @@ -19,15 +19,6 @@ extern "C" #define BUFFER_SIZE 1024 #define libplc_build_dir "./build" - /* ------------------------------------------------------------------------- - * Image-table buffers (booleans, bytes, ints, dints, lints, memories). - * - * Populated at program-load time by image_tables_bind_located_vars(), - * which walks strucpp::locatedVars[] and points each slot at the - * matching IECVar's underlying primitive storage. Plugins read/write - * these directly under the image-tables mutex. - * --------------------------------------------------------------------- */ - extern IEC_BOOL *bool_input[BUFFER_SIZE][8]; extern IEC_BOOL *bool_output[BUFFER_SIZE][8]; @@ -48,18 +39,6 @@ extern "C" extern IEC_ULINT *lint_memory[BUFFER_SIZE]; extern IEC_BOOL *bool_memory[BUFFER_SIZE][8]; - /* ------------------------------------------------------------------------- - * Resolved .so symbols (populated by symbols_init). - * - * strucpp_set_current_time sets the per-.so __CURRENT_TIME_NS (thread_local - * under STRUCPP_THREADED) for the calling thread; the GCD master-tick - * dispatcher stamps each task's dispatch time and the worker thread calls - * this at the top of its scan so IEC TIME() is stable within a scan. - * strucpp_advance_time is retained for compatibility (unused by the - * dispatcher). base_tick_ns is owned runtime-side (utils.c) and computed in - * symbols_init by walking the loaded configuration. - * --------------------------------------------------------------------- */ - extern void (*ext_strucpp_advance_time)(uint64_t tick_ns); /* Sets IEC TIME() for the CALLING thread. Call on the worker thread at the * top of its scan with the dispatch-stamped time. */ @@ -76,31 +55,15 @@ extern "C" uint16_t len); extern uint16_t (*ext_strucpp_debug_read) (uint8_t arr, uint16_t elem, uint8_t *dest); - /* Soft write — updates the variable's underlying value via - * IECVar::set(). If the variable is currently forced, the write is - * silently ignored (force remains authoritative). Distinct from - * ext_strucpp_debug_set(forcing=true) which pins the value - * indefinitely. Used by plugins (OPC-UA, BACnet) that want regular - * write semantics rather than debugger-style forcing. */ + /* Soft write via IECVar::set(). Silently ignored if the variable + * is forced (force wins). Distinct from debug_set(forcing=true). */ extern uint8_t (*ext_strucpp_debug_write) (uint8_t arr, uint16_t elem, const uint8_t *bytes, uint16_t len); - /* ---- Retain marshalling (NODE-94) -------------------------------------- - * - * The WALK lives inside the .so, not here: that is where the debug tables - * and handle_read/handle_write are, and re-implementing the blob format on - * this side would put two copies of one wire format in two repos. - * - * `unpack` takes a write CALLBACK because the runtime owns the write path. - * A retained variable may also be LOCATED (`VAR RETAIN x AT %MW10`), and - * poking such a leaf's IECVar is undone by the next copy-in from the - * process image — so the callback we hand over routes through - * runtime_external_write, which knows to send a located leaf through the - * image journal. - * - * Optional: a program built by an older STruC++ resolves these to NULL and - * the retain path simply never runs. */ + /* Retain marshalling. Walk lives in the .so (where the debug + * tables are). unpack takes a write callback so a LOCATED retained + * leaf can be routed through the image journal. Optional. */ extern size_t (*ext_strucpp_retain_blob_size) (void); extern uint32_t (*ext_strucpp_retain_layout_hash)(void); extern size_t (*ext_strucpp_retain_pack) (uint8_t *out, size_t cap); @@ -108,49 +71,24 @@ extern "C" uint8_t (*write_leaf)(uint8_t, uint16_t, const uint8_t *, uint16_t)); - /* Located-variable classifier. Reports whether a debug (arr, elem) leaf is - * a LOCATED variable and, if so, its image location (area / size / - * byte_index / bit_index). Returns 1 + fills the out-params if located, 0 - * otherwise. The debug-write drain uses it to route located writes/forces - * through the image journal + forced-slot bitmap (copy_in would clobber a - * direct IECVar poke). OPTIONAL: an older .so without it leaves the pointer - * NULL, and the drain treats every leaf as a global (IECVar) write. */ + /* Classifier: returns 1 and fills out-params when (arr, elem) is a + * LOCATED leaf; 0 otherwise. The debug-write drain uses this to + * route located writes through the image journal. Optional. */ extern int (*ext_strucpp_debug_locate) (uint8_t arr, uint16_t elem, uint8_t *area, uint8_t *size, uint16_t *byte_index, uint8_t *bit_index); - /* ------------------------------------------------------------------------- - * Symbol resolution. - * - * Resolves all required entry points from the dlopen'd .so, including - * the strucpp shim entry (strucpp_get_config) the runtime needs to walk - * the configuration. Initializes the runtime-owned image-tables mutex - * (recursive PI) on first call. The mutex is locked by the runtime - * directly; it is not handed to the .so. - * - * Returns 0 on success, -1 if anything required is missing. - * --------------------------------------------------------------------- */ int symbols_init(PluginManager *pm); - /* ------------------------------------------------------------------------- - * Walk strucpp::locatedVars[] and point each image-table slot at the - * corresponding IECVar's underlying primitive storage. Caller must hold - * the image-tables mutex. - * --------------------------------------------------------------------- */ + /* Caller must hold the image-tables mutex. */ void image_tables_bind_located_vars(void); - /* ------------------------------------------------------------------------- - * After binding, fill any unbound image-table slots with private - * backing buffers so plugins reading those addresses don't dereference - * NULL. Caller must hold the image-tables mutex. - * --------------------------------------------------------------------- */ + /* Caller must hold the image-tables mutex. */ void image_tables_fill_null_pointers(void); - /* ------------------------------------------------------------------------- - * Reset all image-table pointers to NULL before unloading a program. - * Caller must hold the image-tables mutex. - * --------------------------------------------------------------------- */ + /* Reset all image-table pointers to NULL before unloading a program. + * Caller must hold the image-tables mutex. */ void image_tables_clear_null_pointers(void); /** @@ -162,62 +100,20 @@ extern "C" */ void image_tables_zero_outputs(void); - /* ------------------------------------------------------------------------- - * Image-tables mutex accessor. Returns a pointer to the runtime-owned - * recursive PI mutex that protects the image tables. The runtime locks - * it directly; the .so never locks anything (generated code runs on its - * own storage), so there is no lock handoff into the .so. - * --------------------------------------------------------------------- */ pthread_mutex_t *image_tables_mutex(void); - /* ------------------------------------------------------------------------- - * Flush-on-lock read API. The canonical way for any consumer to obtain a - * coherent view of the image for READING: - * image_lock() -- take the image mutex, then drain the journal so the - * holder sees every committed write. - * image_unlock() -- release the image mutex. - * Writes never use this lock (they go through journal_write_*). Prefer the - * bulk pattern: lock, copy the region to a local buffer, unlock, then do - * any slow work (network, conversion) on the buffer outside the lock. The - * mutex is recursive PI; the drain is a no-op when nothing is pending. - * --------------------------------------------------------------------- */ void image_lock(void); void image_unlock(void); - /* ------------------------------------------------------------------------- - * Threaded process-image model (the only execution model — the runtime - * compiles every .so itself with -DSTRUCPP_THREADED). - * - * The copy_in/out functions move a located slice [offset, offset+count) of - * locatedVars[] between the runtime-owned image and the program's storage: - * - copy_in : image -> members (called before run(), under the image - * mutex, after the journal drain). - * - copy_out : changed members -> journal (dirty-diff, lock-free; applied - * to the image on the next drain). %I is never committed. - * - * Shared globals no longer use a runtime-owned mutex — each carries its own - * std::mutex inside the .so (strucpp GlobalVar), taken per access in - * run(). The config-scope helpers copy the located CONFIGURATION VAR_GLOBALs - * at the dispatcher's quiescent frame boundary. Those entries are identified - * at load by joining locatedVars[].pointer against the .so's locatedGlobals[] - * on pointer identity — NOT by their position in locatedVars[], and NOT as a - * contiguous slice (see image_tables.cpp). - * - * locatedGlobals[] is OPTIONAL: a program exported before STruC++ emitted it - * does not export the accessors, in which case located configuration globals - * are not synced at all (the program still runs, POU-local located I/O is - * unaffected) and a warning is logged once at load. - * --------------------------------------------------------------------- */ + /* copy_in runs before run(), under the image mutex, after the journal drain. + * copy_out publishes changed slots through the lock-free journal and never + * commits %I. config_globals_* mirror the same ordering for CONFIG globals. */ void image_tables_threaded_copy_in(uint32_t offset, uint32_t count); void image_tables_threaded_copy_out(uint32_t offset, uint32_t count); void image_tables_copy_config_globals_in(void); void image_tables_copy_config_globals_out(void); - /* ------------------------------------------------------------------------- - * Returns the cached strucpp::ConfigurationInstance* (as void* — the - * runtime's .cpp callers static_cast to the right type). NULL until - * symbols_init() succeeds; reset to NULL on image_tables_clear_null_pointers(). - * --------------------------------------------------------------------- */ + /* NULL until symbols_init() succeeds; cleared by image_tables_clear_null_pointers(). */ void *strucpp_config_handle(void); #ifdef __cplusplus diff --git a/core/src/plc_app/journal_buffer.c b/core/src/plc_app/journal_buffer.c index 7862925a..23763efb 100644 --- a/core/src/plc_app/journal_buffer.c +++ b/core/src/plc_app/journal_buffer.c @@ -61,23 +61,10 @@ static int journal_add(uint8_t type, uint16_t index, uint8_t bit, uint64_t value * ============================================================================= */ -/* --------------------------------------------------------------------------- - * Forced-slot bitmap. - * - * A located variable that the debugger / OPC-UA has FORCED must keep its - * forced value in the image regardless of what plugins (or the program's own - * copy_out) write to that slot. journal_force_set() seeds the slot with the - * forced value and marks it; every subsequent journal write to a forced slot - * is then DROPPED at apply time, so the force wins 100% of the cycle — the - * proper "force locks out external writes" semantic. (For globals/internals - * forcing lives in the IECVar; this bitmap is the located/image leg.) - * - * Mutated only from the dispatcher's debug-write drain and read only from - * apply_entry() — both under image_lock — so no atomics are required. - * JBUF_FORCE_SIZE mirrors the image BUFFER_SIZE; a runtime guard keeps this - * safe even if the two ever diverge. - * --------------------------------------------------------------------------- */ +/* Mirrors the image's BUFFER_SIZE; divergence trips the runtime guard. */ #define JBUF_FORCE_SIZE 1024 +/* Written only by journal_force_set/clear from the dispatcher drain and + * read only by apply_entry() — both under image_lock, so no atomics. */ static uint8_t g_forced[JOURNAL_TYPE_COUNT][JBUF_FORCE_SIZE]; static int g_force_count = 0; @@ -109,16 +96,10 @@ static void apply_write_raw(const journal_entry_t *entry) return; } - /* bit_index is only meaningful for the three BOOL cases, where it indexes - * the inner [8] dimension of the bool_* pointer rows. A non-bool write sets - * the 0xFF sentinel (journal_write_byte/int/dint/lint), and the lock-free - * path can hand the consumer a torn or stale-recycled slot whose - * buffer_type reads as BOOL while bit_index carries that sentinel. An - * unchecked bool_*[idx][0xFF] reads a pointer 247 slots past the row, - * harvesting a wild pointer that the store below would write through -- - * corrupting unrelated storage (observed: VAR_GLOBALs in the .so). Reject - * any bool entry whose bit_index is out of range so a torn/stale entry can - * never escalate into an out-of-bounds pointer write. */ + /* bit_index is only meaningful for BOOL. A torn/stale slot can read + * as BOOL with the 0xFF non-bool sentinel; an unchecked + * bool_*[idx][0xFF] would dereference a wild pointer 247 slots past + * the row and corrupt unrelated storage. Reject out-of-range bits. */ if ((entry->buffer_type == JOURNAL_BOOL_INPUT || entry->buffer_type == JOURNAL_BOOL_OUTPUT || entry->buffer_type == JOURNAL_BOOL_MEMORY) && @@ -202,10 +183,9 @@ static void apply_write_raw(const journal_entry_t *entry) } } -/* Apply one drained journal entry, honoring the forced-slot bitmap: a write - * to a forced slot is dropped so the force owns the slot for the whole cycle. - * (copy_out's journal writes and plugin journal writes both flow through here, - * so a forced located output stays pinned no matter who writes it.) */ +/* Apply one drained entry, honoring the forced-slot bitmap: a write to + * a forced slot is dropped so the force owns the slot for the whole + * cycle. All journal writes (copy_out and plugins) flow through here. */ static void apply_entry(const journal_entry_t *entry) { if (is_slot_forced(entry->buffer_type, entry->index, entry->bit_index)) { @@ -214,10 +194,9 @@ static void apply_entry(const journal_entry_t *entry) apply_write_raw(entry); } -/* Pin an image slot to `value` and mark it forced. Seeds the slot immediately - * (bypassing the drop), then every later journal write to it is dropped until - * journal_force_clear. Called only from the dispatcher's debug-write drain, - * under image_lock — the same serialization domain as apply_entry. */ +/* Pin an image slot to `value` and mark it forced. Seeds the slot + * immediately (bypass drop); later writes to it are dropped until + * journal_force_clear. Called under image_lock from the debug drain. */ void journal_force_set(journal_buffer_type_t type, uint16_t index, uint8_t bit, uint64_t value) { @@ -264,27 +243,6 @@ void journal_force_clear(journal_buffer_type_t type, uint16_t index, uint8_t bit #if JOURNAL_LOCKFREE -/* - * ============================================================================= - * Lock-free double-buffer-flip implementation (MPSC) - * ============================================================================= - * - * Control word layout (32-bit): - * bit 31 : active bank index (0 or 1) - * bits 0..30 : write count claimed in the active bank this cycle - * - * Producer: fetch_add(1) atomically claims (bank, slot). It writes the entry - * (plain stores) then publishes with a release store on the per-slot flag. - * - * Consumer (single, under image mutex): one atomic exchange flips the active - * bank and resets the count; the returned old value gives the retired bank and - * its final count. The consumer drains [0, count), acquiring each publish flag - * (a bounded wait covers a producer caught mid-write at the instant of flip). - * - * Only the consumer ever changes the bank bit, and the runtime calls the - * consumer from a single thread, so read-active-then-exchange is race-free. - */ - #define JOURNAL_NBANKS 2 #define JOURNAL_BANK_SHIFT 31u #define JOURNAL_COUNT_MASK 0x7FFFFFFFu @@ -378,12 +336,9 @@ void journal_apply_and_clear(void) return; } - /* Fast path: nothing pending -> nothing to apply, so skip the bank flip. - * The image already reflects every committed write. A producer that adds an - * entry after this load is simply applied on the next drain (one cycle - * later) -- the same ordering guarantee a flush-on-lock read offers. This - * keeps a read-heavy plugin (locking every cycle to read %Q via image_lock) - * from flipping the journal needlessly and racing producers mid-publish. */ + /* Fast path: nothing pending → skip the bank flip. A producer that + * adds an entry after this load gets applied on the next drain. Keeps + * a read-heavy plugin from flipping banks on every image_lock. */ if ((atomic_load_explicit(&g_control, memory_order_relaxed) & JOURNAL_COUNT_MASK) == 0) { return; } diff --git a/core/src/plc_app/located_globals.c b/core/src/plc_app/located_globals.c index 8dbc1ef5..473d38c3 100644 --- a/core/src/plc_app/located_globals.c +++ b/core/src/plc_app/located_globals.c @@ -1,9 +1,7 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* ----------------------------------------------------------------------------- - * located_globals.c — see located_globals.h for the rationale. - * -------------------------------------------------------------------------- */ +/* Joins locatedVars[] with locatedGlobals[] by storage-pointer identity. */ #include @@ -39,10 +37,9 @@ uint32_t located_globals_join_ex(uint32_t lv_count, } } - /* Report how many locatedGlobals[] entries found a home, so the caller can - * detect the two generated arrays disagreeing. Counted separately because a - * single global could in principle be referenced by more than one located - * descriptor (aliasing), which would make the index count misleading. */ + /* Report how many locatedGlobals[] entries matched so the caller + * can detect the two generated arrays disagreeing. Counted separately + * because a single global may be aliased by multiple descriptors. */ if (out_matched) { uint32_t matched = 0; diff --git a/core/src/plc_app/located_globals.h b/core/src/plc_app/located_globals.h index 6ae93e17..258ec06c 100644 --- a/core/src/plc_app/located_globals.h +++ b/core/src/plc_app/located_globals.h @@ -1,37 +1,6 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* ----------------------------------------------------------------------------- - * located_globals.h — resolve which locatedVars[] entries are CONFIGURATION - * VAR_GLOBAL ... AT, by joining against the .so's locatedGlobals[] array. - * - * Pure logic with no runtime dependencies, so it can be unit-tested directly - * (tests/test_located_globals.c). - * - * Why a join, and why this module exists at all: - * - * locatedVars[] mixes two ownership classes. POU-local `VAR ... AT` is serviced - * by the owning IEC task around run(); CONFIGURATION `VAR_GLOBAL ... AT` is - * owned by no task, so the dispatcher copies it at the quiescent frame boundary. - * The LocatedVar descriptors carry no scope, and the runtime cannot recover it - * from the configuration tree — globals are file-scope singletons that appear - * nowhere in ConfigurationInstance -> Resource -> Task -> Program. - * - * The runtime used to INFER the split from position, assuming the layout was - * [program-local ...][config globals ...] and treating the tail uncovered by any - * program's located_range() as the globals. strucpp actually emits config - * globals FIRST, so the program-local block is always the tail: one POU-local - * located variable made the computed count collapse to zero and every located - * global silently stopped being serviced (forum: "%MX locations now invalid"). - * - * strucpp now states the answer. locatedGlobals[] holds the canonical storage - * pointer of each located VAR_GLOBAL — the same raw_ptr() value written into - * locatedVars[].pointer — so an entry is config-scope exactly when its pointer - * appears in that array. Pointer identity: no ordering rule, no layout - * assumption, and no false positives, since distinct objects have distinct - * addresses. - * -------------------------------------------------------------------------- */ - #ifndef OPENPLC_LOCATED_GLOBALS_H #define OPENPLC_LOCATED_GLOBALS_H @@ -47,25 +16,21 @@ extern "C" { typedef const void *(*located_pointer_at_fn)(const void *located_vars, uint32_t index); -/* ----------------------------------------------------------------------------- - * Join locatedVars[] against locatedGlobals[] and collect the config-scope - * indices, ascending. - * - * lv_count number of locatedVars[] entries - * located_vars opaque handle passed back to pointer_at - * pointer_at accessor yielding locatedVars[i]'s storage pointer - * globals locatedGlobals[] — storage pointers of the located globals - * globals_count number of entries in globals - * out_idx receives the config-scope indices; room for lv_count - * out_matched optional; receives how many `globals` entries matched some - * located variable. A value below globals_count means the two - * generated arrays disagree, which the caller should report. - * - * Returns the number of indices written to out_idx. - * - * Entries with a NULL pointer are skipped: an unbound descriptor cannot be - * matched, and guessing a scope for it is exactly the failure this replaces. - * -------------------------------------------------------------------------- */ +/** + * @brief Join located-variable descriptors with their global pointers. + * + * @param lv_count entries in located_vars[]. + * @param located_vars opaque descriptor array; pointer_at() dereferences it. + * @param pointer_at accessor returning each descriptor's storage pointer. + * @param globals array of global storage pointers to probe. + * @param globals_count entries in globals[]. + * @param out_idx caller-provided buffer, room for lv_count entries; + * filled with the located_vars index each global binds to. + * @param out_matched out: number of locatedGlobals[] entries that found a + * descriptor; a value less than globals_count means the + * two generated arrays disagree. + * @return number of entries written to out_idx. + */ uint32_t located_globals_join_ex(uint32_t lv_count, const void *located_vars, located_pointer_at_fn pointer_at, diff --git a/core/src/plc_app/plc_main.c b/core/src/plc_app/plc_main.c index 97b6e7f7..18e42616 100644 --- a/core/src/plc_app/plc_main.c +++ b/core/src/plc_app/plc_main.c @@ -41,11 +41,9 @@ void handle_shutdown_signal(int sig) keep_running = 0; } -/* Process-wide no-op SIGUSR1 handler. The wake mechanism is EINTR - * delivery to a specific thread via pthread_kill(target, SIGUSR1) — the - * handler body itself does nothing. Installed exactly once at startup - * (instead of being re-installed by every thread that wants to be - * woken) so that handlers can never clobber each other. */ +/* No-op SIGUSR1 handler. Wake mechanism is pthread_kill delivering + * EINTR to a target thread; the body does nothing. Installed once so + * concurrent callers cannot overwrite each other's handler. */ static void handle_sigusr1(int sig) { (void)sig; @@ -96,16 +94,9 @@ int main(int argc, char *argv[]) return -1; } - // Handle SIGINT and SIGTERM for graceful shutdown. - // - // SIGTERM matters as much as SIGINT and was missing: RuntimeManager stops the - // runtime with process.terminate() (SIGTERM) and only escalates to - // process.kill() after a 5 s wait, and systemd's default KillSignal is also - // SIGTERM. With no handler installed the default disposition applied, so every - // supervisor-initiated stop killed the process outright -- the PLC program was - // never unloaded, plugins never stopped, the journal never flushed, and the - // program .so never dlclose'd. The escalation to SIGKILL is the backstop for a - // shutdown that hangs, not the normal path. + // Handle SIGINT and SIGTERM: RuntimeManager and systemd both send + // SIGTERM for a graceful stop; without a handler the default + // disposition kills the process and skips program unload. struct sigaction sa; sa.sa_handler = handle_shutdown_signal; sigemptyset(&sa.sa_mask); @@ -113,14 +104,8 @@ int main(int argc, char *argv[]) sigaction(SIGINT, &sa, NULL); sigaction(SIGTERM, &sa, NULL); - // Install the process-wide SIGUSR1 wake handler exactly once. Task - // threads (plc_state_manager.cpp) and EtherCAT bus threads - // (ethercat_plugin.c) both rely on EINTR-from-pthread_kill to break - // out of clock_nanosleep on stop. Previously each of those callers - // re-installed sigaction(SIGUSR1, …) on its own — last writer wins, - // and any future divergence between handlers (e.g. one logs, the - // other resets state) would silently lose half the time depending - // on thread spawn order. One install, here, eliminates the race. + // Install the process-wide SIGUSR1 wake handler exactly once; task + // and bus threads use pthread_kill to EINTR out of clock_nanosleep. struct sigaction wake_sa; wake_sa.sa_handler = handle_sigusr1; sigemptyset(&wake_sa.sa_mask); @@ -161,20 +146,10 @@ int main(int argc, char *argv[]) return -1; } - // Initialize plugin driver system BEFORE loading the PLC program. - // plc_set_state(RUNNING) triggers load_plc_program() which uses the plugin - // driver to update config and re-init plugins, and plc_cycle_thread() calls - // plugin_driver_start() after image tables are populated. - // - // This block runs BEFORE the command socket exists, deliberately. A START - // accepted while it is still executing runs load_plc_program() -> - // plugin_driver_init() on the transition worker at the same time as this - // thread is inside plugin_driver_load_config()/plugin_driver_init(), and - // rebuilding a plugin slot dlcloses the .so -- so a plugin sleeping in its - // own init() on the other thread returns into an unmapped page. Observed as a - // SIGSEGV in the main thread at an address inside the plugin that had just - // been unloaded. Transition arbitration cannot help here: there is only one - // transition, racing driver setup rather than another transition. + // Initialize the plugin driver BEFORE loading the PLC program. + // Must also run before the command socket exists: a concurrent START + // would race load_config()/init() and could dlclose a .so a plugin + // is still sleeping in. plugin_driver = plugin_driver_create(); if (plugin_driver) { @@ -191,14 +166,9 @@ int main(int argc, char *argv[]) log_error("[PLUGIN]: Failed to load plugin configuration"); } - // Release the Python GIL if Python was initialized during plugin loading. - // This prevents a deadlock where the main thread holds the GIL forever - // while sleeping, blocking other threads (like the unix socket thread) - // from using Python when handling commands like START. - // - // Through the driver rather than PyEval_SaveThread() directly: the driver - // has to restore this exact thread state before Py_FinalizeEx() at - // shutdown, and the state was previously discarded here. + // Release the GIL through the driver (not PyEval_SaveThread + // directly) so plugin_driver_destroy can restore this exact + // thread state before Py_FinalizeEx at shutdown. if (Py_IsInitialized()) { plugin_driver_release_gil(); @@ -231,37 +201,18 @@ int main(int argc, char *argv[]) } } - // Start the command socket only now that the plugin driver is fully built. - // Everything the socket can ask for -- START, STOP, PLUGIN_CMD, STATS -- - // reaches into the driver, so serving commands before this point was serving - // them against a half-configured one. The webserver already tolerates the - // socket appearing a moment later: it connects, retries, and polls. + // Start the socket only now that the driver is fully built: every + // socket command reaches into the driver. if (setup_unix_socket() != 0) { log_error("Failed to set up UNIX socket"); return -1; } - // Start PLC (skip in safe mode to allow program upload without loading the - // faulty program that may have caused repeated crashes). - // - // Use plc_begin_transition() rather than plc_set_state() directly so the - // auto-start is arbitrated by plc_claim_transition() like any - // socket-originated START command. This prevents a race where the socket - // listener (started just above) accepts a START before the auto-start - // finishes, causing two concurrent load_plc_program() calls — and two - // dispatcher threads. plc_begin_transition() also makes the start - // asynchronous, which is fine: the main thread just sleeps below. - // Same gate as any other start, but note what it can and cannot see. A VPP - // plugin that owns a physical mode switch is initialised as part of loading - // the program — inside the start transition below — so at this point the - // switch has usually NOT been reported yet and the gate reads the default - // (RUN). A device powered up with the switch in STOP therefore does start, - // and is then corrected: the plugin reports STOP during init, that request is - // dropped because a transition is in flight, and the switch-movement - // reconciliation stops the PLC as soon as the start lands. Safe, but the gate - // only bites here for a switch position already known at this point (e.g. one - // reported by a plugin the runtime loaded independently of the program). + // Auto-start (skipped in safe mode). The switch gate below reads + // the default (RUN) when no plugin has reported yet — a VPP that + // owns the switch is initialised DURING the start transition, so + // a device in STOP will start and movement-reconciliation stops it. if (!safe_mode && !plc_switch_allows_run()) { log_info("Hardware mode switch is in STOP - PLC left stopped"); @@ -280,13 +231,8 @@ int main(int argc, char *argv[]) log_info("Shutting down..."); - // Stop the program BEFORE destroying the driver, not after. - // - // plc_state_manager_cleanup() tears the program down, and its teardown calls - // plugin_driver_stop(plugin_driver) -- so destroying the driver first left that - // call reading freed memory, and the cycle thread could still be running plugin - // cycle hooks through the same pointer while it wound down. The order here is - // the dependency order: no program, then no driver. + // Order matters: program (which calls plugin_driver_stop) must + // tear down BEFORE the driver is destroyed. plc_state_manager_cleanup(); if (plugin_driver) diff --git a/core/src/plc_app/plc_retain.cpp b/core/src/plc_app/plc_retain.cpp index 281a6eab..cbc49d48 100644 --- a/core/src/plc_app/plc_retain.cpp +++ b/core/src/plc_app/plc_retain.cpp @@ -3,7 +3,7 @@ /** * @file plc_retain.cpp - * @brief Retain-variable persistence — the runtime's half (NODE-94). + * @brief Retain-variable persistence — the runtime's half. * * See plc_retain.h for the split: the .so marshals, a plugin stores, and this * file owns the buffer and the call sites. @@ -29,48 +29,23 @@ extern plugin_driver_t *plugin_driver; namespace { -/** - * Cap on the blob this runtime will handle. - * - * Generous compared with baremetal's 512 bytes — there is no SRAM pressure - * here — but bounded on purpose: the buffer is read from the scan path, and an - * unbounded allocation driven by a program's declaration count is not - * something to discover on a running machine. A program needing more is - * refused at init with a message naming both numbers. - */ +/* Cap on the retain blob. Read on the scan path, so an unbounded size + * driven by a program's declaration count is refused at init. */ constexpr size_t RETAIN_BUFFER_MAX = 64 * 1024; std::vector g_buffer; std::atomic g_active{false}; -/** - * Restore writes go through the runtime's external-write path, NOT straight to - * the IECVar. - * - * A retained variable may also be located (`VAR RETAIN x AT %MW10`). Poking - * such a leaf's storage directly is undone by the next copy-in from the process - * image, so the value would appear to restore and then silently revert on the - * first scan. `runtime_external_write` classifies the leaf and routes a located - * one through the image journal — the same path OPC-UA writes take. - * - * DBGW_OP_WRITE, never a force: restoring a retained value must not pin it. The - * program has to be able to move it on the very next scan, and an operator's - * force has to stay authoritative over whatever was stored. - */ +/* Restore via runtime_external_write (not direct IECVar poke): routes a + * located leaf through the image journal so copy_in won't revert it. + * DBGW_OP_WRITE, never a force — a restore must not pin the slot. */ uint8_t retain_write_leaf(uint8_t arr, uint16_t elem, const uint8_t *bytes, uint16_t len) { return runtime_external_write(arr, elem, (uint8_t)DBGW_OP_WRITE, bytes, len) == 0 ? 0x7E : 0x82; } -/** - * A store, whatever kind it is. - * - * Three function pointers and a name. Everything past init() calls through this - * record, so there is exactly one path to storage and no branch anywhere that - * asks whether the bytes are going to a plugin or to a file. Adding a third - * kind of store means filling this in from somewhere new and changing nothing - * else. - */ +/* Uniform store record (name + 3 fn pointers). Past init() every path + * calls through this, so plugin and file store share one code path. */ struct RetainDriver { const char *name; @@ -81,13 +56,8 @@ struct RetainDriver }; /* Written only by plc_retain_init(), read from the scan thread. - * - * `g_active` IS THE PUBLICATION BARRIER for this record. init() stores false - * before mutating it and true after, both seq_cst, and every reader checks - * g_active before touching g_driver — so a reader that sees active==true is - * guaranteed to see the completed record. Nothing else orders these writes, so - * an early return that skips the `store(true)`, or a relaxed memory order on - * either store, would break it silently. */ + * g_active is the publication barrier: init() stores false (seq_cst) + * before mutating g_driver and true after; readers gate on g_active. */ RetainDriver g_driver = {nullptr, nullptr, nullptr, nullptr}; /* The plugin acting as the store, when a plugin claimed it. Held only so the @@ -125,10 +95,8 @@ void plc_retain_init(void) g_plugin_store = nullptr; g_driver = {nullptr, nullptr, nullptr, nullptr}; g_buffer.clear(); - /* Re-read retain.conf on every program load, so settings that arrived with - * a program upload take effect on the next PLC start without needing the - * daemon restarted. Stopping first is what forces the re-read, and it also - * commits anything the previous run was still holding. */ + /* Re-read retain.conf per program load; stopping first forces the + * re-read and commits anything the previous run was still holding. */ plc_retain_file_store_stop(); if (!ext_strucpp_retain_blob_size || !ext_strucpp_retain_pack || !ext_strucpp_retain_unpack) @@ -149,15 +117,8 @@ void plc_retain_init(void) return; } - /* Ask the drivers, in rank order, which will hold the bytes. - * - * A vendor plugin outranks the built-in file store because the vendor knows - * what the box actually has — FRAM, battery-backed SRAM, an NVS partition — - * and a file on the data partition is the runtime's default, not its - * preference. In a correctly declared device only one of them offers itself - * at all: the file store answers no unless retain.conf enabled it, and the - * editor emits no retain.conf for a target whose VPP declared that it owns - * retention. So this is a rank, not an arbitration. */ + /* Pick a store in rank order: a VPP that owns retention wins over the + * built-in file store. In a correct device only one offers itself. */ g_plugin_store = plugin_driver_find_retain_store(plugin_driver); if (g_plugin_store) { @@ -193,25 +154,14 @@ void plc_retain_read(void) { if (!g_active.load() || !driver_bound()) return; - /* The program's identity, so the driver can tell whether the bytes it holds - * belong to the program now running. Resolved from the .so at load time - * (image_tables.cpp), so it is already available here. Exactly 32 - * characters of hex and NOT guaranteed NUL-terminated, which is why the - * length travels with it rather than being recovered with strlen. */ + /* Program identity (32 hex, NOT NUL-terminated: length travels), + * resolved from the .so at load; drivers compare it to tell a new + * program from the old one. */ if (!ext_strucpp_program_md5) { - /* No identity to compare against means a driver cannot tell a new - * program from the old one, and restoring on that basis is how one - * program inherits another's state. Refuse — and stand the store DOWN - * rather than leave saves running. - * - * Returning while `g_active` stayed true left retain half-on: the - * per-scan save kept packing and handing over bytes that no driver could - * ever commit, because the identity a commit needs is only ever set by - * the read this branch skipped. The file store then refused every write - * and logged "short write" once per flush interval, forever, naming a - * cause that was not the real one. One warning, said once, is the whole - * story — so make it true. */ + /* No identity means a driver cannot tell programs apart; stand + * the store down (g_active=false) rather than leave saves running + * against bytes no driver can ever commit. */ log_warn("Retain: the program exports no MD5 — retained variables start at their " "initial values"); g_active.store(false); diff --git a/core/src/plc_app/plc_retain.h b/core/src/plc_app/plc_retain.h index 565f2b99..27414e42 100644 --- a/core/src/plc_app/plc_retain.h +++ b/core/src/plc_app/plc_retain.h @@ -3,7 +3,7 @@ /** * @file plc_retain.h - * @brief Retain-variable persistence — the runtime's half (NODE-94). + * @brief Retain-variable persistence — the runtime's half. * * The runtime MARSHALS and the platform STORES, exactly as on baremetal. The * marshalling itself lives inside the loaded .so (STruC++'s `iec_retain.hpp`, diff --git a/core/src/plc_app/plc_retain_file_store.cpp b/core/src/plc_app/plc_retain_file_store.cpp index 74d08b15..51efbea4 100644 --- a/core/src/plc_app/plc_retain_file_store.cpp +++ b/core/src/plc_app/plc_retain_file_store.cpp @@ -48,11 +48,9 @@ RtMutex g_lock; std::vector g_pending; bool g_dirty = false; -/* The running program's identity, taken from the last load() and committed - * alongside the blob by every save(). Held rather than written at load time so - * a read never mutates storage, and so identity and bytes always reach the disk - * as one unit — a file whose header says "program A" is guaranteed to contain - * program A's values. Empty until the first load(). */ +/* Program identity from the last load(); committed alongside the blob by + * every save() so identity and bytes reach disk as one unit. Empty until + * the first load() so a read never mutates storage. */ std::string g_program_md5; std::string g_path; @@ -111,27 +109,15 @@ void read_config(const char *config_path) g_enabled.store(enabled); } -/** - * Publish the blob. - * - * Write-and-rename, so a power loss mid-write leaves the PREVIOUS good blob - * rather than a half-written one. The runtime's crc would catch a torn write - * and fall back to initial values anyway, but losing the previous values as - * well would be gratuitous. - */ +/* Publish the blob via write-and-rename, so a power loss mid-write keeps + * the PREVIOUS good blob instead of a half-written one. */ void commit(const uint8_t *buf, uint16_t len, const std::string &program_id) { const std::string tmp = g_path + ".tmp"; - /* File layout: [PROGRAM_ID_LEN bytes of md5 hex][blob]. - * - * The identity goes in the same file as the bytes, and the same - * write-and-rename publishes both, so the two can never disagree — a - * separate sidecar could be updated and then lost, leaving one program's - * values labelled with another's. A file that predates this header, or a - * torn one shorter than the header, simply fails the identity check on the - * next load and is discarded, which is the correct outcome for bytes whose - * owner cannot be established. */ + /* File layout: [PROGRAM_ID_LEN bytes of md5 hex][blob]. Identity and + * bytes publish together in one rename, so they cannot disagree. A + * file shorter than the header fails identity check on load. */ FILE *f = fopen(tmp.c_str(), "wb"); if (!f) @@ -169,12 +155,9 @@ void commit(const uint8_t *buf, uint16_t len, const std::string &program_id) return; } - /* fsync the DIRECTORY too. fsync on the file commits its contents; the - * rename that publishes them is a directory operation, and on ext4 it can - * still be lost to a power cut after the data is safely on disk. Without - * this the store can come back holding the previous blob even though the - * new one was written — the failure that looks like retain silently - * skipping an interval. */ + /* fsync the DIRECTORY too: on ext4 the publishing rename can be lost + * to a power cut even after the file contents are on disk. Without + * this the store can come back holding the previous blob. */ std::vector dircopy(g_path.begin(), g_path.end()); dircopy.push_back('\0'); const int dirfd = open(dirname(dircopy.data()), O_RDONLY | O_DIRECTORY); @@ -185,13 +168,9 @@ void commit(const uint8_t *buf, uint16_t len, const std::string &program_id) } } -/** Remove the stored file and forget the buffered blob. - * - * Not gated on `enabled`: what is being discarded belongs to a PREVIOUS - * program, and may have been written while the store was configured - * differently. The identity is deliberately NOT cleared — the caller has just - * set it to the program now running, and the next save has to label its bytes. - */ +/* Remove the stored file and forget the buffered blob. Not gated on + * `enabled` — the bytes belong to a PREVIOUS program. Identity is NOT + * cleared: caller has just set it to the program now running. */ void discard_stored() { { diff --git a/core/src/plc_app/plc_retain_file_store.h b/core/src/plc_app/plc_retain_file_store.h index f79a3814..008b1b5b 100644 --- a/core/src/plc_app/plc_retain_file_store.h +++ b/core/src/plc_app/plc_retain_file_store.h @@ -69,15 +69,10 @@ bool plc_retain_file_store_active(void); /** @brief The configured path, for logging. Empty when disabled. */ const char *plc_retain_file_store_path(void); -/* The three hooks, shaped exactly like the plugin ones (plugin_driver.h) so - * plc_retain.cpp routes to either through one uniform driver record and never - * learns which kind of store it got. 0 on success. - * - * `load` is handed the running program's identity and decides for itself - * whether the bytes on disk still belong to it — a file labelled with a - * different program is removed and reported empty. See the plugin contract in - * plugin_driver.h; this store is one implementation of it, not a special case - * beside it. */ +/* Hooks shaped like the plugin ones in plugin_driver.h; plc_retain.cpp + * routes through one uniform driver record. 0 on success. `load` is + * given the running program's identity and discards a file labelled + * for a different program. */ int plc_retain_file_store_save(const uint8_t *blob, uint16_t len); int plc_retain_file_store_load(const char *program_md5, uint16_t md5_len, uint8_t *out, uint16_t cap, uint16_t *out_len); diff --git a/core/src/plc_app/plc_state_manager.cpp b/core/src/plc_app/plc_state_manager.cpp index 148c905b..877d5b7d 100644 --- a/core/src/plc_app/plc_state_manager.cpp +++ b/core/src/plc_app/plc_state_manager.cpp @@ -1,14 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// plc_state_manager.cpp -// -// Walks the loaded program's ConfigurationInstance via virtual dispatch -// (Phase 5), spawns one SCHED_FIFO pthread per IEC TASK (Phase 6), and -// anchors the per-cycle housekeeping window on the fastest task's -// thread (Phase 7). -// -// Linux-only (the runtime targets Linux). +// Walks the loaded program's ConfigurationInstance via virtual dispatch, +// spawns one SCHED_FIFO pthread per IEC task, and anchors per-cycle +// housekeeping on the fastest task's thread. Linux-only. #ifndef _GNU_SOURCE #define _GNU_SOURCE @@ -59,15 +54,6 @@ PluginManager *plc_program = NULL; extern plugin_driver_t *plugin_driver; -/* ----------------------------------------------------------------------- - * Per-task storage. Allocated when a program loads, freed on stop. - * - * plc_tasks_lock serialises lifecycle (alloc/publish/free) against - * readers (STATS handler in scan_cycle_manager). See plc_state_manager.h - * for the contract. Read-only once published until the next STOP, so the - * lock is held only briefly on the writer side and for one iteration on - * the reader side. Not a recursive lock — callers must not nest. - * --------------------------------------------------------------------- */ PlcTaskCtx *plc_tasks = nullptr; size_t plc_task_count = 0; @@ -83,24 +69,6 @@ extern "C" void plc_tasks_reader_unlock(void) pthread_mutex_unlock(&plc_tasks_lock); } -/* ----------------------------------------------------------------------- - * Task-completion signalling (for off-hot-path cycle_end). - * - * g_tasks_running = number of task scans currently in flight (released by - * the dispatcher but not yet finished). The dispatcher increments it once - * per release; every worker decrements it exactly once when it leaves a scan - * — via ANY exit path (normal completion, C++ exception, hardware-signal - * recovery) — and the worker that brings it to 0 signals done_cond. - * - * The dispatcher waits on done_cond with the next tick's ABSOLUTE - * CLOCK_MONOTONIC deadline (pthread_cond_timedwait). It therefore wakes on - * whichever comes first: all-tasks-done (fire cycle_end now, off the - * task-wake hot path) or the deadline (start the next tick). The condvar's - * clock is set to CLOCK_MONOTONIC so its deadline shares the dispatcher's - * timeline. g_tasks_running is reset to 0 at each program load, so a stale - * count left by a STOP (a worker woken to exit without finishing a scan) - * never carries into the next run. - * --------------------------------------------------------------------- */ static std::atomic g_tasks_running{0}; static pthread_mutex_t done_mutex = PTHREAD_MUTEX_INITIALIZER; static pthread_cond_t done_cond; /* initialised once, CLOCK_MONOTONIC */ @@ -113,10 +81,8 @@ __attribute__((constructor)) static void state_manager_locks_init_pi(void) rt_mutex_upgrade_static(&done_mutex, "done_mutex"); } -/* One-time init of done_cond on the CLOCK_MONOTONIC clock (the default is - * CLOCK_REALTIME, which would mismatch the dispatcher's monotonic deadline and - * jump under NTP). Init-once (not per-load) so a crash that skips teardown - * never leaves a destroyed-then-reinitialised cond. */ +/* One-time init of done_cond on CLOCK_MONOTONIC (not REALTIME) so it + * matches the dispatcher deadline and does not jump under NTP. */ static void init_done_cond(void) { pthread_condattr_t cattr; @@ -140,10 +106,9 @@ static void worker_scan_done(void) } } -/* The bootstrap thread doesn't run any IEC task body — it does setup, - * spawns task threads, waits, and joins. We still want crash recovery - * on it via a separate jmp pair. The active task's ctx is in __thread - * storage so the signal handler knows which siglongjmp target to use. */ +/* Bootstrap thread runs no IEC task body, but needs its own crash + * recovery jmp pair. Active task ctx is __thread so the signal handler + * picks the right siglongjmp target. */ static __thread PlcTaskCtx *current_task_ctx = nullptr; static sigjmp_buf bootstrap_crash_jmp; static volatile sig_atomic_t bootstrap_crash_sig = 0; @@ -202,17 +167,9 @@ static void plc_abort_handler(int sig) } } -/* Drop whichever runtime lock this task thread currently holds. Mirrors the - * signal-handler recovery (the sigsetjmp block below) so a C++ exception - * thrown mid-scan can't leave the image mutex locked when the thread unwinds - * and exits. holding_mutex is set only inside the locked window, so at most - * that lock is released. - * - * Shared globals are no longer synced under a single runtime-owned mutex: - * each shared global carries its own std::mutex inside the .so (strucpp's - * GlobalVar), taken and released around each access within run(). Those - * fine-grained locks are always released before run() returns, so there is - * nothing global for the crash path to unwind here. */ +/* Drop the image mutex if this task is in the locked window. Mirrors + * the signal-handler recovery below so a C++ exception mid-scan does + * not leave it held. Per-global strucpp mutexes self-release. */ static void plc_task_release_locks(PlcTaskCtx *ctx) { if (ctx->holding_mutex) @@ -254,12 +211,8 @@ static void plc_task_body(PlcTaskCtx *ctx) if (ctx->cpu_affinity_mask != 0) { - // CPU affinity uses `cpu_set_t` / `CPU_ZERO` / `CPU_SETSIZE` / - // `pthread_setaffinity_np` — all Linux extensions to ``, - // not present on MSYS2/Cygwin/Windows. Skip silently when - // building for a non-Linux host: Windows doesn't expose - // SCHED_FIFO either, so any "pin to CPU" guarantee would already - // be unachievable there. + // CPU affinity is Linux-only (cpu_set_t family). SCHED_FIFO is + // also unavailable on Windows, so skip silently off-Linux. #if !defined(__CYGWIN__) && !defined(__MSYS__) && defined(__linux__) cpu_set_t cs; CPU_ZERO(&cs); @@ -297,10 +250,8 @@ static void plc_task_body(PlcTaskCtx *ctx) auto *task = static_cast(ctx->task_handle); - /* Worker loop. No clock here: the GCD master-tick dispatcher owns all - * timing. The worker blocks on its release semaphore between scans, runs - * exactly one scan per release, and bumps `completed` so the dispatcher's - * binary-release / overrun logic can see it finished. */ + /* Worker loop. No clock: the GCD dispatcher owns all timing. One + * scan per release; `completed` is bumped so overrun logic sees it. */ while (true) { /* Block until the dispatcher releases us. sem_wait returns EINTR on a @@ -310,32 +261,18 @@ static void plc_task_body(PlcTaskCtx *ctx) while (sem_wait(&ctx->go) != 0 && errno == EINTR) { /* retry */ } if (plc_get_state() != PLC_STATE_RUNNING) break; - /* Apply this task's scan-stable IEC time, stamped by the dispatcher at - * release. Runs ON this worker thread so the thread_local - * __CURRENT_TIME_NS lands where this task's body (and its FB timers) - * read it — TIME() is constant for the whole scan and unaffected by the - * dispatcher advancing the master clock for other tasks. */ + /* Apply the dispatcher-stamped scan-stable IEC time. Must run on + * THIS worker thread so the thread_local __CURRENT_TIME_NS is + * what the task body and its FB timers read. */ if (ext_strucpp_set_current_time) ext_strucpp_set_current_time(ctx->time_at_dispatch); scan_cycle_tracker_start(&ctx->tracker); - /* Process-image model: a short locked window drains located inputs into - * the image; the body then runs against the .so storage directly. - * Shared globals are NOT copied to private per-task storage — each - * global's own mutex (strucpp GlobalVar) serializes concurrent - * access inside run(), so task bodies still execute in parallel and only - * contend on the specific globals they touch. holding_mutex gates the - * crash-handler unlock of the image lock. - * - * The whole scan body runs under try/catch. On this hosted, - * exceptions-enabled build the STruC++ runtime THROWS on an - * unrecoverable fault (null IEC reference, array index out of bounds, - * bad located address). If that throw escaped this pthread entry - * function it would hit std::terminate -> abort the whole process -> - * the daemon would bounce every task. Instead we catch it here: this one - * task releases its lock, marks itself dead, and terminates; the - * dispatcher then stops releasing it and the other tasks keep scanning. */ + /* Short locked window drains located inputs into the image, then + * the body runs against .so storage directly. Shared globals have + * their own strucpp mutexes. try/catch here contains strucpp + * runtime throws so one task's fault does not abort the process. */ try { /* 1. Copy-in located inputs (image_lock drains the journal). */ @@ -582,15 +519,9 @@ void *plc_cycle_thread(void *arg) image_tables_fill_null_pointers(); pthread_mutex_unlock(itm); - /* Retained variables. init() decides once whether retain can run here — - * does the .so export the entry points, does the program retain anything, - * which driver will hold the bytes — and read() asks that driver for what it - * has for THIS program, which is also where a driver discards a previous - * program's values. Both must follow the located-variable binding above: a - * retained variable may also be located, and its image slot has to exist - * before anything writes through it. Both are no-ops when retain is not in - * play, and read() lands before the first task is released, so a new - * program never runs a scan on the old one's state. */ + /* Retained variables. init() decides capability; read() loads values + * for THIS program and discards a prior program's bytes. Must run + * after located-variable binding so image slots exist. */ plc_retain_init(); plc_retain_read(); @@ -650,16 +581,9 @@ void *plc_cycle_thread(void *arg) log_info("Starting main loop"); - /* NOT where RUNNING is published. The state stays TRANSITIONING_TO_RUN until - * the task threads exist and the dispatcher is about to release the first - * scan -- see the publish below the "Spawned N PLC task thread(s)" log. Two - * writes used to happen before this point (here, and in plc_set_state before - * this thread was even created), and both claimed RUNNING while nothing was - * scanning yet. The one here was also a deadlock: a stop landing in the - * window before this thread was first scheduled wrote STOPPED and joined us, - * and this line put RUNNING back -- after which no loop below would ever - * exit, so the join never returned and the runtime refused every command for - * the rest of the process's life. */ + /* State stays TRANSITIONING_TO_RUN until the task threads exist and + * the dispatcher is about to release the first scan. RUNNING is + * published below after "Spawned N PLC task thread(s)". */ clock_gettime(CLOCK_MONOTONIC, &timer_start); @@ -738,11 +662,9 @@ void *plc_cycle_thread(void *arg) log_info("PLC base tick: %llu ns across %zu task(s)", (unsigned long long)base_ns, total_tasks); - /* Sub-millisecond base tick warning. Whole-millisecond task intervals - * always yield a GCD >= 1 ms; a base tick below that means fractional/sub-ms - * intervals that are near-coprime, so the dispatcher must wake faster than - * 1 kHz. We do NOT clamp (that would break the per-task time grid); we run - * at the true GCD and rely on overrun detection if it can't keep up. */ + /* Sub-millisecond base tick means fractional/sub-ms task intervals + * near-coprime. Do NOT clamp (breaks the per-task time grid); run at + * the true GCD and rely on overrun detection. */ if (base_ns < 1000000ULL) { log_warn("PLC base tick is %llu ns (< 1 ms): dispatcher runs at %llu Hz. " @@ -850,22 +772,10 @@ void *plc_cycle_thread(void *arg) plc_tasks[fastest_idx].priority); } - /* Spawn task threads. - * - * Failure mode: if pthread_create succeeds for tasks 0..i-1 and then - * fails for task i, the previously-spawned threads are running at - * SCHED_FIFO (1..PLC_FIFO_TASK_MAX) and reading from - * plc_tasks[]. Returning here without cleanup leaves them orphaned - * — the next load_plc_program reallocates plc_tasks and the old - * threads dereference freed memory. We must: - * - * 1) flip plc_state to ERROR so the surviving task threads exit - * their `while (state == RUNNING)` loop on their next iteration; - * 2) post each surviving thread's release semaphore so it wakes; - * 3) join all spawned threads before freeing the array. - * - * After this rollback, plc_tasks is nullptr and plc_task_count is 0, - * so STATS / next-cycle teardown can run without UAF. */ + /* Spawn task threads. On partial failure (pthread_create succeeds + * for 0..i-1, fails at i): flip state to ERROR, post every + * surviving thread's release so it wakes, join them, free the + * array. Leaves plc_tasks nullptr and count 0 so STATS is UAF-safe. */ size_t spawned = 0; for (; spawned < plc_task_count; ++spawned) { @@ -909,24 +819,10 @@ void *plc_cycle_thread(void *arg) } log_info("Spawned %zu PLC task thread(s)", plc_task_count); - /* RUNNING, at last, and this is the earliest point it is true: the workers - * exist and the very next thing that happens is the dispatcher releasing the - * first scan. Nothing observes RUNNING too early as a result -- the workers - * are parked in sem_wait and are only ever posted from the loop below, and - * the loop itself needs RUNNING visible to run at all. - * - * This is also what ends the transition claimed by plc_claim_transition, so - * every path out of this thread from here on must land a final state: the - * crash recovery above publishes ERROR, and the stop path publishes STOPPED - * once teardown joins. - * - * Conditional, because ending a transition is only ours to do while it is - * still the one in flight. Two paths get here otherwise: the watchdog forced - * ERROR because this start exceeded its bound, and plc_state_manager_cleanup - * published TRANSITIONING_TO_STOP on shutdown and is now blocked joining this - * very thread. Publishing RUNNING in either case erases a state someone else - * landed, and in the second it hangs the process -- the dispatcher loop below - * would never see a non-RUNNING state and the join would never return. */ + /* Publish RUNNING only while OUR start is still the claimed + * transition. Erasing a watchdog-ERROR or shutdown-TRANSITIONING_TO_STOP + * would hang the join below. The crash and stop paths each publish + * their own final state (ERROR / STOPPED). */ if (!plc_publish_running_if_claimed()) { log_warn("PLC bring-up finished but the start is no longer the transition in " @@ -938,23 +834,7 @@ void *plc_cycle_thread(void *arg) return NULL; } - /* --------------------------------------------------------------------- - * GCD master-tick dispatcher. - * - * This thread is the single time authority. It wakes every base_ns on an - * absolute deadline anchored at one t0, and on each tick: - * - feeds the watchdog (watchdog_feed, every tick); - * - computes the due set (task due iff masterTick % divisor == 0); - * - on a task-bearing tick: drains the journal (committing the previous - * frame's outputs), fires cycle_end (prev frame) then cycle_start (new - * frame), stamps each due+alive task's dispatch time and releases it - * (binary — never queues a second activation), bumps scan_counter. - * It NEVER waits for a worker body (a long body would stall the clock); a - * worker still running when re-due is an overrun and is simply not - * re-released that tick. A faulted worker (alive==0) is skipped forever. - * - * Runs at PLC_FIFO_DISPATCHER: above every worker, below the watchdog. - * --------------------------------------------------------------------- */ + { pthread_setname_np(pthread_self(), "plc_dispatch"); sched_param dsp{}; @@ -1019,11 +899,9 @@ void *plc_cycle_thread(void *arg) if (any_due) { - /* Worst case: the previous frame's tasks didn't all finish before - * this tick (overrun), so its cycle_end was never retired in Phase A - * below. Fire it now — drain to commit those outputs, then cycle_end - * — before opening the new frame with cycle_start. This is the only - * path where cycle_end lands on the task-wake hot path. */ + /* Overrun: previous frame's cycle_end not yet retired. Fire + * it before opening the new frame. Only path where cycle_end + * lands on the task-wake hot path. */ if (cycle_end_pending) { image_lock(); @@ -1033,13 +911,9 @@ void *plc_cycle_thread(void *arg) } if (plugin_driver) plugin_driver_cycle_start(plugin_driver); - /* Config-scope shared globals: prime the canonical storage from the - * freshly-read input image before releasing this frame's tasks. Only - * when quiescent (g_tasks_running == 0): if a previous frame's task - * is still overrunning it is accessing the canonical storage under - * its own per-global mutex, so a raw copy here would race — skip the - * refresh for this frame (bounded staleness under overrun, matching - * the "sync only on the guarded no-overrun path" contract). */ + /* Prime config-scope shared globals from the input image. + * Guarded by quiescence (no overrunning task is in run()) + * to avoid racing a per-global mutex. */ if (g_tasks_running.load(std::memory_order_acquire) == 0) { image_lock(); @@ -1133,16 +1007,10 @@ void *plc_cycle_thread(void *arg) ++master_tick; - /* ---- Phase A: wait out the period on the absolute deadline, waking - * early to retire cycle_end the instant the frame's tasks all finish. - * - * pthread_cond_timedwait wakes on whichever comes first: the - * completion signal (g_tasks_running hit 0) or the deadline. When the - * frame is done we drain + fire cycle_end here — OFF the task-wake hot - * path — then keep waiting (cycle_end_pending now false) purely for the - * deadline. The predicate (g_tasks_running == 0) is checked under - * done_mutex BEFORE waiting, so a task that finishes before we reach the - * wait can't lose its wakeup. ---- */ + /* Phase A: wait out the period on the absolute deadline, waking + * early to retire cycle_end as soon as the frame completes + * (g_tasks_running hits 0). Predicate checked under done_mutex + * before the wait so a finishing task does not lose its wakeup. */ next_tick.tv_nsec += (long)(base_ns % 1000000000ULL); next_tick.tv_sec += (time_t)(base_ns / 1000000000ULL); if (next_tick.tv_nsec >= 1000000000L) @@ -1164,21 +1032,16 @@ void *plc_cycle_thread(void *arg) { pthread_mutex_unlock(&done_mutex); image_lock(); /* drain: commit this frame's outputs */ - /* Config-scope shared globals (VAR_GLOBAL ... AT): g_tasks_running - * == 0 so no worker is mid-scan touching the canonical storage — - * journal changed output/memory globals here, before the drain - * applies them to the image. Safe without the per-global mutex - * (quiescence is the synchronization). */ + /* Journal config-scope globals before the drain copies + * them to the image. Safe under quiescence; no per-global + * mutex needed. */ image_tables_copy_config_globals_out(); /* Apply queued external writes/forces (debugger, OPC-UA) here: * g_tasks_running == 0 so no worker is mid-scan, and we hold * image_lock. Cheap no-op when nothing is queued. */ debug_write_journal_drain(); - /* Retained values, once per scan. This window is the only place - * they can be read safely: g_tasks_running == 0, so no worker is - * inside a body mutating them — the same guarantee - * copy_config_globals_out relies on. The plugin decides whether - * these bytes are actually committed now; no-op with no store. */ + /* Retained values once per scan, under quiescence. The + * plugin decides whether to actually commit now. */ plc_retain_save(); image_unlock(); if (plugin_driver) plugin_driver_cycle_end(plugin_driver); @@ -1222,13 +1085,8 @@ extern "C" int load_plc_program(PluginManager *pm) if (plugin_manager_load(pm)) { - /* Progress, not a state change. This used to publish PLC_STATE_INIT, - * which is now actively wrong: it overwrites TRANSITIONING_TO_RUN, so the - * runtime would stop reporting a transition in flight and let a stop be - * claimed while the start was still landing -- the same hole the - * TRANSITIONING states exist to close. (It is also what made STATUS - * flicker to INIT mid-start.) INIT survives in the enum as a startup - * value; nothing writes it during a transition. */ + /* Log only: must NOT publish PLC_STATE_INIT here, which would + * erase TRANSITIONING_TO_RUN and let a stop be claimed mid-start. */ log_info("Loading PLC application"); if (plugin_driver) @@ -1307,12 +1165,9 @@ extern "C" int load_plc_program(PluginManager *pm) plc_state = PLC_STATE_EMPTY; pthread_mutex_unlock(&state_mutex); log_info("PLC State: EMPTY"); - // Without this, plc_program survives the failed dlopen with a - // stale so_path. The build script rotates the libplc filename - // (libplc_.so) on every successful build, so the - // next "Start PLC" would reuse the deleted path and fail with - // `cannot open shared object file`. Drop the manager and let - // plc_set_state(RUNNING) re-run find_libplc_file next time. + // Clear plc_program so the next Start re-runs find_libplc_file. + // The build rotates libplc_.so, so the stale so_path would + // fail with "cannot open shared object file". if (pm == plc_program) plc_program = NULL; plugin_manager_destroy(pm); return -1; @@ -1341,18 +1196,9 @@ static int unload_plc_program_locked(PluginManager *pm) { if (pm && pm == plc_program) { - /* The dispatcher and its workers leave their loops on anything that is - * not RUNNING, and the join below depends on that. A claimed stop has - * already published TRANSITIONING_TO_STOP, which is that signal. - * - * Anything that is not ERROR is overwritten, not just RUNNING, because - * this also covers the shutdown path (plc_state_manager_cleanup), which - * tears down without claiming a transition first. On SIGTERM during a - * start the state is TRANSITIONING_TO_RUN, and a guard that only matched - * RUNNING published nothing at all -- so the cycle thread went on to - * publish RUNNING and the join below never returned. Pairs with the - * conditional publish in plc_cycle_thread: this write is what makes that - * one decline. ERROR is left alone -- it must survive teardown. */ + /* Publish TRANSITIONING_TO_STOP to break the dispatcher and + * workers out of their RUNNING loops. Overwrite anything but + * ERROR, which must survive teardown. */ pthread_mutex_lock(&state_mutex); if (plc_state != PLC_STATE_ERROR) { @@ -1362,13 +1208,9 @@ static int unload_plc_program_locked(PluginManager *pm) pthread_join(plc_thread, NULL); - /* Retained values: ask the store to commit whatever it is still - * holding. Placed exactly here on purpose — AFTER the join, so no scan - * is mid-save and the bytes are a consistent snapshot, and BEFORE - * plugin_driver_stop(), because a plugin-backed store has to still be - * alive to answer. A driver that already commits inside save() has - * nothing to do; what this buys is that a clean stop loses nothing on - * one that buffers. */ + /* Commit retained values: AFTER the join (consistent snapshot) + * and BEFORE plugin_driver_stop (plugin-backed stores need to + * still be alive to answer). */ plc_retain_flush(); journal_cleanup(); @@ -1535,10 +1377,8 @@ extern "C" void plc_publish_final_state(PLCState final_state) extern "C" bool plc_publish_running_if_claimed(void) { - /* Land RUNNING only while the start we are completing is still the transition - * in flight. The check and the write share one critical section: reading the - * state and then publishing in two steps would let a stop be claimed in - * between, and RUNNING would go down on top of it. */ + /* Check and publish RUNNING under one critical section, so a stop + * cannot be claimed between the read and the write. */ pthread_mutex_lock(&state_mutex); if (plc_state != PLC_STATE_TRANSITIONING_TO_RUN) { @@ -1553,18 +1393,10 @@ extern "C" bool plc_publish_running_if_claimed(void) extern "C" bool plc_set_state(PLCState new_state) { - // Performs a transition already claimed via plc_claim_transition(), which - // published TRANSITIONING_TO_RUN or TRANSITIONING_TO_STOP. No state is - // written here: writing the target up front is what used to let a stop's - // STOPPED be resurrected by a start still landing. The final state is - // published by whoever knows the transition actually finished -- - // plc_cycle_thread for RUNNING (just before it releases the first scan) and - // unload_plc_program for STOPPED (after the teardown joins) -- with the - // failure paths below publishing ERROR or EMPTY. - // - // The current TRANSITIONING state is itself the signal the task and - // dispatcher loops need: they run while plc_get_state() == RUNNING, so - // TRANSITIONING_TO_STOP breaks them exactly as the old early STOPPED did. + // Executes a transition already claimed via plc_claim_transition. + // No state is written here: the final state is published by whoever + // confirms the transition finished (plc_cycle_thread for RUNNING, + // unload_plc_program for STOPPED). if (new_state == PLC_STATE_RUNNING) { @@ -1625,20 +1457,9 @@ extern "C" bool plc_set_state(PLCState new_state) extern "C" void plc_state_manager_cleanup(void) { - /* Let an in-flight transition finish before tearing anything down. - * - * Shutdown is the one state change that does not go through - * plc_claim_transition, so it can land on top of a start that is still - * running. Tearing down from there is not safe at any point of it: during - * plugin bring-up load_plc_program has not assigned plc_thread yet, so the - * join below would run on a handle that was never set; a moment later the - * cycle thread is mid-bring-up and would have RUNNING published underneath - * the teardown. Both disappear if the transition is allowed to land first -- - * then this is an ordinary stop of a RUNNING (or ERROR, or EMPTY) runtime. - * - * Bounded by the same constant the transition worker waits on, so a - * transition that will never land cannot hold the process open forever; the - * teardown then proceeds and does what it can. */ + /* Wait for any in-flight transition to land before teardown, so we + * never join an unassigned plc_thread or race the cycle thread's + * RUNNING publish. Bounded by the transition-landing timeout. */ const int poll_ms = 20; int waited = 0; while (plc_state_is_transitioning() && waited < PLC_TRANSITION_LANDING_TIMEOUT_MS) diff --git a/core/src/plc_app/plc_state_manager.h b/core/src/plc_app/plc_state_manager.h index eca51f26..703d517a 100644 --- a/core/src/plc_app/plc_state_manager.h +++ b/core/src/plc_app/plc_state_manager.h @@ -30,25 +30,9 @@ typedef atomic_uint_least64_t plc_atomic_u64_t; typedef atomic_int_least64_t plc_atomic_i64_t; #endif -/** - * Runtime states. - * - * The two TRANSITIONING values are APPENDED, never inserted: the first five are - * wire-visible (FC 0x46 reports them, and plugin_get_plc_state maps them for - * vendor status indicators), so renumbering would quietly change what boards - * report. - * - * RUNNING means running -- it is published at the moment the dispatcher is about - * to release the first scan, not when a start is requested. Everything in - * between is a TRANSITIONING state, and because both compare unequal to - * PLC_STATE_RUNNING, every loop that gates on RUNNING treats them correctly - * without modification: a stop's teardown still gets its exit signal, and a - * half-started PLC cannot scan. - * - * The direction is carried (rather than one flat TRANSITIONING) so that any code - * found to need the target state before it lands can test for the specific - * direction instead of having to reintroduce a premature RUNNING. - */ +/* Runtime states. TRANSITIONING values APPEND, never insert: the + * first five are wire-visible (FC 0x46). RUNNING is published at + * first-scan release, not at request. Both TRANSITIONING != RUNNING. */ typedef enum { PLC_STATE_INIT, @@ -60,18 +44,6 @@ typedef enum PLC_STATE_TRANSITIONING_TO_STOP } PLCState; -/* ----------------------------------------------------------------------- - * Per-IEC-task execution context. - * - * One PlcTaskCtx per task declared in the user's CONFIGURATION. Lives - * for the duration of a loaded program; freed on stop. - * - * Per-thread state — crash_jmp, crash_sig, holding_mutex — must NOT be - * shared across threads. Each task thread owns its own context - * exclusively once spawned; the runtime stashes a __thread pointer to - * the active ctx so the signal handler can siglongjmp to the right - * recovery point. - * --------------------------------------------------------------------- */ typedef struct PlcTaskCtx { size_t idx; /* index into plc_tasks[] */ @@ -83,27 +55,7 @@ typedef struct PlcTaskCtx pthread_t thread; char name[32]; - /* ------------------------------------------------------------------------- - * GCD master-tick dispatcher plumbing. - * - * The dispatcher releases this worker by posting `go`; the worker blocks on - * sem_wait(go) between scans. `divisor` = interval_ns / base_tick_ns, so the - * worker is due on master tick N iff N % divisor == 0. - * - * Binary release + overrun detection use released/completed: the dispatcher - * bumps `released` and posts only when released == completed (worker idle); - * if released > completed at a due tick the worker is still in its previous - * scan (overrun) and is NOT re-posted, so activations never queue. The - * worker bumps `completed` at the end of each scan. - * - * `time_at_dispatch` is stamped by the dispatcher at release and applied by - * the worker via ext_strucpp_set_current_time() before run() — giving each - * task a scan-stable IEC TIME() snapshot (§ scheduler design doc). - * - * `alive` (1/0): a worker that hits an unrecoverable fault sets this to 0 - * and returns; the dispatcher then never releases it again (the faulted task - * drops out of the schedule while the others keep running). - * --------------------------------------------------------------------- */ + sem_t go; uint64_t divisor; int64_t time_at_dispatch; @@ -122,33 +74,18 @@ typedef struct PlcTaskCtx plc_atomic_u64_t local_tick; - /* Per-task scan/cycle/latency tracker. Each task thread updates its - * own tracker around its scan body; the STATS handler walks all - * trackers to emit per-task entries. Replaces the old single global - * plc_timing_stats from scan_cycle_manager.c which only tracked the - * fastest task. */ + /* Per-task scan/cycle/latency tracker. The task thread updates it + * around its scan body; STATS walks all trackers for per-task entries. */ scan_cycle_tracker_t tracker; } PlcTaskCtx; extern PlcTaskCtx *plc_tasks; extern size_t plc_task_count; -/* Lifecycle lock for plc_tasks / plc_task_count. - * - * The plc_cycle_thread owns the array — it allocates after walking the - * configuration (load) and frees after joining task threads (stop). - * Concurrently, the unix-socket thread services STATS by iterating the - * array under format_timing_stats_response. The TRANSITIONING state gates new - * commands but doesn't bracket an in-flight STATS call: a plugin-initiated - * stop can fire mid-iteration, free plc_tasks, and the STATS reader - * dereferences freed memory. - * - * Readers (STATS) hold this lock for the duration of the iteration. - * The writer (plc_cycle_thread) holds it while allocating, while - * publishing the count, and while freeing. STOP itself doesn't need the - * lock: task threads exit via plc_state observation; the lock only - * brackets the array swap. Held briefly enough that adding latency to - * STATS during a STOP transition is acceptable. */ +/* Lifecycle lock for plc_tasks / plc_task_count. plc_cycle_thread owns + * the array; without this lock a plugin STOP could free plc_tasks + * while STATS iterates. Readers hold it for the walk; writer holds + * it around alloc, publish and free. */ void plc_tasks_reader_lock(void); void plc_tasks_reader_unlock(void); diff --git a/core/src/plc_app/plc_switch.c b/core/src/plc_app/plc_switch.c index 3fcd70e8..ddeb0f17 100644 --- a/core/src/plc_app/plc_switch.c +++ b/core/src/plc_app/plc_switch.c @@ -26,19 +26,10 @@ /* Default RUN: no plugin implementing the interface means no gating. */ static atomic_int switch_position = PLC_SWITCH_RUN; -/** - * Set when the switch moves, cleared when someone acts on it. - * - * Exists because state-change requests are dropped while a transition is in - * flight: a flip during a start or stop is refused, and without a record of it - * the switch and the PLC end up disagreeing with nobody retrying. Only the fact - * of movement is kept, never a queue of requests -- `switch_position` above - * already holds where the switch came to rest, which is the only position that - * matters once the dust settles. - * - * Deliberately platform-agnostic: every VPP that owns a switch reports through - * plc_set_switch_position(), so no plugin needs to know reconciliation exists. - */ +/* Set when the switch moves, cleared when someone acts on it. State + * requests are dropped during a transition, so without this bit a + * flip during start/stop would never be retried. Only the fact of + * movement is kept; switch_position holds where it came to rest. */ static atomic_bool switch_moved = false; void plc_set_switch_position(plc_switch_t position) diff --git a/core/src/plc_app/python_loader.c b/core/src/plc_app/python_loader.c index 732c9830..40eb4302 100644 --- a/core/src/plc_app/python_loader.c +++ b/core/src/plc_app/python_loader.c @@ -1,12 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// This file is responsible for loading function blocks written in Python. -// Python function blocks communicate with the PLC runtime via shared memory. -// -// Logging is done via function pointers that are set by the runtime after -// loading libplc.so. This avoids symbol resolution issues between the -// shared library and the main executable. +// Loader for function blocks written in Python; they talk to the PLC +// runtime via shared memory. Logging is wired through function pointers +// set after libplc.so is loaded, to avoid cross-library symbol lookup. #include #include diff --git a/core/src/plc_app/scan_cycle_manager.c b/core/src/plc_app/scan_cycle_manager.c index 954250e1..858fc7e3 100644 --- a/core/src/plc_app/scan_cycle_manager.c +++ b/core/src/plc_app/scan_cycle_manager.c @@ -12,17 +12,14 @@ #include "scan_cycle_manager.h" #include "utils/utils.h" -// Use CLOCK_MONOTONIC everywhere to match the clock used by sleep_until() -// (clock_nanosleep with CLOCK_MONOTONIC). Using CLOCK_MONOTONIC_RAW here -// would cause progressive drift against the sleep clock due to NTP slew -// adjustments, eventually leading to false overrun detection after ~30-60 -// minutes of continuous operation. +// CLOCK_MONOTONIC matches clock_nanosleep's clock in sleep_until(). +// CLOCK_MONOTONIC_RAW would drift against the sleep clock through NTP +// slew and cause false overruns after ~30-60 minutes. #define OPENPLC_CLOCK CLOCK_MONOTONIC -// Target wall-clock window for the time-based EWMA averages, in microseconds. -// Matches the editor's 2 s polling cadence so the displayed avg stays stable -// between polls and tracks recent drift rather than freezing as a Welford -// historical mean would after ~10^7 cycles. +// EWMA window (us) for time-based averages. Matches the editor's 2 s poll +// so the displayed avg tracks recent drift and does not freeze like a +// Welford historical mean would after ~10^7 cycles. #define EWMA_TARGET_WINDOW_US 2000000 static uint64_t ts_now_us(void) @@ -192,13 +189,9 @@ int format_timing_stats_response(char *buffer, size_t buffer_size) if (n < 0) return 0; offset += (size_t)n; - /* Hold plc_tasks_reader_lock for the whole iteration. the TRANSITIONING state - * gates new commands but does not bracket an in-flight STATS call — - * a plugin-initiated STOP can fire while we're mid-loop, the bootstrap - * thread joins task threads and frees plc_tasks[], and we'd then read - * freed memory (or worse, lock a destroyed tracker mutex). The lock - * makes the alloc/free critical section in plc_cycle_thread mutually - * exclusive with the iteration here. */ + /* Hold plc_tasks_reader_lock for the whole iteration: a plugin STOP + * can fire mid-loop and free plc_tasks[]. The lock makes the alloc/ + * free critical section in plc_cycle_thread mutually exclusive. */ plc_tasks_reader_lock(); for (size_t i = 0; i < plc_task_count; ++i) { diff --git a/core/src/plc_app/scan_cycle_manager.h b/core/src/plc_app/scan_cycle_manager.h index 9a55482b..563abda6 100644 --- a/core/src/plc_app/scan_cycle_manager.h +++ b/core/src/plc_app/scan_cycle_manager.h @@ -31,21 +31,10 @@ typedef struct int64_t overruns; } plc_timing_stats_t; -/* Per-task scan-cycle tracker. Each IEC task thread owns one and is the - * exclusive writer; the snapshot reader (the STATS handler) acquires the - * mutex briefly to copy out a consistent view. - * - * `interval_ns` is this task's scheduling period, used to project the - * next-expected start time so latency = actual_start - expected_start - * stays meaningful across tasks with different periods. - * - * Averages use a time-based EWMA targeting a wall-clock window that - * matches the editor's polling cadence (so the displayed avg doesn't - * decorrelate between consecutive polls). Each `*_sum` field holds an - * approximate sum of the last `avg_window` samples; the per-cycle update - * is `sum += sample - sum/avg_window` and the read is `avg = sum/avg_window`. - * This accumulator-then-divide form avoids the integer-precision stall - * of the incremental `avg += (sample - avg)/N` shape when delta < N. */ +/* Per-task tracker. The IEC task thread is the sole writer; STATS + * briefly takes the mutex to snapshot. `interval_ns` projects + * next-expected start for latency. EWMA sum form (sum += sample - + * sum/N; avg = sum/N) avoids integer-precision stalls when delta #include diff --git a/core/strucpp_runtime/runtime_v4_entry.cpp b/core/strucpp_runtime/runtime_v4_entry.cpp index 09d72627..8ab2a9c1 100644 --- a/core/strucpp_runtime/runtime_v4_entry.cpp +++ b/core/strucpp_runtime/runtime_v4_entry.cpp @@ -1,31 +1,10 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// runtime_v4_entry.cpp -// -// Static C-linkage shim compiled into every user .so. Identical for every -// project — no per-project codegen. Lives here in the runtime repo -// because the build (scripts/compile.sh) is the consumer; the editor's -// upload bundle does not ship this file. -// -// Responsibilities: -// -// 1. Instantiate strucpp::Configuration_CONFIG0 g_config — the actual -// object the runtime walks. Must have external linkage so -// generated_debug.cpp's compile-time address-of expressions resolve. -// 2. Export strucpp_get_config() — C-linkage entry the runtime dlsyms -// to obtain a ConfigurationInstance* pointer. -// 3. Export strucpp_get_located_vars / strucpp_get_located_var_count -// — re-expose strucpp::locatedVars[] (a per-project namespaced -// symbol) under stable C linkage. -// 4. Activate STRUCPP_V4_DEBUG_EXPORTS_DEFINE — emits the C-linkage -// strucpp_debug_* PDU helpers from debug_dispatch.hpp. -// 5. Export strucpp_advance_time() — bumps the per-.so -// strucpp::__CURRENT_TIME_NS by the runtime-supplied tick. The -// runtime owns the tick (computed from g_config); the shim just -// provides the cross-DSO advance entry point. -// 6. Export strucpp_program_md5 — the project MD5, surfaced by FC 0x45 -// so the editor can verify it's debugging the matching source. +// Static C-linkage shim in every user .so. Instantiates g_config, +// re-exposes strucpp::locatedVars[] and the strucpp_debug_* PDU +// helpers under C linkage, provides the cross-DSO time-advance entry, +// and exports strucpp_program_md5 for FC 0x45. #define STRUCPP_V4_DEBUG_EXPORTS_DEFINE #include "debug_dispatch.hpp" @@ -33,10 +12,9 @@ #include "iec_std_lib.hpp" // ConfigurationInstance + __CURRENT_TIME_NS #include "generated.hpp" -// Retain marshalling is conditional on the STruC++ that built this upload — -// see the block at the bottom of this file, and retain_probe.cpp for how the -// build decides. The include lives behind the same gate so a header set that -// predates the retain API is never even asked for it. +// Retain marshalling is conditional on the STruC++ version — see +// retain_probe.cpp for how the build decides. Include is behind the +// same gate so older header sets are never asked for it. #ifdef STRUCPP_SHIM_HAS_RETAIN #include "iec_retain.hpp" // retain blob format + pack/unpack #endif @@ -53,11 +31,9 @@ extern "C" strucpp::ConfigurationInstance* strucpp_get_config(void) { return &g_config; } -// strucpp::locatedVars / locatedVarsCount are top-level externs declared in -// iec_located.hpp and defined per-project by generated.cpp. The runtime -// (loaded once, sees many .so files) can't reach them by mangled name -// portably, so the shim re-exports them via C linkage. Same pattern as -// strucpp_get_config — the runtime walks via these accessors. +// strucpp::locatedVars/locatedVarsCount are externs defined per-project +// by generated.cpp. Re-exported via C linkage so the runtime (one copy, +// many .so files) can reach them portably by dlsym. extern "C" const strucpp::LocatedVar *strucpp_get_located_vars(void) { return strucpp::locatedVars; @@ -67,19 +43,10 @@ extern "C" uint32_t strucpp_get_located_var_count(void) { return strucpp::locatedVarsCount; } -// Located-variable classifier for the unified external-write path. -// -// Given a debug (arr, elem) leaf, report whether it is a LOCATED variable -// and, if so, its image location (area / size / byte_index / bit_index). -// The runtime's runtime_external_write() uses this to route a write/force -// targeting a located var through the image journal (and the forced-slot -// bitmap) — copy_in would otherwise clobber a direct IECVar poke. Globals -// and program-internal leaves return 0 (applied straight to the IECVar via -// the debug-write journal). -// -// The match is by storage pointer: read_entry(arr,elem).ptr is the leaf's -// IECVar raw_ptr(), the same pointer recorded in locatedVars[].pointer — a -// pure pointer-identity check, no memory-layout assumption. +// Classifier for runtime_external_write(): reports whether (arr, elem) +// is LOCATED and fills image location out-params so the write is routed +// through the image journal (copy_in would clobber a direct IECVar +// poke). Match is pure pointer identity. extern "C" int strucpp_debug_locate(uint8_t arr, uint16_t elem, uint8_t *area, uint8_t *size, uint16_t *byte_index, uint8_t *bit_index) { @@ -98,52 +65,29 @@ extern "C" int strucpp_debug_locate(uint8_t arr, uint16_t elem, return 0; } -// Project MD5. Used by FC 0x45 to let the editor verify it's debugging -// the program it has the source for. The editor emits -// core/generated/defines.h next to generated.cpp during compile, -// defining PROGRAM_MD5 with the actual program hash. PROGRAM_MD5 is -// the same macro name the Arduino sketch's defines.h uses, keeping a -// single MD5 contract across targets. -// -// No fallback: a program loaded without defines.h is broken and must -// fail to compile (missing file) or link (undefined PROGRAM_MD5). The -// editor's v4 build path always emits defines.h. +// Project MD5 for FC 0x45. defines.h is emitted by the editor next to +// generated.cpp during compile with PROGRAM_MD5. No fallback: a program +// loaded without defines.h must fail to compile or link. #include "defines.h" -// Define as a non-const char array so: -// 1. The symbol has external linkage (in C++, namespace-scope -// `const` gives INTERNAL linkage, which would hide the symbol -// from dlsym → runtime sees NULL → FC 0x45 returns NOT_LOADED). -// 2. The symbol's address is the start of the string itself, not a -// pointer variable. The runtime's symbols_init does -// `*(void**)&ext_strucpp_program_md5 = dlsym(...)` and indexes -// ext_strucpp_program_md5[i] directly — a `const char *foo = "..."` -// definition would surface the raw pointer bytes as garbage. -// -// extern "C" block expresses C language linkage without the -// "extern initialized" g++ warning that the single-decl form triggers. +// Non-const char array: (1) external linkage for dlsym (namespace-scope +// `const` gives internal linkage in C++); (2) symbol address IS the +// string start, since the runtime indexes it directly. +// extern "C" block avoids the g++ "extern initialized" warning. extern "C" { char strucpp_program_md5[] = PROGRAM_MD5; } -// Advances the strucpp runtime's scan-cycle clock by `tick_ns` on the CALLING -// thread. Retained for compatibility (and for any single-threaded host); the -// GCD master-tick dispatcher does NOT use it — under STRUCPP_THREADED -// __CURRENT_TIME_NS is thread_local, so a dispatcher-side increment would only -// bump the dispatcher's own (unused) copy. The dispatcher uses -// strucpp_set_current_time() on each worker instead. +// Advances the strucpp scan-cycle clock on the CALLING thread. Kept for +// single-threaded hosts; the GCD dispatcher uses strucpp_set_current_time +// per worker because __CURRENT_TIME_NS is thread_local when STRUCPP_THREADED. extern "C" void strucpp_advance_time(uint64_t tick_ns) { strucpp::__CURRENT_TIME_NS += static_cast(tick_ns); } -// Sets the IEC TIME() base for the CALLING thread. Under STRUCPP_THREADED -// __CURRENT_TIME_NS is thread_local (see iec_std_lib.hpp), so each task worker -// thread that calls this gets its own scan-stable time. The GCD master-tick -// dispatcher stamps each task's dispatch time and the worker calls this at the -// top of its scan, before run() — giving correct multi-rate IEC timing -// (TIME() constant within a scan, no cross-task interference, slow tasks keep -// their own snapshot while the master clock advances). Must be called ON the -// worker thread for the thread_local to land where the body reads it. +// Sets IEC TIME() base for the CALLING thread (thread_local under +// STRUCPP_THREADED). Called by each worker at the top of its scan before +// run() so TIME() is scan-stable per task. Must run on the worker thread. extern "C" void strucpp_set_current_time(int64_t ns) { strucpp::__CURRENT_TIME_NS = ns; } @@ -152,49 +96,8 @@ extern "C" void strucpp_set_current_time(int64_t ns) { // compiles every .so itself with -DSTRUCPP_THREADED, so the threaded // process-image model is the only one; there is nothing to detect. -// --------------------------------------------------------------------------- -// Retain marshalling. -// -// GATED ON THE UPLOAD'S STruC++ VERSION. `strucpp::retain` and -// `strucpp::debug::retain_layout_hash` arrived in STruC++ v0.6.5, and this file -// is compiled against the runtime headers the EDITOR shipped inside -// program.zip — which on every OpenPLC Editor released to date are older than -// that. Compiling this block unconditionally made an older editor's upload fail -// to build (runtime v4.2.0), so scripts/Makefile.strucpp probes the header set -// per upload and defines STRUCPP_SHIM_HAS_RETAIN only when the API is there. -// -// When it is not, these four exports are simply absent from the .so. That is a -// state the runtime already handles rather than a degraded one to apologise -// for: image_tables.cpp resolves all four as OPTIONAL symbols, and -// plc_retain_init() stands the store down and says so in the log. Retained -// variables then behave as NON_RETAIN, exactly as they did before these exports -// existed. -// -// Anything added here that touches strucpp::retain or retain_layout_hash MUST -// stay inside this #ifdef and be mirrored into retain_probe.cpp — a probe that -// checks less than the shim uses would pass for a header set that cannot -// actually build. Tests pin both halves -// (tests/pytest/compile/test_retain_capability_probe.py). -// --------------------------------------------------------------------------- #ifdef STRUCPP_SHIM_HAS_RETAIN -// --------------------------------------------------------------------------- -// The WALK lives here, inside the .so, because that is where the debug tables -// and `handle_read` / `handle_write` are. The runtime is built once and loads -// many .so files, so it cannot reach `strucpp::retain` by mangled name — and -// re-implementing the blob format on its side would put two copies of a wire -// format in two repos, which is exactly the drift `iec_retain.hpp` exists to -// prevent. -// -// What the runtime DOES own is the write path, which is why unpack takes a -// callback instead of using `handle_write` directly: a retained variable may -// also be LOCATED (`VAR RETAIN x AT %MW10`), and poking such a leaf's IECVar -// is undone by the next copy-in from the process image. The runtime passes a -// thunk that routes through `runtime_external_write`, which knows to send a -// located leaf through the image journal. Reads need no such care — a read -// sees whatever the last copy-in left — so pack uses `handle_read` here. -// --------------------------------------------------------------------------- - static uint16_t retain_read_leaf(uint8_t arr, uint16_t elem, uint8_t* dest) { return strucpp::debug::handle_read(arr, elem, dest); } @@ -218,15 +121,9 @@ extern "C" size_t strucpp_retain_pack(uint8_t* out, size_t cap) { return strucpp::retain::pack(out, cap, retain_read_leaf, retain_size_leaf); } -/** - * Restore every retained leaf, writing through the runtime's own callback. - * - * Returns `strucpp::retain::LoadResult` as a byte. Anything but 0 (Ok) means - * nothing was written and every variable keeps its declared initial value — - * the correct outcome for a corrupt or stale store, since a machine starting - * from its defaults is recoverable and one starting from plausible-looking - * garbage is not. - */ +/* Restore every retained leaf via write_leaf. Returns + * strucpp::retain::LoadResult as a byte; non-zero means nothing was + * written and variables keep their declared initial values. */ extern "C" uint8_t strucpp_retain_unpack( const uint8_t* blob, size_t len, diff --git a/docs/JOURNAL_BUFFER_ARCHITECTURE.md b/docs/JOURNAL_BUFFER_ARCHITECTURE.md deleted file mode 100644 index 4cba0e3e..00000000 --- a/docs/JOURNAL_BUFFER_ARCHITECTURE.md +++ /dev/null @@ -1,596 +0,0 @@ -# Journal Buffer Architecture - -## Overview - -This document describes the architecture for a journaled buffer system to handle plugin writes to OpenPLC image tables. The journal buffer provides a race-condition-free mechanism for multiple plugins (both native and Python) to write to shared I/O buffers with deterministic conflict resolution. - -## Problem Statement - -### Current Architecture Issues - -The current plugin architecture has several limitations when multiple plugins need to write to the same image table locations: - -1. **Race Conditions**: Plugins run in separate threads and can write to buffers at arbitrary times, causing race conditions. - -2. **Zero vs. Uninitialized Ambiguity**: When a plugin copies its buffer to the core image tables, there's no way to distinguish between "intentionally wrote zero" and "never touched this location." - -3. **Asynchronous Plugin Timing**: Python plugins run in their own threads without `cycle_start`/`cycle_end` hooks, so their writes can occur at any point relative to the PLC scan cycle. - -4. **No Conflict Resolution**: When two plugins write to the same location, the outcome depends on thread timing, leading to unpredictable behavior. - -### Example Scenario - -Consider a system with both S7Comm (native) and Modbus Master (Python) plugins: - -``` -Timeline: -|---- PLC Scan Cycle N ----|--- Sleep ---|---- PLC Scan Cycle N+1 ----| - cycle_start cycle_end cycle_start cycle_end - │ │ │ │ - S7→OpenPLC OpenPLC→S7 S7→OpenPLC OpenPLC→S7 - ↑ - Overwrites! - │ - Modbus Master writes - new input value here -``` - -1. **cycle_end N**: OpenPLC inputs (value X) → S7 buffer -2. **Sleep period**: Modbus Master reads remote I/O, writes new value Y to OpenPLC inputs -3. **cycle_start N+1**: S7 buffer (still has old value X) → OpenPLC inputs (overwrites Y with X!) - -## Proposed Solution: Journal Buffer - -### Core Concept - -Instead of plugins writing directly to image tables, all writes go through a journal buffer. The journal is applied atomically at the start of each PLC scan cycle, with "last writer wins" semantics based on write sequence. - -``` -┌─────────────────────────────────────────────────────────────────┐ -│ PLC Scan Cycle │ -├─────────────────────────────────────────────────────────────────┤ -│ │ -│ ┌─────────────┐ ┌─────────────┐ │ -│ │ JOURNAL │ ──── Apply ────────► │ CORE │ │ -│ │ BUFFER │ (cycle_start) │ IMAGE │ │ -│ │ │ │ TABLES │ │ -│ │ seq=1: ... │ │ │ │ -│ │ seq=2: ... │ │ bool_input │ │ -│ │ seq=3: ... │ │ bool_output │ │ -│ │ seq=4: ... │ │ int_input │ │ -│ └──────▲──────┘ │ ... │ │ -│ │ └──────┬──────┘ │ -│ │ │ │ -│ WRITES READS │ -│ (journal) (direct) │ -│ │ │ │ -│ ┌──────┴────────────────────────────────────────▼──────┐ │ -│ │ │ │ -│ │ ┌─────────┐ ┌─────────┐ ┌─────────┐ ┌─────────┐│ │ -│ │ │Plugin A │ │Plugin B │ │Plugin C │ │Plugin D ││ │ -│ │ │(S7Comm) │ │(Modbus) │ │(Python) │ │(Native) ││ │ -│ │ └─────────┘ └─────────┘ └─────────┘ └─────────┘│ │ -│ │ │ │ -│ └───────────────────────────────────────────────────────┘ │ -└─────────────────────────────────────────────────────────────────┘ -``` - -### Key Design Principles - -1. **Writes Go to Journal**: All plugin writes are recorded in a journal buffer with a sequence number. - -2. **Reads Are Direct**: Plugins read directly from image tables, getting the most recent applied values. - -3. **Atomic Application**: The entire journal is applied at `cycle_start`, before plugin hooks run. - -4. **Last Writer Wins**: Entries are applied in sequence order. If multiple plugins write to the same location, the one with the highest sequence number wins. - -5. **Sequence Reset Per Cycle**: The sequence counter resets to zero when the journal is cleared after application. - -## Detailed Architecture - -### Journal Entry Structure - -```c -typedef struct { - uint32_t sequence; /* Auto-increment, determines apply order */ - uint8_t buffer_type; /* journal_buffer_type_t enum */ - uint8_t bit_index; /* For bool types: 0-7, for others: 0xFF */ - uint16_t index; /* Buffer array index */ - uint64_t value; /* Value to write (sized for largest type) */ -} journal_entry_t; -``` - -**Size**: 16 bytes per entry (with padding for alignment, may be 20 bytes) - -### Buffer Type Enumeration - -```c -typedef enum { - JOURNAL_BOOL_INPUT = 0, - JOURNAL_BOOL_OUTPUT, - JOURNAL_BOOL_MEMORY, - JOURNAL_BYTE_INPUT, - JOURNAL_BYTE_OUTPUT, - JOURNAL_INT_INPUT, - JOURNAL_INT_OUTPUT, - JOURNAL_INT_MEMORY, - JOURNAL_DINT_INPUT, - JOURNAL_DINT_OUTPUT, - JOURNAL_DINT_MEMORY, - JOURNAL_LINT_INPUT, - JOURNAL_LINT_OUTPUT, - JOURNAL_LINT_MEMORY, - JOURNAL_TYPE_COUNT -} journal_buffer_type_t; -``` - -### Static Journal Buffer - -```c -#define JOURNAL_MAX_ENTRIES 40960 - -static journal_entry_t g_entries[JOURNAL_MAX_ENTRIES]; -static size_t g_count = 0; -static uint32_t g_next_sequence = 0; -static pthread_mutex_t g_journal_mutex = PTHREAD_MUTEX_INITIALIZER; -``` - -**Memory Usage**: ~20KB for 1024 entries (fixed, no dynamic allocation) - -### Thread Safety Model - -Two mutexes are involved: - -1. **Journal Mutex** (`g_journal_mutex`): Protects journal buffer state (entries, count, sequence) -2. **Image Table Mutex** (`image_mutex`): Protects core image tables (existing runtime mutex) - -**Lock Ordering** (to prevent deadlock): Always acquire `image_mutex` before `journal_mutex` when both are needed. - -### Write Flow - -``` -Plugin calls journal_write_*() - │ - ▼ -┌─────────────────────────┐ -│ Acquire journal_mutex │ -└───────────┬─────────────┘ - │ - ▼ - ┌───────────────┐ - │ Journal Full? │ - └───────┬───────┘ - │ - Yes │ No - ┌───────┴───────┐ - │ │ - ▼ │ -┌─────────────┐ │ -│ Emergency │ │ -│ Flush │ │ -│ (see below) │ │ -└──────┬──────┘ │ - │ │ - └─────┬──────┘ - │ - ▼ -┌─────────────────────────┐ -│ Add entry with │ -│ next sequence number │ -└───────────┬─────────────┘ - │ - ▼ -┌─────────────────────────┐ -│ Release journal_mutex │ -└─────────────────────────┘ -``` - -### Apply Flow (at cycle_start) - -``` -PLC Cycle Manager calls journal_apply_and_clear() -(image_mutex already held) - │ - ▼ -┌─────────────────────────┐ -│ Acquire journal_mutex │ -└───────────┬─────────────┘ - │ - ▼ -┌─────────────────────────┐ -│ For each entry in order:│ -│ Apply to image table │ -└───────────┬─────────────┘ - │ - ▼ -┌─────────────────────────┐ -│ Reset count = 0 │ -│ Reset sequence = 0 │ -└───────────┬─────────────┘ - │ - ▼ -┌─────────────────────────┐ -│ Release journal_mutex │ -└─────────────────────────┘ -``` - -### Emergency Flush - -When the journal buffer is full and a new write is attempted: - -``` -(Already holding journal_mutex) - │ - ▼ -┌─────────────────────────┐ -│ Release journal_mutex │ ← Prevent deadlock -└───────────┬─────────────┘ - │ - ▼ -┌─────────────────────────┐ -│ Acquire image_mutex │ ← Lock ordering -└───────────┬─────────────┘ - │ - ▼ -┌─────────────────────────┐ -│ Acquire journal_mutex │ -└───────────┬─────────────┘ - │ - ▼ -┌─────────────────────────┐ -│ Apply all entries │ -│ Clear journal │ -└───────────┬─────────────┘ - │ - ▼ -┌─────────────────────────┐ -│ Release image_mutex │ -└───────────┬─────────────┘ - │ - ▼ -(Continue with write, still holding journal_mutex) -``` - -## Public API - -### C API (for native plugins) - -```c -/* Initialize the journal buffer system */ -int journal_init(const journal_buffer_ptrs_t* buffer_ptrs); - -/* Cleanup journal buffer resources */ -void journal_cleanup(void); - -/* Write functions */ -int journal_write_bool(journal_buffer_type_t type, uint16_t index, - uint8_t bit, bool value); -int journal_write_byte(journal_buffer_type_t type, uint16_t index, - uint8_t value); -int journal_write_int(journal_buffer_type_t type, uint16_t index, - uint16_t value); -int journal_write_dint(journal_buffer_type_t type, uint16_t index, - uint32_t value); -int journal_write_lint(journal_buffer_type_t type, uint16_t index, - uint64_t value); - -/* Apply pending writes (called at cycle_start, image_mutex must be held) */ -void journal_apply_and_clear(void); - -/* Diagnostics */ -size_t journal_pending_count(void); -bool journal_is_initialized(void); -``` - -### Python API (for Python plugins) - -The existing `SafeBufferAccess` class will be extended to route writes through the journal: - -```python -class SafeBufferAccess: - def write_bool(self, buf_type: int, index: int, bit: int, value: bool) -> bool: - """Write boolean value to journal buffer.""" - return self._journal_write_bool(buf_type, index, bit, value) - - def write_int(self, buf_type: int, index: int, value: int) -> bool: - """Write 16-bit integer value to journal buffer.""" - return self._journal_write_int(buf_type, index, value) - - # ... similar for other types - - def read_bool(self, buf_type: int, index: int, bit: int) -> Optional[bool]: - """Read boolean value directly from image table (unchanged).""" - return self._direct_read_bool(buf_type, index, bit) -``` - -## Integration Points - -### 1. PLC State Manager Integration - -In `plc_state_manager.c`, add journal application at cycle start: - -```c -void plc_cycle_start(void) { - /* Acquire image table mutex */ - pthread_mutex_lock(&image_mutex); - - /* Apply journal entries FIRST, before any plugin hooks */ - journal_apply_and_clear(); - - /* Call plugin cycle_start hooks */ - plugin_driver_cycle_start(); - - /* ... rest of cycle start logic */ -} -``` - -### 2. Runtime Initialization - -In runtime initialization, set up journal buffer pointers: - -```c -int runtime_init(void) { - /* ... existing init code ... */ - - /* Initialize journal buffer */ - journal_buffer_ptrs_t ptrs = { - .bool_input = bool_input, - .bool_output = bool_output, - .bool_memory = bool_memory, - .byte_input = byte_input, - .byte_output = byte_output, - .int_input = int_input, - .int_output = int_output, - .int_memory = int_memory, - .dint_input = dint_input, - .dint_output = dint_output, - .dint_memory = dint_memory, - .lint_input = lint_input, - .lint_output = lint_output, - .lint_memory = lint_memory, - .buffer_size = BUFFER_SIZE, - .image_mutex = &bufferLock /* existing buffer mutex */ - }; - - if (journal_init(&ptrs) != 0) { - log_error("Failed to initialize journal buffer"); - return -1; - } - - /* ... rest of init ... */ -} -``` - -### 3. Native Plugin Migration - -Native plugins using direct buffer access need to migrate to journal writes: - -**Before (direct write):** -```c -void sync_to_openplc(void) { - IEC_BOOL** arr = runtime_args->bool_output; - if (arr[index][bit] != NULL) { - *arr[index][bit] = value; /* Direct write */ - } -} -``` - -**After (journal write):** -```c -void sync_to_openplc(void) { - journal_write_bool(JOURNAL_BOOL_OUTPUT, index, bit, value); -} -``` - -### 4. Python Plugin Migration - -Python plugins using `SafeBufferAccess` will automatically use the journal when the `SafeBufferAccess` implementation is updated. No changes required in plugin code. - -## Conflict Resolution Examples - -### Example 1: Two Plugins Write Same Location - -``` -Time 0ms: Plugin A writes %QX0.0 = TRUE → seq=1 -Time 5ms: Plugin B writes %QX0.0 = FALSE → seq=2 -Time 10ms: cycle_start - apply journal - -Apply order: - 1. seq=1: %QX0.0 = TRUE - 2. seq=2: %QX0.0 = FALSE ← This wins (last writer) - -Final value: %QX0.0 = FALSE -``` - -### Example 2: Modbus and S7 Conflict - -``` -Time 0ms: S7 client writes %IX0.0 = TRUE → seq=1 -Time 50ms: Modbus Master reads remote I/O, writes %IX0.0 = FALSE → seq=2 -Time 100ms: cycle_start - apply journal - -Result: %IX0.0 = FALSE (Modbus Master wins because it wrote later) -``` - -### Example 3: Emergency Flush - -``` -Time 0ms: Writes 1-1023 accumulate in journal -Time 50ms: Write 1024 triggers emergency flush - - Apply seq 1-1023 to image tables - - Clear journal - - Add write 1024 as seq=0 -Time 100ms: Writes 1-500 accumulate -Time 200ms: cycle_start - apply journal (seq 0-500) -``` - -## Performance Considerations - -### Memory Usage - -- **Journal Buffer**: 1024 entries × ~20 bytes = ~20KB (static allocation) -- **No per-plugin buffers needed**: Single shared journal -- **Total overhead**: Minimal compared to current architecture - -### Latency - -- **Write latency**: Acquiring journal mutex (~microseconds) -- **Apply latency**: Iterating 1024 entries maximum (~milliseconds) -- **Read latency**: Unchanged (direct buffer access) - -### Throughput - -- **Maximum writes per cycle**: 40960 (configurable via `JOURNAL_MAX_ENTRIES`); drain cost follows the writes actually made, not the capacity -- **Emergency flush**: Handles overflow gracefully without data loss - -## Implementation Phases - -### Phase 1: Core Journal Buffer (C Implementation) - -**Files to create:** -- `core/src/plc_app/journal_buffer.h` - Public API header -- `core/src/plc_app/journal_buffer.c` - Implementation - -**Tasks:** -1. Implement journal entry structure and static buffer -2. Implement write functions with mutex protection -3. Implement `journal_apply_and_clear()` function -4. Implement emergency flush logic -5. Add unit tests for journal operations - -**Estimated effort**: 2-3 days - -### Phase 2: Runtime Integration - -**Files to modify:** -- `core/src/plc_app/plc_state_manager.c` - Add journal apply at cycle_start -- `core/src/plc_app/main.c` or equivalent - Initialize journal at startup - -**Tasks:** -1. Initialize journal buffer during runtime startup -2. Call `journal_apply_and_clear()` at cycle_start -3. Cleanup journal buffer at runtime shutdown -4. Integration testing with PLC scan cycle - -**Estimated effort**: 1-2 days - -### Phase 3: Python API Extension - -**Files to modify:** -- `core/src/drivers/plugins/python/shared/safe_buffer_access.py` - Route writes to journal -- `core/src/drivers/plugins/python/shared/__init__.py` - Export new functions - -**Tasks:** -1. Add ctypes bindings for journal write functions -2. Modify `SafeBufferAccess` write methods to use journal -3. Keep read methods unchanged (direct buffer access) -4. Update documentation - -**Estimated effort**: 1-2 days - -### Phase 4: Native Plugin Migration - -**Files to modify:** -- `core/src/drivers/plugins/native/s7comm/s7comm_plugin.cpp` - Use journal writes -- Any other native plugins with direct buffer writes - -**Tasks:** -1. Replace direct buffer writes with `journal_write_*()` calls -2. Remove plugin-specific double-buffering logic (no longer needed) -3. Simplify `cycle_start`/`cycle_end` hooks -4. Integration testing - -**Estimated effort**: 1-2 days per plugin - -### Phase 5: Python Plugin Migration - -**Files to modify:** -- `core/src/drivers/plugins/python/modbus_master/modbus_master_plugin.py` -- `core/src/drivers/plugins/python/modbus_slave/simple_modbus.py` -- Any other Python plugins - -**Tasks:** -1. Verify plugins work with new `SafeBufferAccess` (should be automatic) -2. Integration testing -3. Performance testing - -**Estimated effort**: 1-2 days - -### Phase 6: Testing and Documentation - -**Tasks:** -1. End-to-end testing with multiple plugins -2. Stress testing (high write volume, emergency flush scenarios) -3. Performance benchmarking -4. Update plugin development documentation -5. Update architecture documentation - -**Estimated effort**: 2-3 days - -## Total Implementation Estimate - -| Phase | Effort | -|-------|--------| -| Phase 1: Core Journal Buffer | 2-3 days | -| Phase 2: Runtime Integration | 1-2 days | -| Phase 3: Python API Extension | 1-2 days | -| Phase 4: Native Plugin Migration | 1-2 days per plugin | -| Phase 5: Python Plugin Migration | 1-2 days | -| Phase 6: Testing and Documentation | 2-3 days | -| **Total** | **~10-15 days** | - -## Future Enhancements (Out of Scope) - -The following features are explicitly NOT part of this design but could be added later: - -1. **Write Coalescing**: Merge multiple writes to the same location within a cycle -2. **Read-After-Write Consistency**: Allow plugins to read their own pending writes -3. **Journal Persistence**: Save journal to disk for crash recovery -4. **Journal Replay**: Replay journal for debugging/simulation -5. **Per-Plugin Statistics**: Track write counts per plugin for diagnostics - -## Appendix A: Comparison with Alternative Approaches - -### Alternative 1: Per-Plugin Shadow Buffers - -Each plugin gets its own copy of image tables. Merge at cycle boundaries. - -**Pros**: Complete isolation -**Cons**: High memory usage (N copies), complex merge logic, priority configuration needed - -### Alternative 2: Timestamp-Based Conflict Resolution - -Each write includes a timestamp. Latest timestamp wins. - -**Pros**: Most recent data always wins -**Cons**: Clock synchronization issues, storage overhead, complexity - -### Alternative 3: Priority-Based System - -Each plugin has a priority. Higher priority wins conflicts. - -**Pros**: Deterministic, configurable -**Cons**: Requires priority configuration, may not reflect actual "freshness" of data - -### Why Journal Buffer? - -The journal buffer approach was chosen because: - -1. **No configuration needed**: "Last writer wins" is intuitive -2. **Memory efficient**: Single shared buffer -3. **Simple API**: Just replace write calls -4. **Deterministic**: Same sequence = same result -5. **Debuggable**: Can log/inspect journal contents - -## Appendix B: Glossary - -| Term | Definition | -|------|------------| -| **Image Table** | Core OpenPLC buffer arrays (bool_input, int_output, etc.) | -| **Journal Entry** | Single write record with sequence, type, index, and value | -| **Journal Buffer** | Static array holding pending write entries | -| **Sequence Number** | Auto-incrementing counter determining apply order | -| **Emergency Flush** | Immediate journal application when buffer is full | -| **Apply** | Process of writing journal entries to image tables | -| **Cycle Start** | Beginning of PLC scan cycle, when journal is applied | diff --git a/docs/LOGGING_NORMALIZATION_PLAN.md b/docs/LOGGING_NORMALIZATION_PLAN.md deleted file mode 100644 index cab593c6..00000000 --- a/docs/LOGGING_NORMALIZATION_PLAN.md +++ /dev/null @@ -1,337 +0,0 @@ -# Logging Normalization Plan - -**Goal**: Unify all log output to a human-readable format for stdout while maintaining JSON for API responses. - -**Target Format**: -``` -[2024-01-30 12:00:00] [INFO] message -[2024-01-30 12:00:00] [WARN] warning message -[2024-01-30 12:00:00] [ERROR] error message -``` - ---- - -## Current State Analysis - -### Problem - -The codebase has **two parallel output mechanisms**: - -1. **Formal logging system** (JSON output): - - C side: `log_info()`, `log_error()`, etc. in `core/src/plc_app/utils/log.c` - - Python side: `logger.info()`, `logger.error()`, etc. - - Goes through UNIX socket, printed as JSON to stdout - -2. **Direct printf/print calls** (human-readable): - - C side: `printf()`, `fprintf(stderr)` scattered throughout code - - Python side: `print()` statements in various modules - -### Current Output (Mixed) - -``` -[PLUGIN]: Creating Python capsule for args -[PLUGIN]: Python capsule created successfully -[OPCUA INFO] OPC UA Plugin initializing... -{"timestamp": "1769785673", "level": "INFO", "message": "Logging initialized", "id": 8} -{"timestamp": "1769785673", "level": "INFO", "message": "Buffer accessor created", "id": 9} -``` - ---- - -## Implementation Plan - -### Phase 1: Update Python Logging System - -**Goal**: Change stdout output from JSON to human-readable format. - -#### Files to Modify - -| File | Change | -|------|--------| -| `webserver/logger/formatter.py` | Add `HumanReadableFormatter` class | -| `webserver/logger/__init__.py` | Use `HumanReadableFormatter` for `StreamHandler` (line 43) | -| `webserver/logger/logger.py` | Use `HumanReadableFormatter` for `StreamHandler` (line 19) | - -#### HumanReadableFormatter Implementation - -```python -class HumanReadableFormatter(logging.Formatter): - """Format log records as human-readable strings for stdout.""" - - def format(self, record): - msg = record.getMessage() - - # Try to detect pre-formatted JSON (from C runtime) - try: - log_entry = json.loads(msg) - timestamp = log_entry.get("timestamp", "") - level = log_entry.get("level", record.levelname) - message = log_entry.get("message", msg) - - # Convert Unix timestamp to human-readable - if timestamp and str(timestamp).isdigit(): - dt = datetime.fromtimestamp(int(timestamp), tz=timezone.utc) - timestamp = dt.strftime("%Y-%m-%d %H:%M:%S") - elif timestamp: - # Handle ISO 8601 format - dt = datetime.fromisoformat(timestamp.replace('Z', '+00:00')) - timestamp = dt.strftime("%Y-%m-%d %H:%M:%S") - else: - timestamp = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M:%S") - - except (json.JSONDecodeError, ValueError): - # Not JSON - use record fields directly - timestamp = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M:%S") - level = record.levelname - message = msg - - return f"[{timestamp}] [{level}] {message}" -``` - -#### Changes in `__init__.py` (line 43) - -```python -# Current: -stream_handler.setFormatter(JsonFormatter()) - -# New: -stream_handler.setFormatter(HumanReadableFormatter()) -``` - -#### Changes in `logger.py` (line 19) - -```python -# Current: -stream_handler.setFormatter(JsonFormatter()) - -# New: -stream_handler.setFormatter(HumanReadableFormatter()) -``` - -**Note**: `BufferHandler` keeps using `JsonFormatter` for API responses. - ---- - -### Phase 2: Convert C printf/fprintf to log_*() calls - -#### Files and Occurrence Count - -| File | printf | fprintf | Total | -|------|--------|---------|-------| -| `core/src/drivers/plugin_driver.c` | 41 | 32 | 73 | -| `core/src/drivers/plugin_config.c` | 2 | 0 | 2 | -| `core/src/drivers/plugins/native/plugin_logger.c` | 1 | 3 | 4 | -| `core/src/plc_app/plc_main.c` | 0 | 1 | 1 | -| `core/src/plc_app/journal_buffer.c` | 0 | 2 | 2 | -| `core/src/plc_app/utils/watchdog.c` | 0 | 1 | 1 | -| `core/src/plc_app/python_loader.c` | 0 | 1 | 1 | -| **Total** | **44** | **40** | **84** | - -**Skip** (not production code): -- `core/src/drivers/plugins/native/examples/test_plugin_loader.c` (test file) -- `core/src/drivers/README.md` (documentation examples) - -#### Conversion Rules - -| Original Pattern | Converted To | -|------------------|--------------| -| `printf("[PLUGIN]: %s\n", msg)` | `log_info("%s", msg)` | -| `printf("[PLUGIN]: ... successfully\n")` | `log_info("...")` | -| `fprintf(stderr, "[PLUGIN]: Error...\n")` | `log_error("...")` | -| `fprintf(stderr, "Failed to...\n")` | `log_error("Failed to...")` | -| `fprintf(stderr, "Warning:...\n")` | `log_warn("...")` | - -#### Main File: `core/src/drivers/plugin_driver.c` - -This file has the most occurrences. Key conversions: - -```c -// Current: -printf("[PLUGIN]: Config file %s not found, copying from plugins_default.conf\n", config_file); - -// Converted: -log_info("Config file %s not found, copying from plugins_default.conf", config_file); -``` - -```c -// Current: -fprintf(stderr, "[PLUGIN]: Error - driver is NULL\n"); - -// Converted: -log_error("Error - driver is NULL"); -``` - -```c -// Current: -printf("[PLUGIN]: Plugin %s started successfully.\n", plugin->config.name); - -// Converted: -log_info("Plugin %s started successfully", plugin->config.name); -``` - -**Note**: Need to include `log.h` header in files that don't already have it. - ---- - -### Phase 3: Convert Python print() to logger.*() calls - -#### Files and Occurrence Count - -| File | print() calls | Notes | -|------|---------------|-------| -| `webserver/plugin_config_model.py` | 6 | Config file operations | -| `webserver/credentials.py` | 10 | Certificate generation | -| `webserver/config.py` | 2 | Env validation | -| `webserver/app.py` | 2 | Platform detection | -| `core/src/drivers/plugins/python/modbus_master/modbus_master_memory.py` | ~20 | Error messages | -| **Total** | **~40** | | - -#### Conversion Rules - -| Original Pattern | Converted To | -|------------------|--------------| -| `print(f"[PLUGIN]: {msg}")` | `logger.info(msg)` | -| `print(f"[PLUGIN]: Failed to {x}")` | `logger.error(f"Failed to {x}")` | -| `print(f"Warning: {msg}")` | `logger.warning(msg)` | -| `print(f"Error: {msg}")` | `logger.error(msg)` | -| `print(f"Successfully {x}")` | `logger.info(f"Successfully {x}")` | - -#### Example: `webserver/plugin_config_model.py` - -```python -# Current: -print(f"[PLUGIN]: Config file {file_path} not found, copying from {default_file}") - -# Converted: -logger.info(f"Config file {file_path} not found, copying from {default_file}") -``` - -```python -# Current: -print(f"[PLUGIN]: Failed to copy {default_file}: {e}") - -# Converted: -logger.error(f"Failed to copy {default_file}: {e}") -``` - -#### Example: `webserver/credentials.py` - -```python -# Current: -print(f"Generating self-signed certificate for {self.hostname}...") - -# Converted: -logger.info(f"Generating self-signed certificate for {self.hostname}...") -``` - -```python -# Current: -print(f"Certificate saved to {cert_path}") - -# Converted: -logger.info(f"Certificate saved to {cert_path}") -``` - -**Note**: Each file needs to import the logger: -```python -from webserver.logger import get_logger -logger, _ = get_logger(__name__) -``` - ---- - -### Phase 4: Update Plugin Fallback Logging - -#### File: `core/src/drivers/plugins/python/opcua/opcua_logging.py` - -The fallback `print()` statements should match the standard format when runtime logging isn't available. - -```python -# Current fallback: -print(f"[OPCUA INFO] {message}", file=sys.stdout) - -# Updated fallback (matching standard format): -timestamp = datetime.now().strftime("%Y-%m-%d %H:%M:%S") -print(f"[{timestamp}] [INFO] {message}", file=sys.stdout) -``` - -Apply to all 4 fallback print statements (info, warn, error, debug). - ---- - -## Implementation Order - -1. **Phase 1** - Python logging formatter (3 files) - - Lowest risk, immediate visible improvement - - Test: Run runtime, verify JSON messages now appear human-readable - -2. **Phase 3** - Python print conversions (~40 calls across 5 files) - - Medium effort, easy to test - - Test: Run runtime, verify print messages go through logger - -3. **Phase 2** - C printf/fprintf conversions (~84 calls across 7 files) - - Highest effort, requires rebuild - - Test: Run runtime, verify all plugin messages use logging system - -4. **Phase 4** - Plugin fallback updates (1 file) - - Final cleanup for edge cases - - Test: Run with logging accessor unavailable - ---- - -## Expected Result - -**Before**: -``` -[PLUGIN]: Creating Python capsule for args -[PLUGIN]: Python capsule created successfully -[OPCUA INFO] OPC UA Plugin initializing... -{"timestamp": "1769785673", "level": "INFO", "message": "Logging initialized", "id": 8} -{"timestamp": "1769785673", "level": "INFO", "message": "Buffer accessor created", "id": 9} -[PLUGIN]: Skipping disabled plugin: s7comm -``` - -**After**: -``` -[2024-01-30 12:00:00] [INFO] Creating Python capsule for args -[2024-01-30 12:00:00] [INFO] Python capsule created successfully -[2024-01-30 12:00:00] [INFO] OPC UA Plugin initializing... -[2024-01-30 12:00:00] [INFO] Logging initialized -[2024-01-30 12:00:00] [INFO] Buffer accessor created -[2024-01-30 12:00:00] [INFO] Skipping disabled plugin: s7comm -``` - -**API Response** (`/api/runtime-logs`) - unchanged, still JSON: -```json -{ - "runtime-logs": [ - {"id": 1, "timestamp": "2024-01-30T12:00:00+00:00", "level": "INFO", "message": "..."} - ] -} -``` - ---- - -## Testing Checklist - -- [ ] Phase 1: JSON logs now display as human-readable on stdout -- [ ] Phase 1: API `/api/runtime-logs` still returns JSON -- [ ] Phase 2: All C plugin messages go through logging system -- [ ] Phase 3: All Python print statements converted to logger -- [ ] Phase 4: Fallback logging matches standard format -- [ ] All timestamps are in UTC -- [ ] Log levels display correctly (INFO, WARN, ERROR, DEBUG) -- [ ] No duplicate messages (both printf and log_*) - ---- - -## Status - -- [x] Phase 1: Python Logging System (COMPLETED) -- [x] Phase 2: C printf/fprintf Conversions (COMPLETED) -- [x] Phase 3: Python print() Conversions (COMPLETED - webserver files) -- [x] Phase 4: Plugin Fallback Updates (COMPLETED) - -### Note on Python Plugin Files -Python plugin files use `PluginLogger` from `shared/plugin_logger.py` or `OpcuaLogger` -from `opcua/opcua_logging.py`. Both have been updated with timestamp-formatted fallbacks. diff --git a/docs/ethercat-ebpfcat-plugin-development-plan.md b/docs/ethercat-ebpfcat-plugin-development-plan.md deleted file mode 100644 index 61ffaeda..00000000 --- a/docs/ethercat-ebpfcat-plugin-development-plan.md +++ /dev/null @@ -1,788 +0,0 @@ -# Plano de Desenvolvimento - Plugin EtherCAT com ebpfcat (Runtime) - -**Produto:** OpenPLC Runtime v4 -**Baseado em:** Levantamento de Requisitos - Protocolo EtherCAT (v1.3) -**Data:** 29 de Janeiro de 2026 -**Tipo:** Plugin Python (ebpfcat) -**Status:** Alternativo ao plugin SOEM -**Revisao:** 2.0 - Arquitetura Editor-driven (discovery e parametrizacao no Editor) - ---- - -## 1. Visao Geral - -Este plano descreve uma implementacao **alternativa** do EtherCAT Master usando a biblioteca -ebpfcat. O objetivo e manter **compatibilidade total com o JSON de configuracao** gerado pelo -Editor, permitindo que o usuario escolha entre o backend SOEM (C/C++) ou ebpfcat (Python/eBPF) -sem modificar a configuracao. - -### 1.1 Separacao de Responsabilidades - -Assim como o plugin SOEM, o plugin ebpfcat **apenas executa** a configuracao recebida do Editor. -Toda a parametrizacao (discovery, ESI, couplers, modulos, channels, located vars) e feita no -Editor. - -``` -┌─────────────────────────────────────────────────────────────────┐ -│ OpenPLC Editor │ -│ (Configuracao completa EtherCAT) │ -│ │ -│ Discovery, ESI parsing, parametrizacao, │ -│ channels, located vars, PDO mapping, SDO config │ -└─────────────────────────────────┬────────────────────────────────┘ - │ - ▼ - ┌────────────────────────┐ - │ ethercat_config.json │ <-- Formato UNICO - │ (Gerado pelo Editor) │ (contrato) - └────────────────────────┘ - │ - ┌───────────────┴───────────────┐ - ▼ ▼ -┌──────────────────────────────┐ ┌──────────────────────────────┐ -│ Plugin SOEM (Nativo C/C++) │ │ Plugin ebpfcat (Python) │ -│ - Raw sockets │ │ - eBPF/XDP │ -│ - Qualquer NIC │ │ - NICs com suporte XDP │ -│ - Kernel 4.x+ │ │ - Kernel 5.x+ │ -│ - Mesmo JSON config │ │ - Mesmo JSON config │ -└──────────────────────────────┘ └──────────────────────────────┘ -``` - -### 1.2 Quando Usar Cada Backend - -| Cenario | Backend Recomendado | -|---------|---------------------| -| Producao geral | SOEM | -| Hardware variado | SOEM | -| ARM embarcado | SOEM | -| Cycle time < 500us | ebpfcat | -| Ambiente de pesquisa | ebpfcat | -| Integracao com EPICS | ebpfcat | - -### 1.3 Escopo do MVP (mesmo do SOEM) - -- Carregamento do JSON de configuracao gerado pelo Editor -- Validacao de topologia (slaves fisicos vs configuracao) -- Mapeamento de PDOs para variaveis do PLC conforme JSON -- Aplicacao de configuracoes SDO nos slaves -- Operacao estavel com cycle time de 4 ms -- Suporte a CoE (SDO e PDO) -- Diagnostico basico de status e erros - -### 1.4 Discovery Service (JA IMPLEMENTADO - Compartilhado) - -O Discovery Service e compartilhado com o plugin SOEM e ja esta implementado. -Ver documentacao completa em `docs/old_docs/ethercat-plugin-development-plan.md` secao 1.0.2. - ---- - -## 2. Arquitetura do Plugin - -### 2.1 Estrutura de Arquivos - -``` -core/src/drivers/plugins/python/ethercat_ebpf/ -├── __init__.py -├── plugin.py # Entry point (init, start_loop, stop_loop) -├── master.py # Wrapper do ebpfcat master -├── config_loader.py # Carrega ethercat_config.json (formato compartilhado) -├── pdo_mapper.py # Mapeia PDOs para buffers OpenPLC -├── sdo_handler.py # Operacoes SDO -├── diagnostics.py # Coleta de status e erros -├── state_machine.py # Gerencia estados dos slaves -├── devices/ # Device classes especificos -│ ├── __init__.py -│ ├── generic_io.py # Dispositivo I/O generico (DS401) -│ ├── beckhoff.py # Suporte especifico Beckhoff -│ └── digital_io.py # Modulos digitais -├── requirements.txt # Dependencias (ebpfcat, etc.) -└── README.md -``` - -### 2.2 Configuracao em plugins.conf - -``` -# Backend SOEM (nativo) - padrao -ethercat,./build/plugins/libethercat_plugin.so,1,1,./core/src/drivers/plugins/native/ethercat/ethercat_config.json, - -# Backend ebpfcat (Python) - alternativo -# ethercat_ebpf,./core/src/drivers/plugins/python/ethercat_ebpf/plugin.py,0,0,./core/src/drivers/plugins/native/ethercat/ethercat_config.json,./venvs/ethercat_ebpf -``` - -**Nota:** Ambos os plugins usam o **mesmo arquivo de configuracao** (`ethercat_config.json`), -gerado pelo Editor. O JSON segue o contrato definido em `docs/old_docs/ethercat-plugin-development-plan.md` -secao 3. - -### 2.3 Formato JSON de Configuracao - -O formato JSON e **identico** ao definido no plano SOEM. Ver secao 3 do documento -`docs/old_docs/ethercat-plugin-development-plan.md` para a especificacao completa. - -Resumo da estrutura (lista flat de slaves, alinhada com `ec_slave[]` da SOEM): -```json -[ - { - "name": "ethercat_master", - "protocol": "ETHERCAT", - "config": { - "master": { - "interface": "eth0", - "cycle_time_us": 1000, - "watchdog_timeout_cycles": 3, - "log_level": "info" - }, - "slaves": [ - { - "position": 1, - "name": "EK1100", - "type": "coupler", - "vendor_id": "0x00000002", - "product_code": "0x044c2c52", - "channels": [], - "sdo_configurations": [], - "rx_pdos": [], - "tx_pdos": [] - }, - { - "position": 2, - "name": "EL1008", - "type": "digital_input", - "vendor_id": "0x00000002", - "product_code": "0x03f03052", - "channels": [ ... ], - "sdo_configurations": [ ... ], - "rx_pdos": [ ... ], - "tx_pdos": [ ... ] - } - ], - "diagnostics": { ... } - } - } -] -``` - ---- - -## 3. Etapas de Desenvolvimento - -### Fase 1: Setup e Infraestrutura (2-3 semanas) - -#### Etapa 1.1: Ambiente e Dependencias -**Duracao estimada:** 3-4 dias - -**Tarefas:** -1. Criar estrutura de diretorios do plugin -2. Configurar virtual environment dedicado -3. Instalar ebpfcat e dependencias -4. Verificar requisitos de sistema (kernel, XDP) -5. Criar script de verificacao de compatibilidade - -**Arquivos:** -- `requirements.txt` -- `__init__.py` -- `scripts/check_ebpf_support.sh` - -**requirements.txt:** -``` -ebpfcat>=1.0.0 -``` - -**Criterio de Aceite:** -- Plugin carrega sem erros -- Dependencias instaladas -- Verificacao de compatibilidade funcional - -#### Etapa 1.2: Loader de Configuracao Compartilhada -**Duracao estimada:** 2-3 dias - -**Tarefas:** -1. Criar parser do JSON de configuracao (formato identico ao SOEM) -2. Validar formato (mesmo schema) -3. Converter para estruturas internas do ebpfcat -4. Implementar validacao estrutural - -**Arquivos:** -- `config_loader.py` - -**Implementacao:** -```python -# config_loader.py -import json -from dataclasses import dataclass, field -from typing import List, Optional -from pathlib import Path - - -@dataclass -class PdoEntry: - index: str - subindex: int - bit_length: int - name: str = "" - data_type: str = "BOOL" - - -@dataclass -class Pdo: - index: str - name: str = "" - entries: List[PdoEntry] = field(default_factory=list) - - -@dataclass -class SdoConfig: - index: str - subindex: int - value: any = 0 - data_type: str = "UINT16" - name: str = "" - description: str = "" - - -@dataclass -class Channel: - index: int - name: str - type: str - bit_length: int - iec_location: str - pdo_index: str - pdo_entry_index: str - pdo_entry_subindex: int - - -@dataclass -class Slave: - position: int - name: str - type: str - vendor_id: str - product_code: str - revision: str = "" - channels: List[Channel] = field(default_factory=list) - sdo_configurations: List[SdoConfig] = field(default_factory=list) - rx_pdos: List[Pdo] = field(default_factory=list) - tx_pdos: List[Pdo] = field(default_factory=list) - - -@dataclass -class MasterConfig: - interface: str - cycle_time_us: int = 1000 - watchdog_timeout_cycles: int = 3 - log_level: str = "info" - - -@dataclass -class DiagnosticsConfig: - log_connections: bool = True - log_data_access: bool = False - log_errors: bool = True - max_log_entries: int = 10000 - status_update_interval_ms: int = 500 - - -@dataclass -class EtherCATConfig: - master: MasterConfig - slaves: List[Slave] - diagnostics: DiagnosticsConfig - - -def load_config(config_path: str) -> EtherCATConfig: - """Carrega configuracao do JSON gerado pelo Editor.""" - with open(config_path, "r") as f: - data = json.load(f) - - # Formato padrao OpenPLC: array com primeiro elemento - plugin_data = data[0] if isinstance(data, list) else data - ecat = plugin_data.get("config", plugin_data) - - master = MasterConfig( - interface=ecat["master"]["interface"], - cycle_time_us=ecat["master"].get("cycle_time_us", 1000), - watchdog_timeout_cycles=ecat["master"].get("watchdog_timeout_cycles", 3), - log_level=ecat["master"].get("log_level", "info"), - ) - - # Lista flat de slaves (alinhada com ec_slave[] da SOEM) - slaves = [] - for slave_data in ecat.get("slaves", []): - channels = [ - Channel(**ch) for ch in slave_data.get("channels", []) - ] - sdos = [ - SdoConfig(**sdo) for sdo in slave_data.get("sdo_configurations", []) - ] - rx_pdos = [ - Pdo( - index=p["index"], - name=p.get("name", ""), - entries=[PdoEntry(**e) for e in p.get("entries", [])], - ) - for p in slave_data.get("rx_pdos", []) - ] - tx_pdos = [ - Pdo( - index=p["index"], - name=p.get("name", ""), - entries=[PdoEntry(**e) for e in p.get("entries", [])], - ) - for p in slave_data.get("tx_pdos", []) - ] - slaves.append(Slave( - position=slave_data["position"], - name=slave_data["name"], - type=slave_data["type"], - vendor_id=slave_data["vendor_id"], - product_code=slave_data["product_code"], - revision=slave_data.get("revision", ""), - channels=channels, - sdo_configurations=sdos, - rx_pdos=rx_pdos, - tx_pdos=tx_pdos, - )) - - diag_data = ecat.get("diagnostics", {}) - diagnostics = DiagnosticsConfig( - log_connections=diag_data.get("log_connections", True), - log_data_access=diag_data.get("log_data_access", False), - log_errors=diag_data.get("log_errors", True), - max_log_entries=diag_data.get("max_log_entries", 10000), - status_update_interval_ms=diag_data.get("status_update_interval_ms", 500), - ) - - return EtherCATConfig( - master=master, - slaves=slaves, - diagnostics=diagnostics, - ) -``` - -**Criterio de Aceite:** -- JSON parseado corretamente -- Estruturas identicas ao esperado pelo SOEM -- Validacao de campos obrigatorios - -#### Etapa 1.3: Plugin Entry Point -**Duracao estimada:** 3-4 dias - -**Tarefas:** -1. Implementar `init()` com PyCapsule -2. Implementar `start_loop()` e `stop_loop()` -3. Integrar com SafeBufferAccess -4. Integrar com sistema de logging -5. Carregar e validar JSON config - -**Arquivos:** -- `plugin.py` - -**Implementacao:** -```python -# plugin.py -""" -EtherCAT Master Plugin using ebpfcat. -Compativel com o mesmo JSON de configuracao do plugin SOEM. -O JSON e gerado pelo Editor com toda a parametrizacao. -""" -import threading -from typing import Optional - -from shared import ( - SafeBufferAccess, - safe_extract_runtime_args_from_capsule, - PluginLogger, - SafeLoggingAccess, -) - -from .config_loader import load_config, EtherCATConfig -from .master import EbpfcatMaster - -# Globais do plugin -_runtime_args = None -_buffer_accessor: Optional[SafeBufferAccess] = None -_logger: Optional[PluginLogger] = None -_config: Optional[EtherCATConfig] = None -_master: Optional[EbpfcatMaster] = None -_master_thread: Optional[threading.Thread] = None -_stop_event = threading.Event() - - -def init(args_capsule) -> bool: - """Inicializa o plugin com argumentos do runtime.""" - global _runtime_args, _buffer_accessor, _logger, _config - - _runtime_args, error_msg = safe_extract_runtime_args_from_capsule(args_capsule) - if _runtime_args is None: - print(f"[ETHERCAT_EBPF] Failed to extract runtime args: {error_msg}") - return False - - logging_accessor = SafeLoggingAccess(_runtime_args) - _logger = PluginLogger() - _logger.initialize(logging_accessor) - _logger.info("Initializing EtherCAT ebpfcat plugin") - - _buffer_accessor = SafeBufferAccess(_runtime_args) - if not _buffer_accessor.is_valid: - _logger.error("Failed to create buffer accessor") - return False - - # Carregar JSON config gerado pelo Editor - config_path, err = _buffer_accessor.get_config_path() - if err: - _logger.error(f"Failed to get config path: {err}") - return False - - try: - _config = load_config(config_path) - _logger.info( - f"Loaded config: interface={_config.master.interface}, " - f"cycle_time={_config.master.cycle_time_us}us, " - f"slaves={len(_config.slaves)}" - ) - except Exception as e: - _logger.error(f"Failed to load config: {e}") - return False - - _logger.info("EtherCAT ebpfcat plugin initialized successfully") - return True - - -def start_loop() -> bool: - """Inicia o loop do master EtherCAT.""" - global _master, _master_thread - - if _config is None or _buffer_accessor is None: - _logger.error("Plugin not initialized") - return False - - _logger.info("Starting EtherCAT master loop") - _stop_event.clear() - - try: - _master = EbpfcatMaster( - config=_config, - buffer_accessor=_buffer_accessor, - logger=_logger, - ) - - # Inicializar: scan, validar topologia, aplicar SDOs - if not _master.initialize(): - _logger.error("Failed to initialize EtherCAT master") - return False - - _master_thread = threading.Thread( - target=_master_loop, - name="ethercat_ebpf_master", - daemon=True, - ) - _master_thread.start() - - _logger.info("EtherCAT master loop started") - return True - - except Exception as e: - _logger.error(f"Failed to start master: {e}") - return False - - -def _master_loop(): - """Loop principal do master (executa em thread separada).""" - while not _stop_event.is_set(): - try: - _master.run_cycle() - except Exception as e: - _logger.error(f"Error in master cycle: {e}") - - -def stop_loop() -> bool: - """Para o loop do master.""" - global _master, _master_thread - - _logger.info("Stopping EtherCAT master loop") - _stop_event.set() - - if _master_thread is not None: - _master_thread.join(timeout=5.0) - _master_thread = None - - if _master is not None: - _master.shutdown() - _master = None - - _logger.info("EtherCAT master loop stopped") - return True - - -def cleanup(): - """Limpa recursos do plugin.""" - global _runtime_args, _buffer_accessor, _logger, _config - - if _logger: - _logger.info("Cleaning up EtherCAT ebpfcat plugin") - - _runtime_args = None - _buffer_accessor = None - _logger = None - _config = None -``` - -**Criterio de Aceite:** -- Plugin carrega e inicializa -- Logging funciona -- Configuracao carregada do JSON - ---- - -### Fase 2: Integracao com ebpfcat (3-4 semanas) - -#### Etapa 2.1: Wrapper do Master ebpfcat -**Duracao estimada:** 1-2 semanas - -**Tarefas:** -1. Criar classe wrapper para ebpfcat.Master -2. Implementar scan de rede e validacao de slaves contra JSON -3. Implementar inicializacao de slaves -4. Gerenciar ciclo de comunicacao - -**Arquivos:** -- `master.py` - -**Fluxo de inicializacao (mesmo do SOEM):** -``` -initialize(): - 1. Criar ebpfcat.Master na interface do JSON - 2. Scan da rede - 3. Validar slaves: slaves fisicos == lista slaves no JSON - - Comparar vendor_id, product_code de cada slave por position - - Se divergir -> logar erro e retornar False - 4. Aplicar SDOs conforme sdo_configurations de cada slave no JSON - 5. Configurar PDO mapping conforme JSON - 6. Transicionar slaves para OP -``` - -**Criterio de Aceite:** -- Master inicializa com ebpfcat -- Slaves validados contra JSON -- Scan detecta slaves -- Ciclo executa sem erros - -#### Etapa 2.2: Mapeamento de PDOs -**Duracao estimada:** 1 semana - -**Tarefas:** -1. Calcular offsets de bytes/bits a partir dos PDOs no JSON -2. Considerar padding entries no calculo -3. Mapear PDOs do ebpfcat para buffers OpenPLC -4. Suportar tipos de dados (bool, int, dint, lint) -5. Usar journal writes para saidas -6. Implementar leitura de entradas - -**Arquivos:** -- `pdo_mapper.py` - -**Criterio de Aceite:** -- PDOs lidos dos slaves -- Valores escritos nos buffers corretos -- Journal writes funcionando - -#### Etapa 2.3: Aplicacao de SDOs -**Duracao estimada:** 3-4 dias - -**Tarefas:** -1. Implementar leitura de SDO -2. Implementar escrita de SDO -3. Iterar sobre sdo_configurations de cada slave no JSON -4. Aplicar configuracoes no startup (Pre-Op) - -**Arquivos:** -- `sdo_handler.py` - -**Criterio de Aceite:** -- SDOs lidos corretamente -- Configuracoes aplicadas no startup conforme JSON - -#### Etapa 2.4: Maquina de Estados -**Duracao estimada:** 3-4 dias - -**Tarefas:** -1. Implementar transicoes Init -> Pre-Op -> Safe-Op -> Op -2. Tratar erros de transicao -3. Monitorar estado atual - -**Arquivos:** -- `state_machine.py` - -**Criterio de Aceite:** -- Slaves transicionam para OP -- Erros detectados e reportados - ---- - -### Fase 3: Diagnostico e API (1-2 semanas) - -#### Etapa 3.1: Sistema de Diagnostico -**Duracao estimada:** 3-4 dias - -**Tarefas:** -1. Coletar status do master -2. Coletar status de cada slave -3. Registrar erros com timestamp -4. Expor via estrutura interna - -**Arquivos:** -- `diagnostics.py` - -**Criterio de Aceite:** -- Diagnosticos coletados -- Formato compativel com SOEM - -#### Etapa 3.2: Integracao com API REST -**Duracao estimada:** 2-3 dias - -**Tarefas:** -1. Expor diagnosticos nos mesmos endpoints do SOEM -2. Formato de resposta identico - -**Criterio de Aceite:** -- Mesmos endpoints funcionais -- Respostas compativeis - ---- - -### Fase 4: Testes e Documentacao (1-2 semanas) - -#### Etapa 4.1: Testes -**Duracao estimada:** 1 semana - -**Tarefas:** -1. Testes unitarios -2. Testes de integracao -3. Testes com hardware real (se disponivel) -4. Comparacao de performance com SOEM - -**Criterio de Aceite:** -- Testes passando -- Documentacao de limitacoes - -#### Etapa 4.2: Documentacao -**Duracao estimada:** 2-3 dias - -**Tarefas:** -1. Documentar requisitos de sistema (kernel, XDP) -2. Documentar diferenças em relação ao SOEM -3. Guia de troubleshooting - -**Criterio de Aceite:** -- Documentacao completa - ---- - -## 4. Requisitos de Sistema (ebpfcat) - -### 4.1 Kernel Linux - -| Requisito | Especificacao | -|-----------|---------------| -| Versao minima | 5.4+ (recomendado 5.10+) | -| Configuracoes | CONFIG_BPF=y, CONFIG_BPF_SYSCALL=y | -| Permissoes | CAP_BPF, CAP_NET_ADMIN ou root | - -### 4.2 Driver de Rede - -O driver deve suportar XDP. Drivers compativeis incluem: - -| Driver | Suporte XDP | -|--------|-------------| -| mlx5 (Mellanox) | Completo | -| i40e (Intel) | Completo | -| ixgbe (Intel) | Completo | -| igb (Intel) | Parcial | -| e1000e (Intel) | Limitado | -| virtio-net | Sim (VMs) | -| r8169 (Realtek) | Nao | - -### 4.3 Verificacao de Compatibilidade - -Script para verificar sistema: - -```bash -#!/bin/bash -# check_ebpf_support.sh - -echo "=== Verificacao de Suporte eBPF/XDP ===" - -# Kernel version -KERNEL=$(uname -r) -echo "Kernel: $KERNEL" - -# eBPF support -if [ -d /sys/fs/bpf ]; then - echo "eBPF filesystem: OK" -else - echo "eBPF filesystem: NOT MOUNTED" -fi - -# XDP support na interface -IFACE=${1:-eth0} -if ethtool -i $IFACE 2>/dev/null | grep -q driver; then - DRIVER=$(ethtool -i $IFACE | grep driver | awk '{print $2}') - echo "Interface $IFACE driver: $DRIVER" - - # Testar XDP - if ip link set dev $IFACE xdp off 2>/dev/null; then - echo "XDP support on $IFACE: OK" - else - echo "XDP support on $IFACE: UNKNOWN (needs test)" - fi -else - echo "Interface $IFACE: NOT FOUND" -fi - -# Capabilities -if capsh --print | grep -q cap_bpf; then - echo "CAP_BPF: Available" -else - echo "CAP_BPF: Requires root or capabilities" -fi -``` - ---- - -## 5. Comparacao de Implementacao - -| Aspecto | Plugin SOEM | Plugin ebpfcat | -|---------|-------------|----------------| -| Linguagem | C/C++ | Python | -| Tipo plugin | Nativo (type=1) | Python (type=0) | -| Ciclo hooks | cycle_start/cycle_end | Thread separada | -| Buffer access | Direto com mutex | SafeBufferAccess | -| Configuracao | ethercat_config.json | ethercat_config.json | -| JSON format | Identico | Identico | -| Kernel | 4.x+ | 5.x+ | -| NIC | Qualquer | XDP compativel | -| Plataformas | x86, ARM64, ARMv7 | x86 (ARM experimental) | - ---- - -## 6. Riscos Especificos do ebpfcat - -| Risco | Probabilidade | Impacto | Mitigacao | -|-------|---------------|---------|-----------| -| Driver sem XDP | Alta | Bloqueante | Documentar NICs compativeis | -| API ebpfcat instavel | Media | Alto | Fixar versao, testar atualizacoes | -| Performance em Python | Media | Medio | Ciclo critico em eBPF, nao Python | -| Kernel antigo | Media | Bloqueante | Documentar requisitos | -| Suporte ARM | Alta | Bloqueante | Marcar como experimental | - ---- - -## 7. Referencias - -- ebpfcat: https://ebpfcat.readthedocs.io/ -- ebpfcat GitHub: https://github.com/tecki/ebpfcat -- eBPF Documentation: https://ebpf.io/ -- XDP Tutorial: https://github.com/xdp-project/xdp-tutorial -- Plugin Python existente (modbus_master): `core/src/drivers/plugins/python/modbus_master/` -- JSON config spec: `docs/old_docs/ethercat-plugin-development-plan.md` secao 3 -- Discovery Service (ja implementado): `webserver/discovery/` diff --git a/docs/old_docs/C_PYTHON_DATA_SHARING_PROPOSAL.md b/docs/old_docs/C_PYTHON_DATA_SHARING_PROPOSAL.md deleted file mode 100644 index 975311d8..00000000 --- a/docs/old_docs/C_PYTHON_DATA_SHARING_PROPOSAL.md +++ /dev/null @@ -1,587 +0,0 @@ -# Proposal: Robust C-Python Runtime Data Sharing - -## Executive Summary - -This document analyzes the current approach for sharing runtime functions and buffers between C and Python plugins in OpenPLC, identifies its weaknesses, and proposes a more robust and simplified architecture. - ---- - -## 1. Current Implementation Analysis - -### 1.1 How It Works Today - -The current system uses a monolithic C struct (`plugin_runtime_args_t`) that is: -1. Allocated and populated in C (`plugin_driver.c`) -2. Wrapped in a PyCapsule -3. Passed to Python plugins -4. Extracted using ctypes with a manually-maintained mirror struct - -``` -┌─────────────┐ PyCapsule ┌─────────────┐ ctypes ┌─────────────┐ -│ C Struct │ ──────────────> │ Capsule │ ──────────> │ Python Struct│ -│ (456 bytes) │ │ (pointer) │ │ (mirror) │ -└─────────────┘ └─────────────┘ └─────────────┘ -``` - -### 1.2 Current Problems - -| Problem | Impact | Severity | -|---------|--------|----------| -| **Manual struct synchronization** | Any field order change in C requires manual Python update | Critical | -| **No version checking** | Incompatible changes cause silent memory corruption | Critical | -| **Complex nested pointers** | `IEC_BOOL *(*bool_input)[8]` is error-prone in ctypes | High | -| **Monolithic struct** | Adding one field requires updating entire struct on both sides | High | -| **No compile-time validation** | Mismatches only discovered at runtime (crashes) | High | -| **Tight coupling** | Python code depends on exact C memory layout | Medium | - -### 1.3 Root Cause of Recent Bug - -The crash was caused by field order mismatch: - -```c -// C struct order (plugin_types.h): -mutex_take -mutex_give -buffer_mutex // <-- Position 3 -get_var_list // <-- Position 4 -get_var_size -get_var_count -``` - -```python -# Python struct order (plugin_runtime_args.py) - WRONG: -mutex_take -mutex_give -get_var_list # <-- Position 3 (WRONG!) -get_var_size -get_var_count -buffer_mutex # <-- Position 6 (WRONG!) -``` - -This caused Python to read garbage values, leading to segfaults. - ---- - -## 2. Proposed Solution: Layered API Architecture - -### 2.1 Design Principles - -1. **Separation of Concerns**: Split the monolithic struct into logical groups -2. **Explicit Versioning**: Include version info for compatibility checking -3. **Simplified Interface**: Hide pointer complexity behind C helper functions -4. **Validation First**: Validate compatibility before any data access -5. **Single Source of Truth**: Generate Python bindings from C definitions - -### 2.2 Architecture Overview - -``` -┌─────────────────────────────────────────────────────────────────────────┐ -│ LAYER 3: Plugin API │ -│ High-level Python interface (SafeBufferAccess, OpcuaServer, etc.) │ -└─────────────────────────────────────────────────────────────────────────┘ - │ - ▼ -┌─────────────────────────────────────────────────────────────────────────┐ -│ LAYER 2: C Bridge Functions │ -│ Simple C functions called via ctypes (no complex pointer passing) │ -│ - plc_read_variable(index) -> value │ -│ - plc_write_variable(index, value) -> success │ -│ - plc_get_var_info(index) -> {size, type, name} │ -└─────────────────────────────────────────────────────────────────────────┘ - │ - ▼ -┌─────────────────────────────────────────────────────────────────────────┐ -│ LAYER 1: Minimal Bootstrap Struct │ -│ Only contains: version, function pointers to bridge, config path │ -│ Small, stable, rarely changes │ -└─────────────────────────────────────────────────────────────────────────┘ -``` - ---- - -## 3. Detailed Implementation - -### 3.1 Layer 1: Minimal Bootstrap Struct - -Replace the large monolithic struct with a minimal bootstrap struct: - -```c -// plugin_api.h - NEW FILE - -#define PLUGIN_API_VERSION_MAJOR 2 -#define PLUGIN_API_VERSION_MINOR 0 - -// Simple struct with only essentials - STABLE, rarely changes -typedef struct { - // Version info for compatibility checking - uint32_t api_version_major; - uint32_t api_version_minor; - uint32_t struct_size; // For validation - - // Function pointers to bridge layer (Layer 2) - void* bridge_handle; // Opaque handle to bridge context - - // Essential function pointers - int (*read_variable)(void* handle, uint16_t index, void* value, size_t* size); - int (*write_variable)(void* handle, uint16_t index, const void* value, size_t size); - int (*get_variable_count)(void* handle, uint16_t* count); - int (*get_variable_info)(void* handle, uint16_t index, VariableInfo* info); - - // Logging (simple interface) - void (*log_message)(void* handle, int level, const char* message); - - // Config - char config_path[256]; - -} PluginBootstrap; - -typedef struct { - uint16_t index; - uint8_t type; // IEC type enum - uint8_t direction; // INPUT, OUTPUT, MEMORY - size_t size; // Size in bytes - char name[64]; // Variable name (optional) -} VariableInfo; -``` - -**Benefits:** -- Only 11 fields vs 25+ in current struct -- No complex nested pointers -- Version checking built-in -- `struct_size` allows runtime validation - -### 3.2 Layer 2: C Bridge Functions - -Implement simple C functions that hide the complexity: - -```c -// plugin_bridge.c - NEW FILE - -typedef struct { - plugin_driver_t* driver; - pthread_mutex_t* mutex; - // Internal state -} BridgeContext; - -// Read any variable by index - handles all types internally -int bridge_read_variable(void* handle, uint16_t index, void* value, size_t* size) { - BridgeContext* ctx = (BridgeContext*)handle; - - // Lock mutex - pthread_mutex_lock(ctx->mutex); - - // Get variable info - size_t var_size = ext_get_var_size(index); - void* var_addr = ext_get_var_addr(index); - - if (!var_addr || var_size == 0) { - pthread_mutex_unlock(ctx->mutex); - return PLUGIN_ERR_INVALID_INDEX; - } - - // Copy value - memcpy(value, var_addr, var_size); - *size = var_size; - - pthread_mutex_unlock(ctx->mutex); - return PLUGIN_OK; -} - -// Write any variable by index -int bridge_write_variable(void* handle, uint16_t index, const void* value, size_t size) { - BridgeContext* ctx = (BridgeContext*)handle; - - pthread_mutex_lock(ctx->mutex); - - size_t var_size = ext_get_var_size(index); - void* var_addr = ext_get_var_addr(index); - - if (!var_addr || var_size == 0 || size != var_size) { - pthread_mutex_unlock(ctx->mutex); - return PLUGIN_ERR_INVALID_INDEX; - } - - memcpy(var_addr, value, size); - - pthread_mutex_unlock(ctx->mutex); - return PLUGIN_OK; -} - -// Get variable metadata -int bridge_get_variable_info(void* handle, uint16_t index, VariableInfo* info) { - info->index = index; - info->size = ext_get_var_size(index); - info->type = determine_iec_type(index); // Internal helper - info->direction = determine_direction(index); - // name populated if available - return PLUGIN_OK; -} -``` - -**Benefits:** -- Mutex handling is internal - Python doesn't manage locks -- Type handling is internal - Python just passes bytes -- Error codes instead of crashes -- No pointer arithmetic in Python - -### 3.3 Layer 3: Python Simple Interface - -```python -# plugin_api.py - NEW FILE - -import ctypes -from enum import IntEnum - -class PluginError(IntEnum): - OK = 0 - INVALID_INDEX = 1 - INVALID_SIZE = 2 - MUTEX_ERROR = 3 - VERSION_MISMATCH = 4 - -class PluginBootstrap(ctypes.Structure): - """Minimal bootstrap struct - matches C exactly""" - _fields_ = [ - ("api_version_major", ctypes.c_uint32), - ("api_version_minor", ctypes.c_uint32), - ("struct_size", ctypes.c_uint32), - ("bridge_handle", ctypes.c_void_p), - ("read_variable", ctypes.CFUNCTYPE( - ctypes.c_int, # return - ctypes.c_void_p, # handle - ctypes.c_uint16, # index - ctypes.c_void_p, # value (output) - ctypes.POINTER(ctypes.c_size_t) # size (output) - )), - ("write_variable", ctypes.CFUNCTYPE( - ctypes.c_int, - ctypes.c_void_p, - ctypes.c_uint16, - ctypes.c_void_p, - ctypes.c_size_t - )), - ("get_variable_count", ctypes.CFUNCTYPE( - ctypes.c_int, - ctypes.c_void_p, - ctypes.POINTER(ctypes.c_uint16) - )), - ("get_variable_info", ctypes.CFUNCTYPE( - ctypes.c_int, - ctypes.c_void_p, - ctypes.c_uint16, - ctypes.c_void_p # VariableInfo* - )), - ("log_message", ctypes.CFUNCTYPE( - None, - ctypes.c_void_p, - ctypes.c_int, - ctypes.c_char_p - )), - ("config_path", ctypes.c_char * 256), - ] - - -class PLCBridge: - """High-level Python interface to PLC runtime""" - - EXPECTED_VERSION_MAJOR = 2 - EXPECTED_VERSION_MINOR = 0 - - def __init__(self, capsule): - # Extract bootstrap struct - self._bootstrap = self._extract_bootstrap(capsule) - - # Validate version FIRST - self._validate_version() - - # Cache handle for function calls - self._handle = self._bootstrap.bridge_handle - - def _validate_version(self): - """Check API compatibility before any operations""" - major = self._bootstrap.api_version_major - minor = self._bootstrap.api_version_minor - size = self._bootstrap.struct_size - - if major != self.EXPECTED_VERSION_MAJOR: - raise RuntimeError( - f"API version mismatch: expected {self.EXPECTED_VERSION_MAJOR}.x, " - f"got {major}.{minor}" - ) - - expected_size = ctypes.sizeof(PluginBootstrap) - if size != expected_size: - raise RuntimeError( - f"Struct size mismatch: expected {expected_size}, got {size}. " - f"This indicates a build mismatch between C and Python." - ) - - def read_variable(self, index: int) -> tuple[bytes, int]: - """Read a PLC variable by index. Returns (value_bytes, error_code)""" - value_buffer = ctypes.create_string_buffer(8) # Max 64-bit - size = ctypes.c_size_t(0) - - result = self._bootstrap.read_variable( - self._handle, - ctypes.c_uint16(index), - ctypes.cast(value_buffer, ctypes.c_void_p), - ctypes.byref(size) - ) - - if result != PluginError.OK: - return None, result - - return value_buffer.raw[:size.value], PluginError.OK - - def write_variable(self, index: int, value: bytes) -> int: - """Write a PLC variable by index. Returns error_code""" - return self._bootstrap.write_variable( - self._handle, - ctypes.c_uint16(index), - value, - len(value) - ) - - def get_variable_count(self) -> tuple[int, int]: - """Get total number of variables. Returns (count, error_code)""" - count = ctypes.c_uint16(0) - result = self._bootstrap.get_variable_count( - self._handle, - ctypes.byref(count) - ) - return count.value, result - - def log(self, level: int, message: str): - """Log a message through the C runtime""" - self._bootstrap.log_message( - self._handle, - level, - message.encode('utf-8') - ) -``` - ---- - -## 4. Migration Path - -### Phase 1: Add Version Checking (Low Risk) - -Add version fields to existing struct without breaking compatibility: - -```c -// Add to beginning of existing plugin_runtime_args_t -typedef struct { - // NEW: Version info (add at START for easy access) - uint32_t api_version; // = 0x00010000 for v1.0 - uint32_t struct_size; // = sizeof(plugin_runtime_args_t) - - // ... existing fields unchanged ... -} plugin_runtime_args_t; -``` - -```python -# Update Python to check version first -def validate_struct(args): - if args.api_version != 0x00010000: - raise RuntimeError(f"Version mismatch: {args.api_version:#x}") - if args.struct_size != ctypes.sizeof(PluginRuntimeArgs): - raise RuntimeError(f"Size mismatch: {args.struct_size}") -``` - -### Phase 2: Add Bridge Functions (Medium Risk) - -Add new bridge functions alongside existing implementation: - -```c -// New bridge functions coexist with direct buffer access -// Plugins can choose which to use -``` - -### Phase 3: Deprecate Direct Buffer Access (Breaking Change) - -Once all plugins migrate to bridge functions, remove direct buffer pointers from the API. - ---- - -## 5. Alternative: Auto-Generated Bindings - -### 5.1 Generate Python from C Header - -Use a tool to automatically generate Python ctypes from C header: - -```bash -# Using ctypesgen (example) -ctypesgen -o plugin_types_generated.py plugin_types.h -``` - -### 5.2 Compile-Time Struct Validation - -Add a C program that validates struct layout at build time: - -```c -// validate_struct_layout.c - Run during build -#include "plugin_types.h" -#include -#include - -int main() { - printf("STRUCT_SIZE=%zu\n", sizeof(plugin_runtime_args_t)); - printf("OFFSET_mutex_take=%zu\n", offsetof(plugin_runtime_args_t, mutex_take)); - printf("OFFSET_mutex_give=%zu\n", offsetof(plugin_runtime_args_t, mutex_give)); - printf("OFFSET_buffer_mutex=%zu\n", offsetof(plugin_runtime_args_t, buffer_mutex)); - printf("OFFSET_get_var_list=%zu\n", offsetof(plugin_runtime_args_t, get_var_list)); - // ... etc - return 0; -} -``` - -Python can read this at runtime to validate: - -```python -def validate_offsets(): - """Compare expected vs actual field offsets""" - expected = read_offsets_from_build_output() - for field, offset in PluginRuntimeArgs._fields_: - actual = getattr(PluginRuntimeArgs, field).offset - if actual != expected[field]: - raise RuntimeError(f"Offset mismatch for {field}") -``` - ---- - -## 6. Comparison - -| Aspect | Current | Proposed (Bridge) | Proposed (Auto-gen) | -|--------|---------|-------------------|---------------------| -| **Complexity** | High | Low | Medium | -| **Maintenance** | Manual sync required | Minimal | Automated | -| **Performance** | Direct memory | Function call overhead | Direct memory | -| **Safety** | Crash on mismatch | Error codes | Validated at build | -| **Breaking Changes** | Silent corruption | Version check fails | Build fails | -| **Implementation Effort** | N/A | Medium | Low | - ---- - -## 7. Recommendation - -### Short Term (Immediate) -1. Add `api_version` and `struct_size` fields to existing struct -2. Add validation in Python before accessing any fields -3. Add build-time offset validation script - -### Medium Term (Next Release) -1. Implement bridge functions for variable access -2. Migrate plugins to use bridge functions -3. Add comprehensive error handling - -### Long Term (Future) -1. Deprecate direct buffer pointer access -2. Simplify bootstrap struct to minimal interface -3. Consider using a proper FFI library (cffi) instead of ctypes - ---- - -## 8. Code Examples - -### 8.1 Quick Fix: Add Version Validation - -```c -// plugin_types.h - Add at the BEGINNING of struct -#define PLUGIN_API_VERSION 0x00020001 // v2.0.1 - -typedef struct { - uint32_t api_version; // MUST be first field - uint32_t struct_size; // MUST be second field - - // ... rest of existing fields ... -} plugin_runtime_args_t; - -// plugin_driver.c - Set version when creating -args->api_version = PLUGIN_API_VERSION; -args->struct_size = sizeof(plugin_runtime_args_t); -``` - -```python -# plugin_runtime_args.py - Add validation -EXPECTED_API_VERSION = 0x00020001 - -class PluginRuntimeArgs(ctypes.Structure): - _fields_ = [ - ("api_version", ctypes.c_uint32), # NEW - must be first - ("struct_size", ctypes.c_uint32), # NEW - must be second - # ... rest unchanged ... - ] - - def validate(self): - if self.api_version != EXPECTED_API_VERSION: - raise RuntimeError( - f"API version mismatch: C={self.api_version:#x}, " - f"Python={EXPECTED_API_VERSION:#x}" - ) - expected_size = ctypes.sizeof(PluginRuntimeArgs) - if self.struct_size != expected_size: - raise RuntimeError( - f"Struct size mismatch: C={self.struct_size}, " - f"Python={expected_size}" - ) -``` - -### 8.2 Build-Time Validation Script - -```python -#!/usr/bin/env python3 -# scripts/validate_struct_layout.py - -import subprocess -import ctypes -import sys - -# Compile and run C validation program -result = subprocess.run( - ['./build/validate_struct_layout'], - capture_output=True, text=True -) - -# Parse C offsets -c_offsets = {} -for line in result.stdout.strip().split('\n'): - key, value = line.split('=') - c_offsets[key] = int(value) - -# Compare with Python -from plugin_runtime_args import PluginRuntimeArgs - -py_size = ctypes.sizeof(PluginRuntimeArgs) -if py_size != c_offsets['STRUCT_SIZE']: - print(f"ERROR: Size mismatch C={c_offsets['STRUCT_SIZE']} Python={py_size}") - sys.exit(1) - -# Check each field offset -for name, ctype in PluginRuntimeArgs._fields_: - field = getattr(PluginRuntimeArgs, name) - py_offset = field.offset - c_key = f'OFFSET_{name}' - if c_key in c_offsets: - if py_offset != c_offsets[c_key]: - print(f"ERROR: {name} offset mismatch C={c_offsets[c_key]} Python={py_offset}") - sys.exit(1) - -print("All struct validations passed!") -sys.exit(0) -``` - ---- - -## 9. Conclusion - -The current implementation's fragility stems from: -1. Manual synchronization of complex struct layouts -2. No version checking -3. Direct memory access without validation - -The proposed solution addresses these by: -1. Adding explicit version and size validation -2. Providing a simpler bridge API that hides complexity -3. Optionally auto-generating bindings from C headers - -**Recommended immediate action**: Add `api_version` and `struct_size` fields to catch mismatches early, preventing silent memory corruption and crashes. diff --git a/docs/old_docs/ethercat-plugin-development-plan.md b/docs/old_docs/ethercat-plugin-development-plan.md deleted file mode 100644 index 4ecca958..00000000 --- a/docs/old_docs/ethercat-plugin-development-plan.md +++ /dev/null @@ -1,1271 +0,0 @@ -# Plano de Desenvolvimento - Plugin EtherCAT (Runtime) - -**Produto:** OpenPLC Runtime v4 -**Baseado em:** Levantamento de Requisitos - Protocolo EtherCAT (v1.3) -**Data:** 29 de Janeiro de 2026 -**Tipo:** Plugin Nativo C/C++ -**Revisao:** 3.0 - Arquitetura Editor-driven (discovery e parametrizacao no Editor) - ---- - -## 1. Visao Geral - -Este plano detalha a implementacao do plugin EtherCAT Master para o OpenPLC Runtime v4, -utilizando a biblioteca SOEM (Simple Open EtherCAT Master) conforme definido nos requisitos. - -### 1.0 Separacao de Responsabilidades: Editor vs Runtime - -A arquitetura do sistema EtherCAT segue o mesmo padrao dos demais plugins do OpenPLC: -o **Editor** e responsavel por toda a configuracao e parametrizacao, e o **Runtime** e -responsavel apenas pela execucao. - -``` -┌─────────────────────────────────────────────────────────────────────────┐ -│ OpenPLC EDITOR │ -│ │ -│ - Discovery: scan de rede via Runtime Discovery Service (ja feito) │ -│ - Upload e parsing de arquivos ESI (XML) │ -│ - Parametrizacao de couplers e modulos │ -│ - Configuracao de channels e located vars (%IX, %QX, etc.) │ -│ - Mapeamento de PDOs para variaveis IEC │ -│ - Configuracao de SDOs (parametros dos slaves) │ -│ - Geracao do JSON de configuracao final │ -│ │ -│ Resultado: ethercat_config.json (contrato Editor -> Runtime) │ -└─────────────────────────────────┬───────────────────────────────────────┘ - │ - │ Upload via programa (program.zip) - │ JSON em core/generated/conf/ - ▼ -┌─────────────────────────────────────────────────────────────────────────┐ -│ OpenPLC RUNTIME │ -│ │ -│ - Recebe JSON de configuracao completo (via plugins.conf) │ -│ - Inicializa SOEM master na interface configurada │ -│ - Valida topologia: slaves fisicos vs configuracao JSON │ -│ - Aplica configuracoes SDO nos slaves (Pre-Op) │ -│ - Configura mapeamento de PDOs conforme JSON │ -│ - Transiciona slaves para estado Operational │ -│ - Executa ciclo de comunicacao (cycle_start / cycle_end) │ -│ - Monitora diagnosticos e erros │ -│ │ -│ NAO faz: discovery, parsing ESI, parametrizacao de UI │ -└─────────────────────────────────────────────────────────────────────────┘ -``` - -### 1.0.1 Fluxo de Configuracao (igual aos demais plugins) - -O fluxo segue exatamente o padrao existente no OpenPLC para todos os plugins: - -``` -1. Editor parametriza dispositivos EtherCAT (couplers, modulos, channels) -2. Editor gera ethercat_config.json com toda a configuracao -3. Editor envia program.zip contendo o JSON em conf/ethercat_config.json -4. Runtime extrai para core/generated/conf/ -5. update_plugin_configurations() copia JSON para diretorio do plugin -6. plugins.conf e atualizado com caminho do config -7. Plugin init() recebe caminho via plugin_runtime_args_t.plugin_specific_config_file_path -8. Plugin carrega JSON, inicializa SOEM, e opera -``` - -### 1.0.2 Discovery Service (JA IMPLEMENTADO) - -O Discovery Service no Runtime ja esta implementado e funcional. Ele fornece endpoints -REST para que o Editor faca scan de rede e deteccao de dispositivos: - -**Status: CONCLUIDO** - -Arquivos implementados: -- `webserver/discovery/discovery_routes.py` - Endpoints REST -- `webserver/discovery/ethercat_discovery.py` - Logica de discovery -- `scripts/discovery/ethercat_scan.py` - Scanner usando pysoem -- `scripts/setup_discovery_venv.sh` - Setup do venv de discovery -- `tests/pytest/discovery/` - Testes unitarios (40+ testes) - -Endpoints disponiveis: -- `GET /api/discovery/interfaces` - Lista interfaces de rede -- `GET /api/discovery/ethercat/status` - Status do servico -- `POST /api/discovery/ethercat/scan` - Scan da rede -- `POST /api/discovery/ethercat/validate` - Valida configuracao -- `POST /api/discovery/ethercat/test` - Testa conexao com slave - -### 1.1 Escopo do MVP - -Conforme o documento de requisitos, o MVP do **plugin Runtime** deve incluir: - -- Carregamento do JSON de configuracao gerado pelo Editor -- Validacao de topologia (slaves fisicos vs configuracao) -- Mapeamento de PDOs para variaveis do PLC conforme JSON -- Aplicacao de configuracoes SDO nos slaves -- Operacao estavel com cycle time de 4 ms -- Suporte a CoE (SDO e PDO) -- Diagnostico basico de status e erros -- Suporte ao perfil DS401 (I/O Devices) - -### 1.2 Itens Fora do Escopo (MVP) - -- Distributed Clocks (RF04) -- Multiplos Masters (RF05) -- Hot-connect de dispositivos (RF08) -- Redundancia de cabo (RF10) -- EoE - Ethernet over EtherCAT (RF12) -- FoE - File over EtherCAT (RF13) -- Perfil DS402 - Motion Control (RF14) -- Controle de servo drives - -### 1.3 Itens que NAO pertencem ao Runtime - -Os seguintes itens sao responsabilidade exclusiva do **Editor**: - -- Scan de rede / discovery de dispositivos (Editor usa Discovery Service do Runtime) -- Upload e parsing de arquivos ESI (XML) -- Interface de parametrizacao de couplers e modulos -- Interface de configuracao de channels e located vars -- Mapeamento visual de PDOs -- Geracao do JSON de configuracao - ---- - -## 2. Arquitetura do Plugin - -### 2.1 Estrutura de Arquivos - -``` -core/src/drivers/plugins/native/ethercat/ -├── CMakeLists.txt # Build configuration -├── ethercat_plugin.c # Main plugin entry (init, start, stop, cycle hooks) -├── ethercat_plugin.h # Plugin interface definitions -├── ethercat_config.h # Configuration structures -├── ethercat_config.c # JSON config parser (cJSON) -├── ethercat_config.json # Default/empty configuration file -├── ethercat_master.c # SOEM wrapper - master operations -├── ethercat_master.h -├── ethercat_pdo_mapper.c # PDO to IEC variable mapping (from JSON config) -├── ethercat_pdo_mapper.h -├── ethercat_diagnostics.c # Status and error reporting -├── ethercat_diagnostics.h -├── ethercat_state_machine.c # EtherCAT state transitions -├── ethercat_state_machine.h -└── libs/ - └── soem/ # SOEM library (submodule or vendored) -``` - -**Nota:** Nao ha `ethercat_esi_parser.c` nem `ethercat_slave_manager.c`. O parsing -de ESI e feito no Editor, e o gerenciamento de slaves e baseado no JSON de configuracao. - -### 2.2 Integracao com Plugin System - -O plugin seguira o padrao nativo existente (similar ao S7Comm): - -```c -// Funcoes obrigatorias -int init(void *args); // Inicializacao com runtime_args -void start_loop(void); // Inicia thread EtherCAT -void stop_loop(void); // Para thread EtherCAT -void cleanup(void); // Liberacao de recursos - -// Funcoes de ciclo (chamadas com mutex held) -void cycle_start(void); // Leitura de inputs dos slaves -void cycle_end(void); // Escrita de outputs para slaves -``` - -### 2.3 Configuracao (plugins.conf) - -``` -ethercat,./build/plugins/libethercat_plugin.so,1,1,./core/src/drivers/plugins/native/ethercat/ethercat_config.json, -``` - ---- - -## 3. Contrato JSON: Editor -> Runtime - -O JSON de configuracao e o **contrato** entre Editor e Runtime. O Editor gera este JSON -com **todos** os parametros necessarios para o funcionamento do plugin. O Runtime nao -faz discovery nem parsing ESI - tudo ja vem resolvido no JSON. - -### 3.1 Estrutura Completa do JSON - -A lista `slaves` e **flat** (plana), exatamente como o array `ec_slave[]` da SOEM. -Cada slave ocupa uma posicao no barramento, independente de ser coupler ou modulo. -O campo `position` corresponde diretamente ao indice `ec_slave[position]` da SOEM. - -```json -[ - { - "name": "ethercat_master", - "protocol": "ETHERCAT", - "config": { - "master": { - "interface": "eth0", - "cycle_time_us": 1000, - "watchdog_timeout_cycles": 3, - "log_level": "info" - }, - "slaves": [ - { - "position": 1, - "name": "EK1100", - "type": "coupler", - "vendor_id": "0x00000002", - "product_code": "0x044c2c52", - "revision": "0x00120000", - "channels": [], - "sdo_configurations": [], - "rx_pdos": [], - "tx_pdos": [] - }, - { - "position": 2, - "name": "EL1008", - "type": "digital_input", - "vendor_id": "0x00000002", - "product_code": "0x03f03052", - "revision": "0x00120000", - "channels": [ - { - "index": 0, - "name": "Input 1", - "type": "digital_input", - "bit_length": 1, - "iec_location": "%IX0.0", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 1 - }, - { - "index": 1, - "name": "Input 2", - "type": "digital_input", - "bit_length": 1, - "iec_location": "%IX0.1", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 2 - }, - { - "index": 2, - "name": "Input 3", - "type": "digital_input", - "bit_length": 1, - "iec_location": "%IX0.2", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 3 - }, - { - "index": 3, - "name": "Input 4", - "type": "digital_input", - "bit_length": 1, - "iec_location": "%IX0.3", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 4 - }, - { - "index": 4, - "name": "Input 5", - "type": "digital_input", - "bit_length": 1, - "iec_location": "%IX0.4", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 5 - }, - { - "index": 5, - "name": "Input 6", - "type": "digital_input", - "bit_length": 1, - "iec_location": "%IX0.5", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 6 - }, - { - "index": 6, - "name": "Input 7", - "type": "digital_input", - "bit_length": 1, - "iec_location": "%IX0.6", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 7 - }, - { - "index": 7, - "name": "Input 8", - "type": "digital_input", - "bit_length": 1, - "iec_location": "%IX1.0", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 8 - } - ], - "sdo_configurations": [], - "rx_pdos": [], - "tx_pdos": [ - { - "index": "0x1A00", - "name": "TxPDO-Map Inputs", - "entries": [ - { - "index": "0x6000", - "subindex": 1, - "bit_length": 1, - "name": "Input 1", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 2, - "bit_length": 1, - "name": "Input 2", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 3, - "bit_length": 1, - "name": "Input 3", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 4, - "bit_length": 1, - "name": "Input 4", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 5, - "bit_length": 1, - "name": "Input 5", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 6, - "bit_length": 1, - "name": "Input 6", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 7, - "bit_length": 1, - "name": "Input 7", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 8, - "bit_length": 1, - "name": "Input 8", - "data_type": "BOOL" - } - ] - } - ] - }, - { - "position": 3, - "name": "EL2008", - "type": "digital_output", - "vendor_id": "0x00000002", - "product_code": "0x07d83052", - "revision": "0x00120000", - "channels": [ - { - "index": 0, - "name": "Output 1", - "type": "digital_output", - "bit_length": 1, - "iec_location": "%QX0.0", - "pdo_index": "0x1600", - "pdo_entry_index": "0x7000", - "pdo_entry_subindex": 1 - }, - { - "index": 1, - "name": "Output 2", - "type": "digital_output", - "bit_length": 1, - "iec_location": "%QX0.1", - "pdo_index": "0x1600", - "pdo_entry_index": "0x7000", - "pdo_entry_subindex": 2 - } - ], - "sdo_configurations": [], - "rx_pdos": [ - { - "index": "0x1600", - "name": "RxPDO-Map Outputs", - "entries": [ - { - "index": "0x7000", - "subindex": 1, - "bit_length": 1, - "name": "Output 1", - "data_type": "BOOL" - }, - { - "index": "0x7000", - "subindex": 2, - "bit_length": 1, - "name": "Output 2", - "data_type": "BOOL" - } - ] - } - ], - "tx_pdos": [] - }, - { - "position": 4, - "name": "EL3062", - "type": "analog_input", - "vendor_id": "0x00000002", - "product_code": "0x0bf63052", - "revision": "0x00120000", - "channels": [ - { - "index": 0, - "name": "Analog Input 1", - "type": "analog_input", - "bit_length": 16, - "iec_location": "%IW0", - "pdo_index": "0x1A00", - "pdo_entry_index": "0x6000", - "pdo_entry_subindex": 17 - }, - { - "index": 1, - "name": "Analog Input 2", - "type": "analog_input", - "bit_length": 16, - "iec_location": "%IW1", - "pdo_index": "0x1A01", - "pdo_entry_index": "0x6010", - "pdo_entry_subindex": 17 - } - ], - "sdo_configurations": [ - { - "index": "0x8000", - "subindex": 6, - "value": 0, - "data_type": "UINT16", - "name": "Filter setting Ch.1", - "description": "Filter constant for channel 1 (0=50Hz, 1=60Hz)" - }, - { - "index": "0x8000", - "subindex": 21, - "value": true, - "data_type": "BOOL", - "name": "Enable user scale Ch.1", - "description": "Enable user-defined scaling for channel 1" - }, - { - "index": "0x8000", - "subindex": 17, - "value": 0, - "data_type": "INT16", - "name": "User scale offset Ch.1", - "description": "User-defined offset for channel 1" - }, - { - "index": "0x8000", - "subindex": 18, - "value": 32767, - "data_type": "INT32", - "name": "User scale gain Ch.1", - "description": "User-defined gain for channel 1" - } - ], - "rx_pdos": [], - "tx_pdos": [ - { - "index": "0x1A00", - "name": "TxPDO-Map Ch.1", - "entries": [ - { - "index": "0x6000", - "subindex": 1, - "bit_length": 1, - "name": "Underrange", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 2, - "bit_length": 1, - "name": "Overrange", - "data_type": "BOOL" - }, - { - "index": "0x0000", - "subindex": 0, - "bit_length": 4, - "name": "padding", - "data_type": "PAD" - }, - { - "index": "0x6000", - "subindex": 7, - "bit_length": 1, - "name": "Error", - "data_type": "BOOL" - }, - { - "index": "0x0000", - "subindex": 0, - "bit_length": 7, - "name": "padding", - "data_type": "PAD" - }, - { - "index": "0x1800", - "subindex": 7, - "bit_length": 1, - "name": "TxPDO State", - "data_type": "BOOL" - }, - { - "index": "0x1800", - "subindex": 9, - "bit_length": 1, - "name": "TxPDO Toggle", - "data_type": "BOOL" - }, - { - "index": "0x6000", - "subindex": 17, - "bit_length": 16, - "name": "Value", - "data_type": "INT16" - } - ] - }, - { - "index": "0x1A01", - "name": "TxPDO-Map Ch.2", - "entries": [ - { - "index": "0x6010", - "subindex": 17, - "bit_length": 16, - "name": "Value", - "data_type": "INT16" - } - ] - } - ] - } - ], - "diagnostics": { - "log_connections": true, - "log_data_access": false, - "log_errors": true, - "max_log_entries": 10000, - "status_update_interval_ms": 500 - } - } - } -] -``` - -### 3.2 Descricao dos Campos - -#### 3.2.1 Nivel Raiz (padrao OpenPLC plugin config) - -| Campo | Tipo | Obrigatorio | Descricao | -|-------|------|-------------|-----------| -| `name` | string | Sim | Identificador da instancia (ex: "ethercat_master") | -| `protocol` | string | Sim | Sempre "ETHERCAT" | -| `config` | object | Sim | Configuracao completa do plugin | - -#### 3.2.2 config.master (parametros obrigatorios) - -| Campo | Tipo | Obrigatorio | Default | Descricao | -|-------|------|-------------|---------|-----------| -| `interface` | string | Sim | - | Interface de rede (ex: "eth0", "enp2s0") | -| `cycle_time_us` | int | Nao | 1000 | Tempo de ciclo em microssegundos (min: 100) | -| `watchdog_timeout_cycles` | int | Nao | 3 | Ciclos sem resposta para acionar watchdog | -| `log_level` | string | Nao | "info" | Nivel de log: "debug", "info", "warn", "error" | - -#### 3.2.3 config.slaves[] (lista flat de slaves) - -Lista plana de todos os slaves no barramento EtherCAT, ordenada por `position`. -Corresponde diretamente ao array `ec_slave[]` da SOEM. Couplers e modulos estao -no mesmo nivel - a distincao e apenas pelo campo `type`. - -**Slave:** - -| Campo | Tipo | Obrigatorio | Descricao | -|-------|------|-------------|-----------| -| `position` | int | Sim | Posicao no barramento EtherCAT (1-based), corresponde a `ec_slave[position]` | -| `name` | string | Sim | Nome do dispositivo (ex: "EK1100", "EL1008") | -| `type` | string | Sim | Tipo: "coupler", "digital_input", "digital_output", "analog_input", "analog_output" | -| `vendor_id` | string | Sim | Vendor ID hexadecimal (ex: "0x00000002") | -| `product_code` | string | Sim | Product Code hexadecimal | -| `revision` | string | Nao | Revision Number hexadecimal | -| `channels` | array | Sim | Lista de channels (pontos de I/O). Vazia para couplers | -| `sdo_configurations` | array | Nao | Configuracoes SDO para aplicar no startup | -| `rx_pdos` | array | Sim | RxPDOs (dados enviados para o slave - outputs) | -| `tx_pdos` | array | Sim | TxPDOs (dados recebidos do slave - inputs) | - -#### 3.2.4 channels[] (pontos de I/O mapeados) - -Cada channel representa um ponto de I/O individual que foi mapeado para uma -located var no programa IEC do PLC. - -| Campo | Tipo | Obrigatorio | Descricao | -|-------|------|-------------|-----------| -| `index` | int | Sim | Indice do channel no slave (0-based) | -| `name` | string | Sim | Nome descritivo (ex: "Input 1") | -| `type` | string | Sim | Tipo de I/O: "digital_input", "digital_output", "analog_input", "analog_output" | -| `bit_length` | int | Sim | Tamanho em bits: 1 (BOOL), 8 (BYTE), 16 (INT/UINT), 32 (DINT/UDINT) | -| `iec_location` | string | Sim | Located var IEC 61131-3 (ex: "%IX0.0", "%QW2", "%IW0") | -| `pdo_index` | string | Sim | Indice do PDO que contem este channel (hex) | -| `pdo_entry_index` | string | Sim | Indice do entry no PDO (hex) | -| `pdo_entry_subindex` | int | Sim | Sub-indice do entry no PDO | - -**Formato de iec_location:** -- `%IX.` - Input digital (BOOL) -- `%QX.` - Output digital (BOOL) -- `%IB` - Input byte (BYTE) -- `%QB` - Output byte (BYTE) -- `%IW` - Input word (INT/UINT, 16 bits) -- `%QW` - Output word (INT/UINT, 16 bits) -- `%ID` - Input double word (DINT/UDINT, 32 bits) -- `%QD` - Output double word (DINT/UDINT, 32 bits) -- `%IL` - Input long word (LINT/ULINT, 64 bits) -- `%QL` - Output long word (LINT/ULINT, 64 bits) - -#### 3.2.5 sdo_configurations[] (parametros de startup) - -Configuracoes SDO sao aplicadas durante a fase Pre-Operational, antes dos slaves -entrarem em modo Operational. - -| Campo | Tipo | Obrigatorio | Descricao | -|-------|------|-------------|-----------| -| `index` | string | Sim | Indice do objeto SDO (hex, ex: "0x8000") | -| `subindex` | int | Sim | Sub-indice do objeto SDO | -| `value` | varies | Sim | Valor a ser escrito (tipo depende de data_type) | -| `data_type` | string | Sim | Tipo de dado: "BOOL", "INT8", "UINT8", "INT16", "UINT16", "INT32", "UINT32" | -| `name` | string | Nao | Nome descritivo do parametro | -| `description` | string | Nao | Descricao do parametro | - -#### 3.2.6 rx_pdos[] e tx_pdos[] (mapeamento completo de PDOs) - -Descrevem o layout completo dos PDOs conforme definido no ESI. O Runtime usa esta -informacao para calcular offsets de bytes/bits no process data image. - -**Convencao de nomenclatura:** -- **RxPDO** = dados recebidos pelo slave = **outputs** do PLC -- **TxPDO** = dados transmitidos pelo slave = **inputs** do PLC - -| Campo | Tipo | Obrigatorio | Descricao | -|-------|------|-------------|-----------| -| `index` | string | Sim | Indice do PDO (hex, ex: "0x1600", "0x1A00") | -| `name` | string | Nao | Nome descritivo do PDO | -| `entries` | array | Sim | Lista de entries dentro do PDO | - -**PDO Entry:** - -| Campo | Tipo | Obrigatorio | Descricao | -|-------|------|-------------|-----------| -| `index` | string | Sim | Indice do objeto (hex). "0x0000" para padding | -| `subindex` | int | Sim | Sub-indice. 0 para padding | -| `bit_length` | int | Sim | Tamanho em bits | -| `name` | string | Nao | Nome descritivo | -| `data_type` | string | Sim | Tipo: "BOOL", "INT8", "UINT8", "INT16", "UINT16", "INT32", "UINT32", "PAD" | - -**Nota sobre padding:** Entries com `index: "0x0000"` e `data_type: "PAD"` sao -espacos de preenchimento no PDO. O Runtime deve considerar esses bits no calculo -de offset, mas nao mapeia para nenhuma variavel. - -#### 3.2.7 config.diagnostics (parametros opcionais) - -| Campo | Tipo | Obrigatorio | Default | Descricao | -|-------|------|-------------|---------|-----------| -| `log_connections` | bool | Nao | true | Logar conexoes/desconexoes de slaves | -| `log_data_access` | bool | Nao | false | Logar acessos a dados (verbose) | -| `log_errors` | bool | Nao | true | Logar erros de comunicacao | -| `max_log_entries` | int | Nao | 10000 | Maximo de entradas no buffer de log | -| `status_update_interval_ms` | int | Nao | 500 | Intervalo de atualizacao de status | - -### 3.3 Validacao no Runtime - -O Runtime deve validar o JSON recebido em dois niveis: - -**1. Validacao estrutural (no init()):** -- Campos obrigatorios presentes -- Tipos de dados corretos -- Valores dentro de faixas validas (cycle_time >= 1, positions > 0, etc.) -- Located vars com formato IEC valido -- Indices PDO/SDO em formato hexadecimal valido - -**2. Validacao de topologia (no start_loop()):** -- Numero de slaves fisicos na rede == tamanho do array `slaves` no JSON -- Para cada slave: `ec_slave[position].man` == vendor_id e `ec_slave[position].id` == product_code do JSON -- Se houver divergencia, logar erro detalhado e abortar - ---- - -## 4. Etapas de Desenvolvimento - -### Fase 1: Foundation (3-4 semanas) - -#### Etapa 1.1: Setup do Projeto e Integracao SOEM -**Duracao estimada:** 1 semana - -**Tarefas:** -1. Criar estrutura de diretorios do plugin -2. Configurar CMakeLists.txt para compilar com SOEM -3. Integrar SOEM como submodulo git ou biblioteca vendorizada -4. Criar Makefile/script de build -5. Testar compilacao em x86_64, ARM64 e ARMv7 - -**Arquivos:** -- `CMakeLists.txt` -- `ethercat_plugin.h` -- `ethercat_plugin.c` (esqueleto inicial) - -**Criterio de Aceite:** -- Plugin compila em todas as arquiteturas alvo -- SOEM linkado corretamente - -#### Etapa 1.2: Estrutura Basica do Plugin e Config Parser -**Duracao estimada:** 1-2 semanas - -**Tarefas:** -1. Implementar funcoes de lifecycle (`init`, `start_loop`, `stop_loop`, `cleanup`) -2. Implementar copia segura de `plugin_runtime_args_t` -3. Integrar sistema de logging (`plugin_logger.h`) -4. Implementar parser do JSON de configuracao usando cJSON -5. Implementar validacao estrutural do JSON -6. Mapear JSON para estruturas C internas - -**Arquivos:** -- `ethercat_plugin.c` -- `ethercat_config.h` -- `ethercat_config.c` - -**Estruturas C para configuracao:** -```c -// Tamanhos maximos -#define ECAT_MAX_SLAVES 64 -#define ECAT_MAX_MODULES 32 -#define ECAT_MAX_CHANNELS 64 -#define ECAT_MAX_PDO_ENTRIES 32 -#define ECAT_MAX_PDOS 16 -#define ECAT_MAX_SDOS 32 -#define ECAT_MAX_NAME_LEN 64 -#define ECAT_MAX_IEC_LOC_LEN 16 - -typedef struct { - char index[12]; // hex string "0x6000" - uint8_t subindex; - uint8_t bit_length; - char name[ECAT_MAX_NAME_LEN]; - char data_type[12]; // "BOOL", "INT16", "PAD", etc. -} ecat_pdo_entry_t; - -typedef struct { - char index[12]; // hex string "0x1A00" - char name[ECAT_MAX_NAME_LEN]; - ecat_pdo_entry_t entries[ECAT_MAX_PDO_ENTRIES]; - int entry_count; -} ecat_pdo_t; - -typedef struct { - char index[12]; // hex string "0x8000" - uint8_t subindex; - int32_t value; - char data_type[12]; - char name[ECAT_MAX_NAME_LEN]; -} ecat_sdo_config_t; - -typedef struct { - int index; - char name[ECAT_MAX_NAME_LEN]; - char type[20]; // "digital_input", "analog_output", etc. - uint8_t bit_length; - char iec_location[ECAT_MAX_IEC_LOC_LEN]; - char pdo_index[12]; - char pdo_entry_index[12]; - uint8_t pdo_entry_subindex; -} ecat_channel_t; - -typedef struct { - int position; // ec_slave[position] na SOEM (1-based) - char name[ECAT_MAX_NAME_LEN]; - char type[20]; // "coupler", "digital_input", etc. - uint32_t vendor_id; - uint32_t product_code; - uint32_t revision; - ecat_channel_t channels[ECAT_MAX_CHANNELS]; - int channel_count; - ecat_sdo_config_t sdo_configs[ECAT_MAX_SDOS]; - int sdo_count; - ecat_pdo_t rx_pdos[ECAT_MAX_PDOS]; - int rx_pdo_count; - ecat_pdo_t tx_pdos[ECAT_MAX_PDOS]; - int tx_pdo_count; -} ecat_slave_t; - -typedef struct { - char interface[32]; - int cycle_time_us; - int watchdog_timeout_cycles; - char log_level[8]; -} ecat_master_config_t; - -typedef struct { - bool log_connections; - bool log_data_access; - bool log_errors; - int max_log_entries; - int status_update_interval_ms; -} ecat_diagnostics_config_t; - -typedef struct { - ecat_master_config_t master; - ecat_slave_t slaves[ECAT_MAX_SLAVES]; // flat list - int slave_count; - ecat_diagnostics_config_t diagnostics; -} ecat_config_t; - -// API -int ecat_config_parse(const char *config_path, ecat_config_t *config); -int ecat_config_validate(const ecat_config_t *config); -void ecat_config_init_defaults(ecat_config_t *config); -void ecat_config_free(ecat_config_t *config); -``` - -**Criterio de Aceite:** -- Plugin carrega e descarrega sem erros -- JSON parseado corretamente para estruturas C -- Validacao rejeita JSON invalido com mensagens claras -- Logs aparecem no sistema centralizado - -#### Etapa 1.3: Inicializacao do Master EtherCAT -**Duracao estimada:** 1 semana - -**Tarefas:** -1. Implementar wrapper para `ec_init()` do SOEM -2. Configurar interface de rede (raw sockets) -3. Implementar scan e validacao de topologia contra JSON -4. Verificar permissoes necessarias (CAP_NET_RAW) -5. Implementar tratamento de erros de inicializacao - -**Arquivos:** -- `ethercat_master.c` -- `ethercat_master.h` - -**Fluxo de inicializacao:** -``` -init(): - 1. Carregar JSON config - 2. Validar estrutura do JSON - -start_loop(): - 3. ec_init(interface) - 4. ec_config_init() - scan da rede - 5. Validar topologia: slaves fisicos == JSON config - - Comparar vendor_id, product_code de cada slave - - Se divergir -> logar erro e abortar - 6. Continuar com configuracao dos slaves -``` - -**Requisitos Atendidos:** -- RNF06: Compatibilidade com qualquer NIC com raw sockets -- RNF07: Linux kernel 4.x ou superior - -**Criterio de Aceite:** -- Master inicializa corretamente em interface de rede -- Topologia validada contra configuracao JSON -- Erros de permissao e topologia reportados adequadamente - ---- - -### Fase 2: Core Features (4-6 semanas) - -#### Etapa 2.1: Maquina de Estados EtherCAT -**Duracao estimada:** 1-2 semanas - -**Tarefas:** -1. Implementar transicoes: Init -> Pre-Op -> Safe-Op -> Op -2. Implementar transicoes de erro e recuperacao -3. Gerenciar estado individual de cada slave -4. Implementar timeout de transicao - -**Arquivos:** -- `ethercat_state_machine.c` -- `ethercat_state_machine.h` - -**Requisitos Atendidos:** -- RF07: Transicionar dispositivos pelos estados EtherCAT -- Criterio: Transicao completa em menos de 2 segundos - -**Criterio de Aceite:** -- Todos os slaves transicionam para estado OP -- Erros de transicao detectados e reportados - -#### Etapa 2.2: Aplicacao de SDOs (Configuracao Pre-Operational) -**Duracao estimada:** 1 semana - -**Tarefas:** -1. Implementar leitura de SDOs (`ec_SDOread`) -2. Implementar escrita de SDOs (`ec_SDOwrite`) -3. Iterar sobre `sdo_configurations` de cada slave no JSON -4. Aplicar SDOs na fase Pre-Op antes de transicionar para Safe-Op -5. Tratar erros de SDO (device nao suporta, valor invalido, etc.) - -**Arquivos:** -- Extensao de `ethercat_master.c` - -**Fluxo:** -``` -Para cada slave no JSON: - Se sdo_configurations nao vazio: - Para cada SDO: - ec_SDOwrite(slave.position, index, subindex, value, sizeof(value)) - Se erro -> logar e decidir (abortar ou continuar) -``` - -**Requisitos Atendidos:** -- RF11: Implementar CoE (CANopen over EtherCAT) -- Criterio: Acesso completo a SDOs - -**Criterio de Aceite:** -- Parametros de slaves configuraveis via SDO conforme JSON -- Erros de SDO tratados adequadamente - -#### Etapa 2.3: Mapeamento de PDOs -**Duracao estimada:** 2 semanas - -**Tarefas:** -1. Calcular offsets de bytes/bits no process data image a partir dos PDOs no JSON -2. Considerar padding entries no calculo de offset -3. Implementar estrutura de mapeamento PDO <-> buffer IEC -4. Parse de iec_location (%IX, %QX, %IW, etc.) para tipo/indice de buffer -5. Construir tabela de mapeamento rapido para uso no ciclo - -**Arquivos:** -- `ethercat_pdo_mapper.c` -- `ethercat_pdo_mapper.h` - -**Estrutura de mapeamento interno:** -```c -typedef enum { - ECAT_DIR_INPUT, // Slave -> PLC (TxPDO) - ECAT_DIR_OUTPUT, // PLC -> Slave (RxPDO) -} ecat_direction_t; - -typedef enum { - ECAT_IEC_BOOL, // 1 bit -> bool_input/bool_output - ECAT_IEC_BYTE, // 8 bit -> byte_input/byte_output - ECAT_IEC_INT, // 16 bit -> int_input/int_output - ECAT_IEC_DINT, // 32 bit -> dint_input/dint_output - ECAT_IEC_LINT, // 64 bit -> lint_input/lint_output -} ecat_iec_type_t; - -typedef struct { - // Localizacao no process data image do SOEM - int slave_index; // Indice do slave no SOEM (0-based) - int pdi_byte_offset; // Offset em bytes no process data do slave - int pdi_bit_offset; // Offset em bits (0-7, para BOOL) - int bit_length; // Tamanho em bits - - // Localizacao no buffer do PLC - ecat_direction_t direction; - ecat_iec_type_t iec_type; - int buffer_index; // Indice no array do buffer - int bit_index; // Bit dentro do buffer (para BOOL) -} ecat_pdo_map_entry_t; - -typedef struct { - ecat_pdo_map_entry_t *entries; - int input_count; - int output_count; - int total_count; -} ecat_pdo_map_t; - -// API -int ecat_pdo_map_build(const ecat_config_t *config, ecat_pdo_map_t *map); -void ecat_pdo_map_free(ecat_pdo_map_t *map); -``` - -**Algoritmo de calculo de offset:** -``` -Para cada slave na lista: - slave_index = slave.position - 1 (0-based para SOEM) - Para cada TxPDO do slave: - bit_offset = 0 - Para cada entry do PDO: - Se entry.data_type != "PAD" E channel mapeado para esta entry: - Criar map_entry com: - - pdi_byte_offset = bit_offset / 8 - - pdi_bit_offset = bit_offset % 8 - - parse iec_location do channel -> direction, iec_type, buffer_index, bit_index - bit_offset += entry.bit_length - (mesmo para RxPDOs) -``` - -**Requisitos Atendidos:** -- RF03: Permitir mapeamento visual de PDOs (suporte no runtime) -- N03: Mapear variaveis do PLC para PDOs - -**Criterio de Aceite:** -- Offsets calculados corretamente a partir do JSON -- Padding considerado no calculo -- Located vars parseadas para tipos/indices de buffer corretos -- Dados fluem entre slaves e image tables - -#### Etapa 2.4: Ciclo de Comunicacao -**Duracao estimada:** 1-2 semanas - -**Tarefas:** -1. Implementar troca ciclica de PDOs (`ec_send_processdata`, `ec_receive_processdata`) -2. Implementar cycle_start: ler process data -> copiar para buffers PLC (inputs) -3. Implementar cycle_end: copiar buffers PLC -> escrever process data (outputs) -4. Usar journal writes para operacoes atomicas nos buffers -5. Verificar working counter a cada ciclo - -**Arquivos:** -- Extensao de `ethercat_plugin.c` -- Extensao de `ethercat_master.c` - -**Fluxo do ciclo:** -``` -cycle_start(): (chamado com mutex held) - 1. ec_receive_processdata(timeout) // Recebe dados dos slaves - 2. Verificar working counter - 3. Para cada input mapping: - Ler bytes/bits do process data image do slave - Escrever no buffer PLC via journal_write_*() - -cycle_end(): (chamado com mutex held) - 1. Para cada output mapping: - Ler bytes/bits do buffer PLC - Escrever no process data image do slave - 2. ec_send_processdata() // Envia dados para os slaves -``` - -**Requisitos Atendidos:** -- RF06: Executar ciclo EtherCAT sincronizado com task -- RNF01: Cycle time minimo de 1 ms -- RNF02: Cycle time recomendado de 4 ms -- RNF03: Jitter maximo de 500 us - -**Criterio de Aceite:** -- Comunicacao ciclica estavel -- Jitter dentro do especificado -- Operacao continua sem erros por 1 hora - ---- - -### Fase 3: Diagnostico e Robustez (2-3 semanas) - -#### Etapa 3.1: Sistema de Diagnostico -**Duracao estimada:** 1-2 semanas - -**Tarefas:** -1. Implementar coleta de status por slave (estado, erros) -2. Implementar contadores de erro (CRC, frame, lost link) -3. Registrar eventos em log com timestamp -4. Expor metricas via estrutura interna - -**Arquivos:** -- `ethercat_diagnostics.c` -- `ethercat_diagnostics.h` - -**Requisitos Atendidos:** -- RF16: Exibir status de cada dispositivo em tempo real -- RF17: Registrar log de erros com timestamp (10.000 eventos) -- RF18: Fornecer contadores de erros por dispositivo -- RF19: Permitir leitura de registradores ESC - -**Estrutura de Diagnostico:** -```c -typedef struct { - int slave_index; - uint16_t al_status; // EtherCAT state - uint16_t al_status_code; // Error code - uint32_t crc_errors; - uint32_t frame_errors; - uint32_t lost_links; - uint64_t last_error_timestamp; -} ecat_slave_diagnostics_t; - -typedef struct { - uint32_t cycle_count; - uint32_t cycle_time_us; - uint32_t max_jitter_us; - uint32_t working_counter_errors; -} ecat_master_diagnostics_t; -``` - -**Criterio de Aceite:** -- Status de slaves disponivel em tempo real -- Erros registrados com timestamp -- Contadores incrementados corretamente - -#### Etapa 3.2: Watchdog de Comunicacao -**Duracao estimada:** 1 semana - -**Tarefas:** -1. Implementar deteccao de perda de comunicacao -2. Implementar acao de watchdog (transicao para Safe-Op) -3. Implementar recuperacao automatica -4. Configurar timeout via `watchdog_timeout_cycles` do JSON - -**Arquivos:** -- Extensao de `ethercat_master.c` - -**Requisitos Atendidos:** -- RF09: Implementar watchdog de comunicacao -- Criterio: Detectar perda em maximo 3 ciclos - -**Criterio de Aceite:** -- Perda de comunicacao detectada rapidamente -- Sistema transiciona para estado seguro -- Recuperacao automatica quando comunicacao retorna - ---- - -### Fase 4: Perfil DS401 e Finalizacao (2-3 semanas) - -#### Etapa 4.1: Suporte ao Perfil DS401 (I/O Devices) -**Duracao estimada:** 1-2 semanas - -**Tarefas:** -1. Validar mapeamento padrao DS401 para I/O digital (funciona via PDO mapping do JSON) -2. Validar mapeamento padrao DS401 para I/O analogico -3. Testar com modulos I/O de diferentes fabricantes -4. Tratar particularidades de cada tipo de modulo (scaling, ranges, etc.) - -**Requisitos Atendidos:** -- RF15: Suportar perfil DS401 (I/O Devices) - -**Criterio de Aceite:** -- Modulos I/O digitais funcionais -- Modulos I/O analogicos funcionais -- Compatibilidade com pelo menos 2 fabricantes - -#### Etapa 4.2: API REST para Status -**Duracao estimada:** 1 semana - -**Tarefas:** -1. Expor status EtherCAT via REST API do webserver -2. Implementar endpoints para: - - Lista de slaves - - Status individual de slave - - Metricas do master - - Logs de eventos - -**Endpoints:** -``` -GET /api/ethercat/slaves # Lista slaves detectados -GET /api/ethercat/slaves/{id} # Status de slave especifico -GET /api/ethercat/master/status # Metricas do master -GET /api/ethercat/diagnostics # Diagnostico completo -``` - -**Criterio de Aceite:** -- Endpoints funcionais -- Dados atualizados em tempo real - -#### Etapa 4.3: Testes e Documentacao -**Duracao estimada:** 1 semana - -**Tarefas:** -1. Criar testes unitarios (cobertura minima 80%) -2. Criar testes de integracao -3. Testar em hardware real (Beckhoff, Omron) -4. Escrever documentacao tecnica -5. Criar 3 projetos de exemplo - -**Requisitos Atendidos:** -- RNF14: Cobertura minima de 80% em testes unitarios -- RNF12: Minimo 3 projetos de exemplo - -**Criterio de Aceite:** -- Testes passando -- Documentacao completa -- Exemplos funcionais - ---- - -## 5. Dependencias e Requisitos de Sistema - -### 5.1 Dependencias de Build -- CMake 3.10+ -- GCC/Clang com suporte a C99 -- SOEM library (Git submodule) -- cJSON (ja utilizado no projeto para parsing JSON) -- pthread - -**Nota:** libxml2 **nao e necessaria** no Runtime. O parsing de ESI (XML) e feito -inteiramente no Editor. - -### 5.2 Dependencias de Runtime -- Linux kernel 4.x+ com suporte a raw sockets -- Permissao CAP_NET_RAW ou execucao como root -- Interface de rede Ethernet - -### 5.3 Plataformas Alvo -- x86_64 (principal) -- ARM64 (Raspberry Pi 4, Orange Pi) -- ARMv7 (Raspberry Pi 3, BeagleBone) - ---- - -## 6. Criterios de Qualidade - -### 6.1 Performance -- Cycle time minimo: 1 ms -- Cycle time recomendado: 4 ms -- Jitter maximo: 500 us -- Operacao continua: 72 horas sem erros criticos - -### 6.2 Codigo -- Seguir padrao do projeto: snake_case, 4-space indent -- Documentacao inline para funcoes publicas -- Sem warnings em compilacao - -### 6.3 Testes -- Cobertura unitaria: 80%+ -- Testes de integracao com hardware real -- Testes de estresse (carga maxima por 24h) - ---- - -## 7. Riscos e Mitigacoes - -| Risco | Mitigacao | -|-------|-----------| -| Complexidade SOEM | Spike tecnico inicial, consulta a documentacao | -| Incompatibilidade de dispositivos | Testar com multiplos fabricantes desde inicio | -| Performance em ARM | Benchmarks continuos em hardware alvo | -| Jitter excessivo | Usar PREEMPT_RT kernel quando necessario | -| JSON do Editor incompleto/invalido | Validacao rigorosa no init() com mensagens claras | -| Divergencia topologia vs JSON | Validacao no start_loop() com log detalhado | - ---- - -## 8. Referencias - -- SOEM: https://github.com/OpenEtherCATsociety/SOEM -- ETG.1000: EtherCAT Specification -- CiA 401: Device Profile for I/O Modules -- Plugin S7Comm existente: `core/src/drivers/plugins/native/s7comm/` -- Plugin Driver API: `core/src/drivers/plugin_driver.h` -- Discovery Service (ja implementado): `webserver/discovery/` diff --git a/docs/opcua/OPCUA_AUTHENTICATION_REVIEW.md b/docs/opcua/OPCUA_AUTHENTICATION_REVIEW.md deleted file mode 100644 index 155274c6..00000000 --- a/docs/opcua/OPCUA_AUTHENTICATION_REVIEW.md +++ /dev/null @@ -1,182 +0,0 @@ -# OPC-UA Plugin Authentication Implementation Report - -**Date:** 2026-01-22 -**Branch:** RTOP-100-OPC-UA -**asyncua Version:** 1.1.8 - -## Executive Summary - -The OpenPLC OPC-UA plugin's username/password authentication implementation **is correctly aligned with asyncua 1.1.8 patterns**. The implementation follows the recommended approach from the asyncua library documentation and community examples. - ---- - -## Comparison Table: OpenPLC vs asyncua 1.1.8 - -| Aspect | asyncua 1.1.8 Pattern | OpenPLC Implementation | Status | -|--------|----------------------|------------------------|--------| -| **UserManager Interface** | Extends `UserManager` base class | `OpenPLCUserManager(UserManager)` | Correct | -| **get_user signature** | `get_user(self, iserver, username=None, password=None, certificate=None)` | Exact same signature at `user_manager.py:88-94` | Correct | -| **Return value** | `User` object with `role` attribute, or `None` | Returns user object with `role` (UserRole enum) or `None` | Correct | -| **Server integration** | `Server(user_manager=UserManager())` | `Server(user_manager=self.user_manager)` at `server.py:188` | Correct | -| **UserRole enum** | `from asyncua.server.user_managers import UserRole` | Same import at `user_manager.py:15` | Correct | -| **Password storage** | No specific requirement | bcrypt hashes (industry standard) | Good | - ---- - -## Detailed Analysis - -### 1. UserManager Class Implementation (`user_manager.py`) - -**Correct Implementation:** -```python -# Line 15: Correct import from asyncua -from asyncua.server.user_managers import UserManager, UserRole - -# Line 41: Proper inheritance -class OpenPLCUserManager(UserManager): - ... - -# Lines 88-94: Correct method signature -def get_user( - self, - iserver, - username: Optional[str] = None, - password: Optional[str] = None, - certificate: Optional[Any] = None -) -> Optional[Any]: -``` - -This matches the asyncua documentation exactly: -```python -# asyncua pattern: -class UserManager: - def get_user(self, iserver, username=None, password=None, certificate=None): - raise NotImplementedError -``` - -### 2. Server Integration (`server.py:188`) - -**Correct Implementation:** -```python -# Line 188: Passes user_manager to Server constructor -self.server = Server(user_manager=self.user_manager) -``` - -This aligns with asyncua's recommended pattern: -```python -# asyncua documentation: -server = Server(user_manager=UserManager()) -``` - -### 3. Role Mapping (`user_manager.py:56-61`) - -**Implementation:** -```python -ROLE_MAPPING = { - "viewer": UserRole.User, # Read-only access - "operator": UserRole.User, # Read/write via callbacks - "engineer": UserRole.Admin # Full access -} -``` - -This is consistent with asyncua's `UserRole` enum which has `User` and `Admin` levels. - -### 4. Password Validation (`user_manager.py:369-389`) - -**Strengths:** -- Uses bcrypt for password hashing (industry standard) -- Fails securely if bcrypt is unavailable -- No plaintext password storage - -**Implementation:** -```python -def _validate_password(self, password: str, password_hash: str) -> bool: - if _bcrypt_available: - try: - return bcrypt.checkpw(password.encode(), password_hash.encode()) - except Exception as e: - log_error(f"bcrypt validation error: {e}") - return False - else: - log_error("bcrypt not available - password authentication disabled for security") - return False -``` - ---- - -## Minor Observations (Not Issues) - -| Item | Current State | asyncua Default | Impact | -|------|--------------|-----------------|--------| -| User return type | `SimpleNamespace` / config `User` | `User` from `asyncua.crypto.permission_rules` | Works correctly - asyncua only checks for `role` attribute | -| Anonymous users | `SimpleNamespace()` with `role` | `User(role=UserRole.User)` | Functionally equivalent | - -The implementation returns user objects that have the required `role` attribute, which is all asyncua needs for authorization decisions. - ---- - -## Test Coverage Gap - -**Finding:** No unit tests exist for the `OpenPLCUserManager` class. - -**Recommendation:** Consider adding tests for: -- Password authentication success/failure -- Certificate authentication success/failure -- Anonymous authentication with profile restrictions -- Role mapping verification - ---- - -## Configuration Validation - -The current config file (`opcua.json`) shows proper usage: - -```json -{ - "users": [ - { - "type": "certificate", - "certificate_id": "engineer_cert", - "role": "engineer" - }, - { - "type": "password", - "username": "operator", - "password_hash": "$2b$10$Y/WT4Z8ku9hObwSPk1bmY...", - "role": "operator" - } - ] -} -``` - ---- - -## Conclusion - -**The implementation is healthy and correctly follows asyncua 1.1.8 patterns.** - -No changes are required for core functionality. The implementation: -1. Uses the correct `UserManager` interface -2. Has the correct `get_user()` signature -3. Integrates properly with asyncua `Server` -4. Uses appropriate security practices (bcrypt hashing) - ---- - -## Optional Improvements (Not Required) - -| Priority | Improvement | Rationale | -|----------|-------------|-----------| -| Low | Add unit tests for `OpenPLCUserManager` | Increase confidence in auth logic | -| Low | Return asyncua's `User` class directly | Closer adherence to asyncua patterns (not required for functionality) | -| Low | Add rate limiting on auth attempts | Security hardening against brute force | - ---- - -## References - -- [Server set User with Password - GitHub Discussion #1386](https://github.com/FreeOpcUa/opcua-asyncio/discussions/1386) -- [Server with Authentication (user/password) and Encryption - GitHub Discussion #934](https://github.com/FreeOpcUa/opcua-asyncio/discussions/934) -- [asyncua PyPI](https://pypi.org/project/asyncua/) -- [asyncua server.py](https://github.com/FreeOpcUa/opcua-asyncio/blob/master/asyncua/server/server.py) -- [asyncua server-with-encryption.py example](https://github.com/FreeOpcUa/opcua-asyncio/blob/master/examples/server-with-encryption.py) diff --git a/docs/opcua/OPCUA_SECURITY_MODE_INSUFFICIENT_ANALYSIS.md b/docs/opcua/OPCUA_SECURITY_MODE_INSUFFICIENT_ANALYSIS.md deleted file mode 100644 index 73eca55b..00000000 --- a/docs/opcua/OPCUA_SECURITY_MODE_INSUFFICIENT_ANALYSIS.md +++ /dev/null @@ -1,146 +0,0 @@ -# OPC-UA BadSecurityModeInsufficient Error Analysis - -**Date:** 2026-01-22 -**Context:** Username/password authentication over insecure (unencrypted) endpoint - -## Summary - -When using only the insecure security profile with Username authentication, OPC-UA clients display the error: - -> Error 'BadSecurityModeInsufficient' was returned during ActivateSession, press 'Ignore' to suppress the error and continue connecting. If you ignore the error it is possible that the password is being sent in clear text. - -This is **NOT a bug** - it's a security feature defined in the OPC-UA specification. - ---- - -## What's Happening - -This error is a **security feature** defined in the OPC-UA specification (Part 4, Section 7.36). - -**The error comes from the OPC-UA client** (like UAExpert), not the server. When you configure: -- Security Policy: `None` -- Security Mode: `None` -- Auth Method: `Username` (password authentication) - -The client detects that it would send the password **in plain text** over the network and warns the user. This is intentional behavior to protect against accidentally exposing credentials. - ---- - -## OPC-UA Security Architecture - -OPC-UA has **two separate security layers**: - -| Layer | Purpose | Description | -|-------|---------|-------------| -| **Channel Security** | Encrypts communication between client/server | Configured via `security_policy` and `security_mode` | -| **Token Security** | Can encrypt user credentials separately | Can have its own SecurityPolicyUri | - -When both are "None", passwords travel unencrypted over the network. - -### Security Policy Options - -| Policy | Mode | Result | -|--------|------|--------| -| `None` | `None` | No encryption (plaintext) | -| `Basic256Sha256` | `Sign` | Messages are signed (integrity) | -| `Basic256Sha256` | `SignAndEncrypt` | Full encryption (confidentiality + integrity) | - ---- - -## Configuration Options - -### Option 1: Use Encrypted Security Profile (Recommended) - -Keep the `SignAndEncrypt` profile enabled alongside the insecure one: - -```json -"security_profiles": [ - { - "name": "insecure", - "enabled": true, - "security_policy": "None", - "security_mode": "None", - "auth_methods": ["Anonymous"] - }, - { - "name": "SignAndEncrypt", - "enabled": true, - "security_policy": "Basic256Sha256", - "security_mode": "SignAndEncrypt", - "auth_methods": ["Username", "Certificate"] - } -] -``` - -This configuration: -- Allows Anonymous access on the insecure endpoint -- Requires encryption for Username/password authentication -- Follows OPC-UA security best practices - -### Option 2: Accept the Risk (Click "Ignore") - -If you're on a trusted local network (like a lab environment), clicking "Ignore" in the OPC-UA client will: -- Send the password in plaintext -- Connection will work normally -- **Only use this in isolated/trusted networks** - -**Warning:** This exposes credentials to network sniffing attacks. - -### Option 3: Token-Level Encryption (Advanced) - -OPC-UA allows the UserIdentityToken to have its own security policy, even when channel security is "None". This means you could theoretically: -- Use `None` for channel security (no message encryption) -- Use `Basic256Sha256` for token security (password is encrypted) - -This requires additional configuration in asyncua and is not currently implemented. - ---- - -## Recommendations - -| Environment | Recommendation | -|-------------|----------------| -| **Production / Industrial** | Use Option 1 - require encryption for password auth | -| **Development / Testing** | Option 2 is acceptable on isolated networks | -| **Internet-facing** | Always use SignAndEncrypt with certificates | - ---- - -## Technical Details - -### Error Code - -- **Status Code:** `BadSecurityModeInsufficient` (0x80E60000) -- **Meaning:** "The operation is not permitted over the current secure channel" - -### Where the Check Occurs - -The security check happens during the `ActivateSession` phase: -1. Client connects to server (OpenSecureChannel) -2. Client creates session (CreateSession) -3. Client activates session with credentials (ActivateSession) - **Error occurs here** - -The client library checks if sending credentials over the current security mode is safe before transmitting. - -### asyncua Behavior - -The asyncua library includes logic to: -1. Warn when creating open endpoints alongside encrypted ones -2. Try to find an encrypting policy for password transmission -3. Log warnings when no encrypting policy is available - -From `asyncua/server/server.py`: -```python -# try to avoid plaintext password, find first policy with encryption -# ... -# No encrypting policy available, password may get transferred in plaintext -``` - ---- - -## References - -- [OPC UA Part 4: Services - 7.37 UserTokenPolicy](https://reference.opcfoundation.org/Core/Part4/v104/docs/7.37) -- [Server with Authentication (user/password) and Encryption - GitHub Discussion #934](https://github.com/FreeOpcUa/opcua-asyncio/discussions/934) -- [Server set User with Password - GitHub Discussion #1386](https://github.com/FreeOpcUa/opcua-asyncio/discussions/1386) -- [asyncua PyPI](https://pypi.org/project/asyncua/) diff --git a/docs/plans/ONLINE_PROGRAMMING_PLAN.md b/docs/plans/ONLINE_PROGRAMMING_PLAN.md deleted file mode 100644 index 3a34e8b1..00000000 --- a/docs/plans/ONLINE_PROGRAMMING_PLAN.md +++ /dev/null @@ -1,192 +0,0 @@ -# Online Programming Plan — OpenPLC Runtime v4 (Linux targets) - -> **Status:** brainstorm / pre-design. Not committed; see *Decisions needed* at the end. - -## Background - -One of the most-requested OpenPLC features is **online programming** — change a small piece of the running program without stopping the PLC. In interpreted-bytecode runtimes this is trivial: swap the program text between scan cycles. We compile, so the unit of change is a freshly-built `.so`, and the trick is preserving program state across the swap. - -This document maps the current runtime-v4 / strucpp architecture to that requirement, proposes a staged delivery, and enumerates the limitations and risks. - ---- - -## What's already in place that helps - -Three pieces of the existing architecture are *already* shaped for hot-swap; we just don't use them that way yet: - -### 1. `.so` is loaded via `dlopen(..., RTLD_NOW)` and the runtime executable holds zero user-program state - -Everything user-defined — POU struct instances, `IECVar` storage, RETAIN vars, FB internals (timer counters, edge-detect flip-flops) — lives in the `.so`'s `.bss` / `.data`. The runtime keeps only: - -- Function pointers (`ext_strucpp_*`) into the `.so`'s exports (resolved in `image_tables.cpp::symbols_init`). -- A cached `ConfigurationInstance*` (`g_config_ptr`) returned by the `.so`'s `strucpp_get_config()`. -- Runtime-owned mutexes (`image_tables_mutex`, `global_vars_mutex`) whose addresses are *handed to* the `.so` via `strucpp_set_locks`. - -Mechanically, unload-and-reload is one `dlclose(handle)` + one `dlopen(new_path, RTLD_NOW)` away. The state-transplant problem is the hard part, not the dynamic-loading machinery. - -### 2. The debug-table gives every leaf variable a stable nominal identity - -strucpp's `debug-table-gen.ts` emits two artifacts per compile: - -- **In the `.so`:** a `debug_arrays[][]` of `Entry { void* ptr; uint8_t tag; }` (declared in `debug_table.hpp`, defined in `generated_debug.cpp`). Flat addressable by `(arrayIdx, elemIdx)`. -- **Out-of-band JSON `debugMap`:** maps fully-qualified IEC paths like `"INSTANCE0.speeds[5]"` → the `(arr, elem)` index in this build's table, plus the `TypeTag`. - -Today the editor uses the JSON to do force / read. The same JSON is exactly what we need to **transplant state by path** across the swap. Path = stable nominal identity; `(arr, elem)` = positional address in the *current* program. - -### 3. The scan loop already has a natural quiescence point - -`plc_state_manager.cpp::plc_task_thread` does: - -``` -mutex_lock(image_tables_mutex) - tracker_start - io_cycle_pre # only for the fastest task - for each program: run() - io_cycle_post # only for the fastest task - tracker_end -mutex_unlock(image_tables_mutex) -clock_nanosleep(TIMER_ABSTIME, next_wakeup) -``` - -The image-tables mutex is a recursive PI mutex. The swap window is "after the last task unlocks, before any task next wakes" — already a no-touch zone for the program data. - ---- - -## The swap mechanism, end-to-end - -``` -┌────────────────────────────────────────────────────────────────────┐ -│ 1. Compile new program out-of-band at SCHED_OTHER nice 19 │ -│ (or SCHED_IDLE) so it can't preempt SCHED_FIFO task threads. │ -│ Produce: libplc_.so + debugMap_new.json. │ -├────────────────────────────────────────────────────────────────────┤ -│ 2. DRIFT GATE: compare debugMap_old vs debugMap_new (see below). │ -│ If REJECT → bail. If WARN → expose preview to editor user. │ -├────────────────────────────────────────────────────────────────────┤ -│ 3. ARM the swap. Set an atomic `pending_swap = new_so_path`. │ -│ Plugins stay running; their state is untouched. │ -├────────────────────────────────────────────────────────────────────┤ -│ 4. SNAPSHOT. The fastest task, AFTER its `io_cycle_post`, BEFORE │ -│ `clock_nanosleep`, sees pending_swap != NULL and: │ -│ a. Suspends every OTHER task thread (cooperative — they │ -│ park at the top of their loop on a per-task condvar). │ -│ b. Walks every leaf in debug_arrays[] in the OLD .so, reads │ -│ IECVar::{value_, forced_, forced_value_} via the │ -│ existing `strucpp_debug_read` plus a new │ -│ `strucpp_debug_read_force_state` accessor, stores them │ -│ keyed by the debugMap_old path → an in-runtime arena. │ -│ c. Captures the `RetainVarInfo` table from each │ -│ `ProgramBase::getRetainVars()` and copies the retain bytes │ -│ by leaf path as well. │ -│ d. Captures `__CURRENT_TIME_NS` (the per-.so monotonic time │ -│ the runtime owns via `ext_strucpp_advance_time`) — the │ -│ new .so resumes from the same time so TON / TOF / CTU │ -│ internal deadlines stay continuous. │ -├────────────────────────────────────────────────────────────────────┤ -│ 5. SWAP. Under image_tables_mutex: │ -│ a. image_tables_clear_null_pointers() │ -│ b. dlclose(old handle) │ -│ c. dlopen(new .so, RTLD_NOW) │ -│ d. symbols_init(new pm) — re-resolves all ext_strucpp_* │ -│ e. image_tables_bind_located_vars() against new .so │ -├────────────────────────────────────────────────────────────────────┤ -│ 6. RESTORE. For each leaf path in debugMap_old: │ -│ - If new program has the same path AND same TypeTag: │ -│ write value_ via strucpp_debug_write, │ -│ re-apply force via strucpp_debug_set(forcing=true) │ -│ - If type promoted (INT → DINT) within widening rules: │ -│ reinterpret + assign (lossless), reapply force. │ -│ - If path missing in new program (variable was deleted): │ -│ journal the dropped value, continue. │ -│ - New paths not in old: keep their compile-time default. │ -│ Re-seed `strucpp_advance_time` with the saved CURRENT_TIME_NS │ -│ so timer FBs continue counting from where they were. │ -├────────────────────────────────────────────────────────────────────┤ -│ 7. RESUME. Release suspended tasks. The fastest task continues │ -│ into its already-scheduled `next_wakeup` — drift visible to │ -│ plugins is bounded by the snapshot+swap+restore cost. │ -└────────────────────────────────────────────────────────────────────┘ -``` - ---- - -## Drift gate — what changes are safe - -The two `debugMap` JSONs are compared server-side before arming. Buckets: - -| Change | Old leaf path | New leaf path | Verdict | -| ----------------------------------------------------------------- | ----------------- | ----------------- | ------------------------------------------------------------------------------------------------------------------------ | -| **Identity preserved** | `A.x: INT` | `A.x: INT` | ✅ transplant value + force | -| **Pure addition** | (absent) | `A.y: INT` | ✅ new leaf gets compile-time default | -| **Pure deletion** | `A.x: INT` | (absent) | ✅ value journaled and dropped | -| **Type widening, lossless** | `A.x: INT` | `A.x: DINT` | ✅ cast + transplant — whitelist: SINT→INT→DINT→LINT, REAL→LREAL, signed-to-wider-signed only | -| **Type narrowing** | `A.x: DINT` | `A.x: INT` | ⚠️ if `|value_| ≤ INT_MAX` proceed with cast, else REJECT (lossy) | -| **Type change cross-family** | `A.x: REAL` | `A.x: STRING` | ❌ REJECT — no defined transplant | -| **Located-variable address remap** | `A.x AT %IX0.0` | `A.x AT %IX0.1` | ⚠️ accepted but live wire signal "moves" at swap instant — flag as DISRUPTIVE | -| **Task topology change** (interval, priority, add, remove) | | | ❌ REJECT — would invalidate step 4a's suspend list (threads no longer exist) | -| **Resource topology change** (resources added / removed) | | | ❌ REJECT — multi-resource swap is a different beast | -| **Struct layout change** within named UDT (add / remove / reorder)| | | ⚠️ if every retained leaf path still resolves with same type → ACCEPT; else inspect per-field | -| **FB internal state reset** (timer FB's PT preset changed) | | | ⚠️ ET (elapsed) preserved if path matches; new PT takes over at next `.EN` edge — flag as SEMANTIC-CHANGE | -| **Code change inside a POU body** (no leaf changes) | | | ✅ trivial — just swap the `.so` | - -The compile-time `debugMap` makes all of these mechanically checkable without parsing source. The editor can show the diff preview ("these 3 leaves will be added, this 1 will be dropped; values for the other 247 will be preserved") before the user commits. - ---- - -## Limitations and risks - -### Hard limitations - -1. **PI mutex held across `dlclose`.** `image_tables_mutex` is created by the runtime, but its address is handed to the `.so` via `strucpp_set_locks`. If a stray plugin thread is blocked on it when we unload, we get a use-after-free on wakeup. Mitigation: drain plugins from that mutex first (every plugin already exits its cycle window before `io_cycle_post` returns). -2. **C/C++ POUs compile into the `.so`.** The user's `c_blocks_code.cpp` static state isn't reachable via the debug table — `static int counter` inside a user POU is lost across swap. Either document this (users shouldn't keep state outside IEC variables) or have strucpp generate a per-POU retain blob. -3. **Pointers stored in IEC variables (REF_TO) become invalid across swap** because target addresses move when the new `.so`'s `.bss` lands at a different vmaddr. Likely needs path-relative pointer canonicalization at snapshot time and re-resolution at restore. -4. **Plugins that opened sockets bound to IEC variable addresses** — Modbus mappings via image-tables, OPC-UA nodes that cache `&IECVar::value_` — need to re-bind after the swap. Extend the `_clear_null_pointers` / `_bind_located_vars` pattern: every plugin gets a `plugin_on_program_swap()` notification. -5. **PREEMPT_RT cycle jitter budget.** Snapshot + swap + restore in step 4–6 has to fit in less than the fastest task's interval, OR we accept a single missed cycle. With ~2000 leaves at ~200 ns each for `debug_read` + ~100 ns mutex ops, snapshot is ~600 µs; **`dlopen` on a Pi 4 from page cache is 5–20 ms** — that's the long pole. For a 10 ms task we cannot do this in one cycle window. Two paths: - - **(a)** Accept a documented "one missed cycle" during swap. - - **(b)** Keep both `.so`s mapped simultaneously and only switch the function-pointer table at the end of `io_cycle_post`, deferring `dlclose` of the old one to a later quiet period. - -### Risks worth spiking early - -- **PIC `.bss` aliasing.** Two `dlopen`'d `.so`s with the same symbol names — the second `dlopen` won't ABI-collide if symbols are bound `RTLD_LOCAL` / `RTLD_DEEPBIND`. strucpp's symbol resolution path currently assumes `RTLD_NOW` from the global namespace. Worth a test before committing to "both `.so`s mapped at once" (the long-pole mitigation above). -- **Python plugin loader** keeps Python interpreter state. The GIL holders and the embedded interpreter are runtime-owned (not in the user `.so`), so this should survive. User-defined Python scripts that bind to specific variable names need a graceful re-resolution. -- **Debug-client liveness.** If the editor is actively in a debugger session when the swap fires, the address tables it cached are stale. Need a "program-revision token" that increments on every swap; the editor protocol re-resolves on mismatch. -- **MD5-based program identity.** `strucpp_program_md5` is already exported and tracked. The drift gate should use `md5_old ≠ md5_new` as the trigger gate (no swap on identical builds), but the **actual** drift comparison must use the structural `debugMap` diff — MD5 is only "did anything change at all". - ---- - -## Staged delivery - -Ship this in three releases rather than a v5 monolith: - -### Stage 1 — Cold reload primitive - -Runtime-side `/api/reload-program` endpoint that does the full swap but **stops the PLC first** (`PLC_STATE_STOPPED → swap → PLC_STATE_RUNNING`). No state preservation. Validates the dlopen / dlclose plumbing without committing to the harder semantics. Editor can use it as "soft restart" — no power cycle, no plugin re-init, but no continuity either. - -### Stage 2 — State transplant, identity-only - -Adds the debugMap-driven snapshot / restore for the IDENTITY-PRESERVED + PURE-ADDITION + PURE-DELETION categories. All other categories REJECT at the gate. This is the version ~80 % of users would call "online programming" — change a POU body, tweak constants, add a variable, all preserve state. - -### Stage 3 — Type / structure drift, located remap, FB internal preservation - -The category whitelist expands; the editor shows the diff preview; users opt in to risky swaps. This is where most of the engineering will actually live. - ---- - -## Decisions needed before coding - -1. **PREEMPT_RT jitter budget — which side of the trade?** Is "one missed scan cycle during swap" acceptable, or is the "both `.so`s mapped simultaneously" approach required from the start? That decision shapes step 5–6 of the swap mechanism. -2. **C/C++ POU static state — owned by the user or by us?** If the answer is "users shouldn't keep state outside IEC variables," document it. If it's "we should preserve it," strucpp needs a retain blob per user POU. -3. **Drift policy default for the gate** — strict (REJECT unless explicitly opted in) or permissive (warn-and-proceed)? Working assumption: strict-by-default + editor-side "Force online change anyway" toggle. - ---- - -## References in the current codebase - -- `core/src/lib/strucpp_abi.hpp` — runtime-side ABI mirror of `ProgramBase` / `TaskInstance` / `ConfigurationInstance`. -- `core/src/plc_app/image_tables.{h,cpp}` — symbol resolution, `image_tables_bind_located_vars`, `image_tables_clear_null_pointers`. -- `core/src/plc_app/plcapp_manager.{h,c}` — `dlopen` / `dlsym` / `dlclose` wrapper. -- `core/src/plc_app/plc_state_manager.cpp::plc_task_thread` — per-task scan loop with mutex / `clock_nanosleep` rhythm. -- `core/src/plc_app/debug_handler.c` — wire protocol on top of `strucpp_debug_{set,read,write}`. -- `strucpp/src/runtime/include/debug_table.hpp` — `Entry` shape, `TypeTag` enum. -- `strucpp/src/runtime/include/debug_dispatch.hpp` — per-type `force_impl` / `unforce_impl` / `read_impl` / `write_impl`. -- `strucpp/src/backend/debug-table-gen.ts` — emits `generated_debug.cpp` *and* the out-of-band `debugMap` JSON keyed by IEC path. diff --git a/docs/pr-reviews/PR_REVIEW_CHECKLIST.md b/docs/pr-reviews/PR_REVIEW_CHECKLIST.md deleted file mode 100644 index 679ac7e1..00000000 --- a/docs/pr-reviews/PR_REVIEW_CHECKLIST.md +++ /dev/null @@ -1,574 +0,0 @@ -# Pull Request Review Checklist - -This document standardizes the review process for OpenPLC Runtime pull requests. Use this checklist to ensure code quality, prevent technical debt, and avoid runtime errors. - -## Quick Checklist - -Before approving any PR, verify: - -- [ ] Pre-commit hooks pass (`pre-commit run --all-files`) -- [ ] All tests pass (`pytest tests/`) -- [ ] No compiler warnings with strict flags -- [ ] Memory management is correct (no leaks) -- [ ] Thread safety verified (mutex usage correct) -- [ ] Security considerations addressed -- [ ] Platform compatibility maintained - ---- - -## 1. Code Style and Formatting - -### C/C++ Code -- [ ] 4-space indentation, no tabs -- [ ] 100-character line limit -- [ ] `snake_case` for functions and variables -- [ ] `snake_case_t` for type definitions -- [ ] `UPPER_CASE` for macros and constants -- [ ] Allman brace style for functions -- [ ] Clang-Format validates: `clang-format --style=file --dry-run -Werror *.c *.h` - -### Python Code -- [ ] Black formatter passes -- [ ] isort import ordering correct -- [ ] Ruff linter passes -- [ ] Type hints on function signatures -- [ ] 100-character line limit -- [ ] Double quotes for strings - -### General -- [ ] No emojis in code, comments, or documentation -- [ ] No trailing whitespace -- [ ] Files end with newline -- [ ] No files larger than 500KB - ---- - -## 2. Architecture and Design - -### Dual-Process Architecture -- [ ] Changes respect process boundaries (Python REST API vs C/C++ Runtime) -- [ ] IPC protocol compatibility maintained (`/run/runtime/plc_runtime.socket`) -- [ ] Socket message format unchanged or versioned properly -- [ ] Log socket protocol compatible (`/run/runtime/log_runtime.socket`) - -### State Machine Integrity -``` -EMPTY -> INIT -> RUNNING <-> STOPPED -> ERROR -``` -- [ ] State transitions are atomic (mutex held) -- [ ] No invalid state transitions introduced -- [ ] State changes logged appropriately -- [ ] Error states handled with recovery path - -### Plugin System -- [ ] Plugin interface contract maintained (`init`, `start_loop`, `stop_loop`, `cycle_start`, `cycle_end`, `cleanup`) -- [ ] `plugins.conf` format compatible -- [ ] Dynamic loading error handling present (`dlopen`/`dlsym` checks) -- [ ] Plugin cleanup called on errors -- [ ] No resource leaks when plugins fail to load - ---- - -## 3. Memory Management - -### C/C++ Memory -- [ ] Every `malloc()`/`calloc()` has corresponding `free()` -- [ ] Memory freed in error paths (early returns) -- [ ] Pointers set to `NULL` after `free()` -- [ ] `calloc()` preferred over `malloc()` (zeroed memory) -- [ ] No buffer overflows: `strncpy()`, `snprintf()` used with correct sizes -- [ ] String buffers null-terminated after `strncpy()` - -### Dynamic Loading -- [ ] `dlopen()` result checked for NULL -- [ ] `dlsym()` errors handled -- [ ] `dlclose()` called on cleanup -- [ ] Error messages include `dlerror()` - -### Python Memory -- [ ] Context managers (`with`) used for files/sockets -- [ ] Large buffers explicitly cleaned up -- [ ] No circular references preventing GC -- [ ] Thread-safe data structures where needed - ---- - -## 4. Thread Safety and Concurrency - -### Mutex Usage -- [ ] Lock/unlock pairs are symmetric (no double-lock) -- [ ] No potential deadlocks (consistent lock ordering) -- [ ] Priority inheritance used for real-time mutexes on Linux -- [ ] Graceful fallback for non-Linux platforms - -### Critical Sections -- [ ] `state_mutex` held during PLC state changes -- [ ] `buffer_mutex` held during image table access -- [ ] Minimal time spent holding locks -- [ ] No blocking operations while holding mutex - -### Thread Lifecycle -- [ ] Threads properly joined on shutdown -- [ ] No thread leaks (count remains deterministic) -- [ ] Thread-local storage cleaned up -- [ ] Signal handlers are async-signal-safe - -### Real-Time Considerations (Linux) -- [ ] `SCHED_FIFO` scheduling preserved for PLC thread -- [ ] `mlockall()` called to prevent page faults -- [ ] No dynamic memory allocation in scan cycle -- [ ] Deterministic timing maintained - ---- - -## 5. Error Handling - -### C Error Patterns -- [ ] Return codes checked (0 = success, -1 = failure) -- [ ] `log_error()` called with context on failures -- [ ] No uninitialized variables -- [ ] All heap allocations checked for NULL -- [ ] Error paths clean up resources - -### Python Error Patterns -- [ ] Specific exceptions caught (not bare `except:`) -- [ ] Exceptions logged with context: `logger.error("msg", exc_info=True)` -- [ ] JSON parsing uses `json.JSONDecodeError` -- [ ] Socket errors caught: `socket.error`, `OSError` -- [ ] No swallowing exceptions without logging - -### Graceful Degradation -- [ ] Platform-specific features have fallbacks -- [ ] Missing optional dependencies handled -- [ ] Network timeouts don't crash the system -- [ ] Plugin failures don't crash the runtime - ---- - -## 6. Security - -### Input Validation -- [ ] All user input validated before use -- [ ] Buffer bounds checked before access -- [ ] Socket commands validated against whitelist -- [ ] Debug frame sizes checked: `MAX_DEBUG_FRAME - 7` -- [ ] Variable indices bounds-checked - -### File Operations -- [ ] Path traversal prevented (validate against base directory) -- [ ] Disallowed extensions rejected: `.exe`, `.dll`, `.sh`, `.bat`, `.js`, `.vbs`, `.scr` -- [ ] ZIP extraction uses `safe_extract()` -- [ ] Compression ratio checked (zip bomb prevention) -- [ ] File size limits enforced (10MB per file, 50MB total) - -### Authentication -- [ ] Protected endpoints use `@jwt_required()` -- [ ] Token expiration handled -- [ ] Tokens blacklisted on logout -- [ ] No secrets in version control - -### Password Handling -- [ ] PBKDF2-SHA256 with 600,000 iterations -- [ ] Cryptographic pepper applied -- [ ] Passwords never logged -- [ ] Constant-time comparison used - -### Network Security -- [ ] Hostname validation prevents injection -- [ ] IP addresses parsed via standard library -- [ ] TLS certificates validated (or self-signed with warning) -- [ ] Socket permissions restrict access - -### Plugin Security -- [ ] Plugins use journal API (not direct buffer access) -- [ ] Plugin configuration validated -- [ ] No arbitrary code execution paths - ---- - -## 7. Performance - -### Scan Cycle -- [ ] No regressions in cycle timing (~50ms default) -- [ ] No blocking I/O in scan cycle thread -- [ ] Journal buffer entries within limit (1024 max) -- [ ] Mutex contention minimized - -### Memory -- [ ] No unnecessary allocations in hot paths -- [ ] Buffer sizes appropriate -- [ ] Circular log buffer size reasonable (2MB) - -### Network -- [ ] Socket timeouts appropriate (1.0s default) -- [ ] WebSocket debug overhead acceptable -- [ ] No busy-waiting loops - ---- - -## 8. Testing - -### Test Coverage -- [ ] New features have tests -- [ ] Bug fixes include regression tests -- [ ] Edge cases tested -- [ ] Error paths tested - -### Test Quality -- [ ] Tests use proper mocking (`@patch`) -- [ ] Fixtures clean up state (`reset_globals()`) -- [ ] No test interdependencies -- [ ] Tests are deterministic - -### Running Tests -```bash -sudo bash scripts/setup-tests-env.sh -pytest tests/ -``` - ---- - -## 9. Build System - -### CMake -- [ ] New source files added to `CMakeLists.txt` -- [ ] Include paths correct -- [ ] Link dependencies specified -- [ ] Compiles without warnings: `-Wall -Werror -Wextra` - -### Compiler Flags -Required flags preserved: -``` --Wall -Werror -Wextra -fstack-protector-strong --D_FORTIFY_SOURCE=2 -O2 -Werror=format-security -fPIC -fPIE -``` - -### CI/CD -- [ ] GitHub workflows pass -- [ ] Docker image builds for all platforms (amd64, arm64, arm/v7) -- [ ] Pre-commit hooks configured - ---- - -## 10. Platform Compatibility - -### Linux (Full Support) -- [ ] Real-time scheduling works (`SCHED_FIFO`) -- [ ] Memory locking works (`mlockall`) -- [ ] Priority inheritance enabled - -### Windows/Cygwin/MSYS2 (Graceful Fallback) -- [ ] Compiles without real-time features -- [ ] Warning suppression for Python header conflicts -- [ ] No priority inheritance (falls back to regular mutex) - -### Docker -- [ ] Capabilities documented: `--cap-add=SYS_NICE --cap-add=SYS_RESOURCE` -- [ ] Multi-arch build works - -### ARM (arm64, arm/v7) -- [ ] Cross-compilation works -- [ ] No x86-specific code - ---- - -## 11. Documentation - -### Code Documentation -- [ ] Complex logic has comments explaining "why" -- [ ] Public APIs have docstrings -- [ ] Magic numbers explained or named - -### Project Documentation -- [ ] `CLAUDE.md` updated if architecture changes -- [ ] `README.md` updated for user-facing changes -- [ ] API changes documented - ---- - -## 12. Backward Compatibility - -### Protocol Compatibility -- [ ] Unix socket command protocol unchanged -- [ ] WebSocket debug protocol unchanged -- [ ] REST API endpoints backward compatible - -### Configuration -- [ ] `plugins.conf` format unchanged -- [ ] Environment variables unchanged -- [ ] Database schema migrations provided if needed - -### Plugin API -- [ ] Plugin function signatures unchanged -- [ ] Image table access patterns unchanged -- [ ] Journal API unchanged - ---- - -## Review Categories by Change Type - -### Bug Fixes -Focus on: -- Root cause identified -- Fix addresses root cause (not symptoms) -- Regression test added -- No side effects introduced - -### New Features -Focus on: -- Architecture fits existing patterns -- Error handling comprehensive -- Tests cover happy path and edge cases -- Documentation updated - -### Refactoring -Focus on: -- Behavior unchanged (tests pass) -- No performance regression -- Code cleaner/more maintainable -- No unnecessary changes bundled - -### Security Fixes -Focus on: -- Vulnerability fully addressed -- No new attack vectors -- Regression test prevents reintroduction -- Coordinated disclosure if needed - -### Performance Improvements -Focus on: -- Benchmark results provided -- No correctness regressions -- Edge cases still handled -- Memory usage acceptable - ---- - -## Common Issues to Watch For - -### Memory Leaks -```c -// BAD: Leak on error -char *buf = malloc(size); -if (condition) { - return -1; // buf leaked -} - -// GOOD: Free before return -char *buf = malloc(size); -if (condition) { - free(buf); - return -1; -} -``` - -### Race Conditions -```c -// BAD: Check-then-act race -if (state == RUNNING) { - // Another thread could change state here - do_something(); -} - -// GOOD: Hold mutex -pthread_mutex_lock(&state_mutex); -if (state == RUNNING) { - do_something(); -} -pthread_mutex_unlock(&state_mutex); -``` - -### Buffer Overflows -```c -// BAD: No bounds check -strcpy(dest, src); - -// GOOD: Bounded copy -strncpy(dest, src, sizeof(dest) - 1); -dest[sizeof(dest) - 1] = '\0'; -``` - -### Exception Swallowing -```python -# BAD: Silent failure -try: - do_something() -except Exception: - pass - -# GOOD: Log the error -try: - do_something() -except SpecificError as e: - logger.error("Operation failed: %s", e) - raise -``` - -### Path Traversal -```python -# BAD: Trusts user input -path = os.path.join(base_dir, user_input) - -# GOOD: Validate path -path = os.path.join(base_dir, user_input) -if not os.path.realpath(path).startswith(os.path.realpath(base_dir)): - raise ValueError("Invalid path") -``` - ---- - -## Technical Debt Indicators - -Watch for these patterns that indicate growing technical debt: - -1. **Copy-pasted code** - Should be refactored to shared function -2. **Magic numbers** - Should be named constants -3. **TODO/FIXME comments** - Should have associated issues -4. **Disabled tests** - Should be fixed or removed -5. **Suppressed warnings** - Should be investigated -6. **Platform-specific #ifdefs proliferating** - Consider abstraction layer -7. **Growing function length** - Should be split -8. **Deep nesting** - Should be flattened -9. **Tight coupling** - Should use interfaces/callbacks -10. **Missing error handling** - Should be added - ---- - -## File Reference - -| Component | Location | -|-----------|----------| -| PLC Runtime Core | `core/src/plc_app/` | -| REST API Server | `webserver/` | -| Plugin System | `core/src/drivers/` | -| Build Configuration | `CMakeLists.txt` | -| Code Style (C) | `.clang-format` | -| Code Style (Python) | `pyproject.toml` | -| Pre-commit Config | `.pre-commit-config.yaml` | -| Tests | `tests/pytest/` | -| Documentation | `docs/` | -| CI/CD Workflows | `.github/workflows/` | - ---- - -## Post-Review Actions - -After completing the review, always communicate findings directly on the PR. - -### Step 1: Create Review Document - -Save detailed review to `docs/pr-reviews/PR__REVIEW.md`: - -```markdown -# PR # Review: - -**Reviewer:** -**Date:** -**Author:** @ - -## Summary - - -## Quick Checklist -| Check | Status | Notes | -|-------|--------|-------| -| Pre-commit hooks pass | :white_check_mark: / :x: | ... | -... - -## Issues Found -### Critical -### Major -### Minor - -## Final Assessment -:white_check_mark: **APPROVE** / :x: **REQUEST CHANGES** -``` - -### Step 2: Post Summary Comment on PR - -Add an overall review comment: - -```bash -gh pr review --comment --body "## PR Review Summary - - - -**Overall:** :white_check_mark: **APPROVED** / :x: **CHANGES REQUESTED** - -See full review: \`docs/pr-reviews/PR__REVIEW.md\`" -``` - -### Step 3: Post Issues as PR Comment - -**Always** add a separate comment listing all issues found with file locations and suggestions: - -```bash -gh pr comment --body "### Issues Found () - ---- - -**1. ** -- **File:** \`path/to/file.py:\` -- **Issue:** -- **Suggestion:** -\`\`\`python - -\`\`\` - ---- - -**2. ** -... - ---- - -Full review: \`docs/pr-reviews/PR__REVIEW.md\`" -``` - -### Comment Format Guidelines - -1. **Be specific**: Always include file path and line number -2. **Be constructive**: Provide code suggestions when possible -3. **Categorize severity**: Critical, Major, Minor, or Suggestion -4. **Indicate blocking status**: Clearly state if issues block merge -5. **Reference documentation**: Link to the full review document - -### Example Comment Structure - -```markdown -### Minor Issues Found (Non-blocking) - -These are suggestions for future improvement - not blocking merge. - ---- - -**1. Duplicate constant definition** -- **File:** `core/src/module.py:22` -- **Issue:** `CONSTANT` is also defined in `other_module.py:54` -- **Suggestion:** Import from a single location: -\`\`\`python -from .other_module import CONSTANT -\`\`\` - ---- - -**2. Type hint could be more precise** -- **File:** `core/src/utils.py:199` -- **Function:** `some_function` -- **Suggestion:** Use `tuple[int, int]` instead of `tuple`: -\`\`\`python -def some_function(arg: int) -> tuple[int, int]: -\`\`\` - ---- - -Full review: `docs/pr-reviews/PR_123_REVIEW.md` -``` - -### Why Post Comments on PR? - -1. **Visibility**: Authors see feedback immediately in GitHub notifications -2. **Traceability**: Comments are linked to the PR permanently -3. **Discussion**: Authors can reply and discuss specific issues -4. **History**: Future reviewers can see past feedback patterns -5. **Accountability**: Clear record of what was reviewed and approved diff --git a/install.sh b/install.sh index 1e35a545..ef5f6459 100755 --- a/install.sh +++ b/install.sh @@ -140,15 +140,9 @@ EOF # Ensure we're in the project directory cd "$OPENPLC_DIR" -# Dispatch: Docker install by default, source build behind --native (RTOP-283). -# -# The Docker path installs no toolchain and compiles nothing, which is what -# makes a version change from the editor possible at all -- and what removes -# the failure this ticket exists to fix, where a half-finished source rebuild -# leaves a device with no build/ and no way in. -# -# --native keeps today's behaviour verbatim for MSYS2 and for targets that -# cannot host a container engine. It is a supported path, not a deprecated one. +# Default install mode is Docker; --native triggers the source build. +# Docker installs no toolchain and compiles nothing. --native stays a +# supported path for MSYS2 and container-less targets. INSTALL_MODE="docker" declare -a DOCKER_INSTALL_ARGS=() for arg in "$@"; do @@ -242,15 +236,9 @@ install_cmake() { echo "CMake $(cmake --version | head -1) installed" } -# `ccache` is added to every package set below. The runtime's -# scripts/Makefile.strucpp picks it up automatically when present and -# uses it to cache compiled .o files keyed by a hash of the -# preprocessed source + compile flags. The editor uploads the full -# project on every build, but ccache compares CONTENT (not file -# mtime), so unchanged TUs hit the cache and skip recompilation -# entirely. Single-POU edits drop incremental rebuilds from minutes -# to a few seconds. Without ccache the runtime still builds — just -# without the per-file reuse. +# `ccache` is installed by every package set. Makefile.strucpp picks +# it up automatically and caches .o files by content hash so repeated +# editor uploads reuse unchanged TUs. Optional: runtime builds without it. # For apt-based distros (Debian, Ubuntu, Linux Mint, Pop!_OS, elementary OS, Zorin, MX Linux, etc.) install_deps_apt() { @@ -328,10 +316,9 @@ install_deps_apk() { # For MSYS2 on Windows install_deps_msys2() { echo "Installing dependencies via pacman (MSYS2)..." - # Note: python-cryptography is installed via pacman because pip cannot build - # Rust-based packages on MSYS2/Cygwin. - # Plugin venvs use --system-site-packages to access these pre-built packages. - # bcrypt is skipped on MSYS2 - the OPC-UA plugin uses PBKDF2 fallback (Python stdlib). + # python-cryptography via pacman: pip cannot build Rust packages on + # MSYS2/Cygwin. Plugin venvs use --system-site-packages. bcrypt is + # skipped (OPC-UA plugin falls back to PBKDF2). local pkgs=( base-devel gcc diff --git a/scripts/compile.sh b/scripts/compile.sh index 7f17c354..20a515e1 100755 --- a/scripts/compile.sh +++ b/scripts/compile.sh @@ -2,25 +2,9 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® -# -# compile.sh — build the user PLC program into core/build/new_libplc.so -# -# This script is the entry point the runtime's webserver calls after -# extracting an upload into core/generated/. The actual build rules -# live in scripts/Makefile.strucpp — invoking `make` lets us: -# -# - run per-file compilation in parallel via `-j$(nproc)` (Pi 4 = 4×, -# workstations more); -# - reuse cached .o files automatically when ccache is installed, -# so incremental rebuilds (one POU edited) drop from minutes to -# a few seconds; -# - adapt to whatever .cpp set STruC++'s codegen split emitted — -# the Makefile uses `wildcard $(GENERATED_DIR)/*.cpp` rather than -# a hard-coded list. -# -# The MatIEC-era files (Config0.c, Res0.c, debug.c, glueVars.c) are -# rejected with a clear error so stale uploads fail loudly instead of -# silently building against the old pipeline. +# compile.sh builds the user PLC program into core/build/new_libplc.so. +# Entry point from the webserver after it extracts core/generated/. +# MatIEC-era files are rejected explicitly so stale uploads fail loudly. set -euo pipefail @@ -56,46 +40,16 @@ check_required_files() { check_required_files -# Build the program — actual rules live in scripts/Makefile.strucpp. -# -# Pick the build parallelism (`-j`) as min(CPU cap, memory cap). Both -# are real constraints on the Pi-class targets we ship to, and either -# bound alone has caused field outages. -# -# CPU cap = nproc - 1. On a 4-core Pi 4 `-j$(nproc)` saturates every -# core with g++ and starves the webserver / runtime monitor of CPU -# during the compile; combined with the Pi's slow SD-card swap, that -# made port-8443 RST new connections for 60+ seconds while the compile -# thrashed. Reserving one core for the Flask webserver, the runtime -# monitor thread, and any plugins keeps the device responsive -# throughout — only ~25 % slower per build on a Pi 4. -# -# Memory cap = total RAM (rounded to nearest GB) — one parallel job -# per gigabyte. Each cc1plus invocation on OpenPLC-generated TUs -# peaks at ~500–700 MB on a Pi 4. With `-j3` on a 2 GB device, -# three concurrent cc1pluses exhaust RAM + swap and the system enters -# a swap-thrash deadlock where no compile process makes progress — -# requires a physical reboot to recover on a headless target. -# Capping at one job per GB gives ~500 MB per cc1plus + headroom for -# the kernel, the webserver, the PLC core, and the plugins. The -# `+512` before dividing rounds MemTotal to the nearest GB so a 2 GB -# Pi (which reports ~1.8 GiB usable after kernel reserves) doesn't -# get wrongly demoted to -j1. -# -# Floor at 1 so single-core or sub-GB targets don't end up with `-j0` -# (which means "unlimited" in GNU make, i.e. fork-bomb). -# -# Worked examples: -# Pi 4 2 GB (nproc=4): cpu=3, mem=2 → -j2 -# Pi 4 4 GB (nproc=4): cpu=3, mem=4 → -j3 -# Pi 4 8 GB (nproc=4): cpu=3, mem=8 → -j3 -# 1 GB / 1 core VM: cpu=1, mem=1 → -j1 -# Workstation 8c/16GB: cpu=7, mem=16 → -j7 +# Build parallelism = min(nproc-1, RAM_GB). The CPU bound keeps the +# webserver responsive; the memory bound avoids swap-thrash from +# cc1plus peaks (~500-700 MB each on Pi-class targets). Floor at 1. CPU_JOBS=$(nproc) [ "$CPU_JOBS" -gt 1 ] && CPU_JOBS=$((CPU_JOBS - 1)) MEM_KB=$(awk '/^MemTotal:/{print $2}' /proc/meminfo) MEM_MB=$((MEM_KB / 1024)) +# Round to nearest GB so a 2 GB Pi (~1.8 GiB) does not demote to -j1. MEM_JOBS=$(( (MEM_MB + 512) / 1024 )) +# Floor at 1: -j0 in GNU make means unlimited. [ "$MEM_JOBS" -lt 1 ] && MEM_JOBS=1 if [ "$CPU_JOBS" -lt "$MEM_JOBS" ]; then JOBS=$CPU_JOBS @@ -104,26 +58,12 @@ else fi make -j"$JOBS" -f scripts/Makefile.strucpp -# ----------------------------------------------------------------------- -# Compile VPP plugin if source is present in the uploaded project -# -# The editor ships an optional vpp_plugin/ subtree alongside the IEC -# program when the project includes a VPP package. The plugin builds -# into BUILD_PATH (next to new_libplc.so) so the runtime's plugin -# loader picks it up under the same lookup rules as built-ins. -# -# checksum.sha256 is a RECOMPILATION CACHE KEY: the editor writes it over the -# files it copied, it travels inside the upload, and it is only ever compared -# against a copy of itself saved by a previous build -- to decide whether the -# plugin source changed since the last compile. -# ----------------------------------------------------------------------- +# Build the optional VPP plugin subtree. checksum.sha256 is the +# recompilation cache key; the editor writes it, this build compares. VPP_PLUGIN_DIR="$GENERATED_DIR/vpp_plugin" VPP_CHECKSUM_FILE="$VPP_PLUGIN_DIR/checksum.sha256" -# VPP outputs land in a dedicated subdir of BUILD_PATH so the cleanup -# glob below can scope itself to VPP-only artefacts. If a future built-in -# plugin ships as a .so dropped into BUILD_PATH directly, the old -# "$BUILD_PATH/lib*_plugin.so" glob would have rm'd it on every upload -# without a vpp_plugin subtree present. +# VPP outputs in a dedicated subdir so cleanup can scope to VPP-only +# artefacts without touching other plugins dropped into BUILD_PATH. VPP_OUTPUT_DIR="$BUILD_PATH/vpp" VPP_CACHED_CHECKSUM="$VPP_OUTPUT_DIR/checksum.sha256" # Seal the loader checks before dlopen (core/src/drivers/vpp_plugin_seal.c). @@ -164,12 +104,9 @@ if [ -d "$VPP_PLUGIN_DIR" ] && [ -f "$VPP_PLUGIN_DIR/Makefile" ]; then if [ -f "$VPP_CHECKSUM_FILE" ] && [ -f "$VPP_CACHED_CHECKSUM" ]; then if diff -q "$VPP_CHECKSUM_FILE" "$VPP_CACHED_CHECKSUM" > /dev/null 2>&1; then if ls "$VPP_OUTPUT_DIR"/lib*_plugin.so 1>/dev/null 2>&1; then - # A cache hit only stands when the SEAL vouches for the - # objects on disk (review 2026-08-20, R3): re-blessing - # whatever sits in build/vpp/ converted a detected tamper - # into a permanent pass on the next re-upload, and an - # upgraded runtime with no seal must REBUILD from the - # just-extracted tree, never bless unknown bytes. + # Cache hit only stands when the SEAL vouches for the + # on-disk objects; otherwise rebuild from the uploaded + # tree so unknown bytes are never blessed. if vpp_object_seal_matches; then echo "[INFO] VPP plugin source unchanged (checksum match), skipping recompilation" NEEDS_COMPILE=0 @@ -197,12 +134,9 @@ if [ -d "$VPP_PLUGIN_DIR" ] && [ -f "$VPP_PLUGIN_DIR/Makefile" ]; then echo "[INFO] VPP plugin compiled successfully" fi - # Record the sha256 of every .so this build produced, so the - # plugin loader can refuse an object swapped in AFTER the compile - # (core/src/drivers/vpp_plugin_seal.c, checked immediately before dlopen). - # ONLY when this run compiled (review 2026-08-20, R3): sealing on the - # cache-hit path blessed whatever bytes sat in build/vpp/; the upgrade- - # without-seal case is served by the forced recompile above. + # Seal every .so this build produced so the loader can refuse an + # object swapped in after compile. Only on compile paths: sealing a + # cache-hit would bless whatever bytes already sat in build/vpp/. if [ "$NEEDS_COMPILE" -eq 1 ]; then : > "$VPP_OBJECT_SEAL" for so in "$VPP_OUTPUT_DIR"/lib*_plugin.so; do diff --git a/scripts/install-docker.sh b/scripts/install-docker.sh index 842ddec5..499b7f4d 100755 --- a/scripts/install-docker.sh +++ b/scripts/install-docker.sh @@ -2,23 +2,9 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® -# OpenPLC Runtime installer -- container edition (RTOP-283). -# -# Two ways in, one script: -# -# curl -fsSL https://runtime.getedge.me | sudo bash no checkout needed -# sudo ./install.sh from a clone; execs this -# -# Compiles nothing: it ensures a container engine, writes the bootloader's -# spec, and starts the bootloader, which pulls the runtime image and brings it -# up. Docker is the only dependency this path adds, and nothing of ours goes -# into systemd -- Docker's restart policy starts the bootloader at boot. -# -# --native keeps the source build, for MSYS2 and targets that cannot host a -# container engine. It needs the repository on disk, so the piped one-liner -# cannot reach it. -# -# --uninstall removes what this script created and puts back what it displaced. +# OpenPLC Runtime container installer. Writes the bootloader spec and +# starts the bootloader (which pulls the runtime image). --native keeps +# the source build. --uninstall reverses the install. set -euo pipefail RED='\033[0;31m'; GREEN='\033[0;32m'; YELLOW='\033[1;33m'; BLUE='\033[0;34m'; NC='\033[0m' @@ -34,12 +20,8 @@ BOOTLOADER_REPOSITORY="${BOOTLOADER_REPOSITORY:-ghcr.io/autonomy-logic/openplc-r RUNTIME_VERSION="${RUNTIME_VERSION:-}" BOOTLOADER_VERSION="${BOOTLOADER_VERSION:-latest}" -# Two directories, deliberately separate. -# -# The runtime's holds what a version change must preserve: .env, restapi.db, -# retain.bin, vpp/ licences, the stored project. The bootloader's holds the -# container spec, including this board's device mounts -- separate because -# "erase all data" wipes the runtime's directory. +# Runtime data (version-preserving) and bootloader state (survives data +# wipe) are deliberately in separate directories. RUNTIME_DATA_DIR="${RUNTIME_DATA_DIR:-/var/lib/openplc-runtime}" BOOTLOADER_STATE_DIR="${BOOTLOADER_STATE_DIR:-/var/lib/openplc-bootloader}" @@ -56,10 +38,8 @@ declare -a EXTRA_ENV=() # which accumulates across installs and is what --uninstall reads. declare -a DISPLACED_THIS_RUN=() -# Where this script lives, for the self-elevation path below. -# Where the piped one-liner fetches this script. runtime.getedge.me serves it -# verbatim from the release branch; the raw GitHub URL works identically, and -# OPENPLC_INSTALLER_URL overrides both, for a fork or an internal mirror. +# Source for the piped one-liner. OPENPLC_INSTALLER_URL overrides for +# forks or internal mirrors. INSTALLER_URL="${OPENPLC_INSTALLER_URL:-https://runtime.getedge.me}" INSTALLER_URL_FALLBACK="https://raw.githubusercontent.com/Autonomy-Logic/openplc-runtime/main/scripts/install-docker.sh" @@ -69,10 +49,8 @@ KEEP_IMAGES=false PURGE_DATA=false REPO_ROOT="" -# systemd units that run a pre-container OpenPLC. Two of them are real and in -# the field: openplc.service is the v3 runtime, openplc-runtime.service is a v4 -# source install. Both bind 8443, so leaving one running means the container -# starts and then fails to serve, with nothing obviously wrong on either side. +# systemd units that bind 8443 from a pre-container install; left +# running they block the container from serving. LEGACY_UNITS=(openplc-runtime.service openplc.service openplc_v3.service openplc-v3.service) # What we stopped, so --uninstall can put it back. Kept in the bootloader's @@ -193,10 +171,8 @@ detect_engine() { install_engine() { log_info "Docker not found; installing it" - # Docker's own convenience script rather than distro packages: it covers - # every distro this runtime targets and always installs a version new - # enough for the API the bootloader uses. Distro packages vary wildly -- - # Debian bookworm's docker.io is old enough to matter. + # Docker's own installer, not distro packages: covers every distro + # and always installs a version new enough for the bootloader's API. if ! command -v curl >/dev/null 2>&1; then log_error "curl is required to install Docker. Install curl, or install" log_error "Docker yourself and re-run this script." @@ -267,13 +243,8 @@ unit_exists() { [ -n "$(systemctl list-unit-files --no-legend "$1" 2>/dev/null)" ] } -# stop_legacy_runtimes clears the way for the container. -# -# A source or v3 install binds 8443 from systemd. Left running, the runtime -# container cannot bind it, while the editor still reaches the old one on that -# port. -# -# Units are stopped and disabled, never deleted: uninstall puts them back. +# stop_legacy_runtimes frees 8443 by stopping any systemd units from a +# source or v3 install. Disabled (not deleted) so --uninstall restores. stop_legacy_runtimes() { have_systemd || return 0 @@ -424,14 +395,9 @@ do_uninstall() { log_warning "Docker is not available; skipping container and image removal." fi - # Data before restoring the old runtime: restarting it first would have it - # recreate this directory, and the delete would then take files the restored - # runtime had written. - # - # Kept by default, because it is not exclusively ours: a native install reads - # and writes the same path (webserver/config.py resolves it on native Linux), - # so deleting it would destroy the data of the runtime this uninstall - # restores. + # Purge data BEFORE restoring the old runtime, so it does not + # recreate the directory first. Kept by default because a native + # install shares the same path. if [ "$PURGE_DATA" = true ] && [ -f "$(disabled_units_file)" ]; then log_warning "Not deleting $RUNTIME_DATA_DIR: a systemd runtime is being" log_warning "restored and shares that directory. Remove it by hand if you" @@ -484,11 +450,8 @@ resolve_runtime_version() { fi } -# resolve_latest_to_a_version turns "latest" into the version it points at. -# -# Recording "latest" would leave the device following the tag on every -# reconcile. The version is read from the image (RUNTIME_VERSION, baked in by -# the release build), so this needs nothing beyond the registry. +# resolve_latest_to_a_version pins "latest" to the baked RUNTIME_VERSION +# so the device does not follow the moving tag on every reconcile. resolve_latest_to_a_version() { [ "$RUNTIME_VERSION" = latest ] || return 0 @@ -498,10 +461,8 @@ resolve_latest_to_a_version() { "$image" 2>/dev/null || true)" if printf '%s' "$baked" | grep -qE '^v[0-9]+\.[0-9]+\.[0-9]+'; then - # The daemon holds this image only as ":latest". Pin the spec to a name - # it does not have and the bootloader sees the container as stale, then - # has to reach the registry to resolve the new tag -- inside the window - # where the device has no runtime, and impossibly if it is offline. + # Tag the local :latest with the baked version so the bootloader + # can resolve it offline (the daemon only holds :latest otherwise). if ! docker tag "$image" "$RUNTIME_REPOSITORY:$baked" 2>/dev/null; then log_warning "Could not tag $image as $baked; the device will follow the 'latest' tag." return 0 @@ -561,15 +522,9 @@ write_spec() { # --- bootloader ---------------------------------------------------------- -# Both images are pulled before anything on the device is disturbed. With the -# pull inside start_bootloader, a device that could not reach the registry had -# already had its systemd runtime disabled by the time the download failed, -# leaving it with no PLC. -# pull_image fetches one image, deliberately not quiet: the runtime image is -# a few hundred megabytes, and with no output the installer looks hung. -# -# Returns 0 when the image is available afterwards, by pull or because a copy -# was already here (air-gapped, or side-loaded with `docker load`). +# pull_image fetches an image, returning 0 when it is available +# afterwards (pulled or already local). Called BEFORE anything on the +# device is disturbed, so a failed pull leaves it unchanged. pull_image() { local image="$1" what="$2" @@ -608,7 +563,7 @@ start_bootloader() { # The runtime data directory is read-only here: the bootloader authenticates # against the runtime's accounts and must not modify one. - # --uts=host so recovery-mode discovery names the DEVICE (RTOP-292). + # --uts=host so recovery-mode discovery names the DEVICE, not the container. docker run -d \ --name "$BOOTLOADER_CONTAINER" \ --restart always \ diff --git a/scripts/manage_plugin_venvs.sh b/scripts/manage_plugin_venvs.sh index 6acbdb9c..a4a236fc 100755 --- a/scripts/manage_plugin_venvs.sh +++ b/scripts/manage_plugin_venvs.sh @@ -59,19 +59,9 @@ check_python() { log_info "Using Python version: $python_version" } -# Install a plugin's requirements into its venv. -# -# On MSYS2/Cygwin the Python interpreter is the cygwin build -# (SOABI cpython-3xx-x86_64-cygwin). Rust-backed wheels (cryptography) cannot -# be compiled there — maturin aborts with "Unsupported platform: x86_64-cygwin". -# install.sh therefore installs python-cryptography via pacman and the venv is -# created with --system-site-packages so the pre-built copy is importable. -# -# That alone is not enough: even though the system cryptography satisfies the -# direct requirement, pip's resolver (dragged along by pyopenssl) still selects -# the newest cryptography release and tries to build its sdist from source, -# which fails. Pin cryptography to the exact version already present in the -# system site-packages so pip reuses the pacman build instead of compiling. +# Install a plugin's requirements. On MSYS2 the cryptography wheel +# cannot be built, so pin it to the pacman-installed version and reuse +# the system site-packages copy. pip_install_requirements() { local venv_path="$1" local requirements_file="$2" diff --git a/scripts/run-pytest.sh b/scripts/run-pytest.sh index 96a23794..2e8683f9 100755 --- a/scripts/run-pytest.sh +++ b/scripts/run-pytest.sh @@ -38,11 +38,8 @@ echo "Installing pytest and local package..." pip install pytest pip install -e "$PROJECT_ROOT" -# Every plugin's requirements, not just modbus_master's. -# -# The plugin suites import their driver modules at collection time, so a -# missing pymodbus or asyncua is a collection error that stops the whole run -# instead of skipping those files -- which is exactly what this script did. +# Every plugin's requirements: suites import drivers at collection +# time, so a missing dep fails the whole run instead of skipping. for req in "$PROJECT_ROOT"/core/src/drivers/plugins/python/*/requirements.txt; do echo "Installing $(basename "$(dirname "$req")") requirements..." pip install -r "$req" diff --git a/tests/host/run.sh b/tests/host/run.sh index 28c9ded4..9da94a8b 100755 --- a/tests/host/run.sh +++ b/tests/host/run.sh @@ -2,18 +2,9 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® -# -# Host tests: plain executables, no framework, no device. -# -# For runtime C++ that Ceedling cannot reach (it is configured for C, and these -# translation units use std::thread / std::mutex) and that does not need the -# lifecycle harness's real `plc_main`. One command, runs anywhere with a C++17 -# compiler — including macOS, which the lifecycle suite cannot do. -# -# ./tests/host/run.sh -# -# Add a test by dropping a `test_*.cpp` here that compiles against the sources -# it needs; list it in TESTS below with those sources. +# Host tests: plain C++17 executables, no framework, no device. For +# runtime code Ceedling cannot reach (uses std::thread/mutex) and that +# needs no plc_main. Add a test by appending it to TESTS below. set -euo pipefail diff --git a/tests/host/test_plc_retain_file_store.cpp b/tests/host/test_plc_retain_file_store.cpp index cd4b3eb9..3bdef336 100644 --- a/tests/host/test_plc_retain_file_store.cpp +++ b/tests/host/test_plc_retain_file_store.cpp @@ -1,48 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * Host test for the built-in retain file store's identity handling. - * - * WHY THIS EXISTS SEPARATELY FROM THE PYTEST SUITE - * ------------------------------------------------ - * `tests/pytest/plugins/test_apply_retain_conf.py` covers the webserver half: - * which retain.conf gets installed, and when the device's copy is removed. It - * says nothing about the half that decides whether stored BYTES still belong to - * the running program, which is the behaviour the whole design rests on: - * - * - the `[32-byte program md5][payload]` on-disk layout. No length field: a - * file has a size, so the payload length is recovered by reading to EOF. - * (Baremetal's flash driver DOES carry an explicit length, because its - * region is fixed-size and trailing erased bytes read as 0xFF — the two - * formats are deliberately not the same, and this test pins this one.); - * - discarding the payload when the identity does not match; - * - treating a file too short to carry the header as unattributable; - * - holding the identity from load() so the next save() can label its bytes. - * - * Until this file existed, that path's only evidence was a by-hand run on an - * SLM-RP4 recorded in a PR body. The case it proves — a program's values are - * refused for a DIFFERENT program even when the retain layout is identical — is - * exactly the one a layout hash cannot catch, so it is worth being able to - * re-run without hardware. - * - * WHY NOT CEEDLING, AND WHY NOT THE LIFECYCLE HARNESS - * -------------------------------------------------- - * Ceedling is configured for C (`:test_file_prefix: test_`, `.c` sources) and - * this store is C++ with `std::thread`/`std::mutex`, so it is not in that - * runner's reach. `tests/lifecycle/` could reach it, but it boots a real - * `plc_main` against a compiled PLC program and needs Linux, Docker and an - * editor payload — far more machinery than file-header logic warrants, and it - * would not run on a developer's machine. - * - * This is a plain executable instead: no framework, no fixtures, one command. - * - * c++ -std=c++17 -I core/src/plc_app -I core/src \ - * tests/host/test_plc_retain_file_store.cpp \ - * core/src/plc_app/plc_retain_file_store.cpp -o /tmp/t && /tmp/t - * - * See tests/host/run.sh, which does exactly that. - */ +/* Host test for the retain file store's identity handling. Pins the + * [32-byte md5][payload] layout, discard-on-mismatch, short files + * unattributable, identity held from load() to the next save(). */ #include #include @@ -57,12 +18,8 @@ #include "plc_retain.h" #include "plc_retain_file_store.h" -// --------------------------------------------------------------------------- -// The store logs through the runtime's logger, which is not worth linking here. -// Captured rather than discarded: two cases below assert that the operator is -// TOLD storage was cleared, because a silent discard of retained values is the -// failure mode this design is most likely to be blamed for later. -// --------------------------------------------------------------------------- +// Capture log calls so test cases can assert the operator is told +// storage was cleared. static std::string g_log; extern "C" void log_info(const char *fmt, ...) @@ -87,9 +44,6 @@ extern "C" void log_warn(const char *fmt, ...) g_log += '\n'; } -// --------------------------------------------------------------------------- -// Harness -// --------------------------------------------------------------------------- static int g_failures = 0; static const char *g_case = ""; @@ -106,10 +60,8 @@ static std::string g_dir; static std::string g_store_path; static std::string g_conf_path; -/* Two identities that differ, both the right length. Deliberately NOT - * NUL-terminated in the calls below — the contract says 32 characters and the - * length travels separately, and a driver reaching for strlen would pass a test - * that used terminated strings and fail in production. */ +/* 32-byte identities, deliberately not NUL-terminated: the contract + * passes length separately, so a strlen-based driver would fail. */ static const char MD5_A[PLC_RETAIN_PROGRAM_ID_LEN] = {'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', 'a', diff --git a/tests/integration/entrypoint.sh b/tests/integration/entrypoint.sh index 1c2fc662..613f7cb8 100644 --- a/tests/integration/entrypoint.sh +++ b/tests/integration/entrypoint.sh @@ -2,12 +2,8 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® -# Start the inner Docker daemon, then hand over to the command. -# -# The daemon has to be up before anything else runs, and "up" means the socket -# answers -- not merely that dockerd was spawned. Racing it is the classic way -# an integration harness fails intermittently and gets blamed on the code under -# test. +# Start the inner Docker daemon (wait for the socket to answer), then +# hand over to the command. set -euo pipefail log() { printf '[testhost] %s\n' "$*" >&2; } diff --git a/tests/integration/harness.sh b/tests/integration/harness.sh index 876b287e..5ff0c8fa 100755 --- a/tests/integration/harness.sh +++ b/tests/integration/harness.sh @@ -2,31 +2,16 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® -# Integration harness for the RTOP-283 bootloader. -# -# Runs a Debian container with its own Docker daemon (see Dockerfile.testhost), -# stands up a registry inside it, and seeds that registry with runtime images. -# The bootloader then does real pulls over a real registry, so the update path -# -- including progress streaming and layer reuse -- is exercised rather than -# stubbed. -# -# What this harness cannot cover, and what the device round is for: hardware. -# There is no /dev/spidev6.0 or /dev/gpiochip0 here, so VPP plugin behaviour -# and real SCHED_FIFO latency must be validated on an SLM-RP4. -# -# Usage: -# ./harness.sh up # build and start the test host -# ./harness.sh seed # load images and fill the inner registry -# ./harness.sh test [filter] # run the suite against it -# ./harness.sh shell # interactive shell on the test host -# ./harness.sh down # tear everything down +# Bootloader integration harness. Runs a Debian container with its own +# Docker daemon and inner registry so the bootloader does real pulls. +# Usage: ./harness.sh {up|seed|test [filter]|shell|down} set -euo pipefail HOST_CONTAINER=openplc-testhost HOST_IMAGE=openplc-testhost:latest -# The device's hostname. Set explicitly: on Docker's container-id default a -# correct reply and the RTOP-292 bug both look like hex. +# The device's hostname. Set explicitly: on Docker's container-id default +# a correct reply and a UTS-namespace leak both look like hex. DEVICE_HOSTNAME="${DEVICE_HOSTNAME:-slm-rp4-testhost}" DOCKER_VOLUME=openplc-testhost-docker REGISTRY=localhost:5000 @@ -35,17 +20,8 @@ REGISTRY=localhost:5000 STUB_REPO="$REGISTRY/openplc-stub" REAL_REPO="$REGISTRY/openplc-runtime" -# Base for the end-to-end case against a real runtime. -# -# A published image by default, so this harness reproduces anywhere. It used to -# default to a tag that existed only on the author's machine, which meant the -# reported pass count could not be reproduced by anyone else -- and quietly -# meant the repository's own Dockerfile was never exercised. -# -# REAL_BASE=build builds from the repository Dockerfile instead. Slower by -# minutes (it is a full source install), and the only setting that covers the -# Dockerfile itself -- which is where `./install.sh` silently switching to the -# container path broke the release build. +# Default to a published image so the harness reproduces anywhere. +# REAL_BASE=build runs the full source install from the Dockerfile. REAL_BASE="${REAL_BASE:-ghcr.io/autonomy-logic/openplc-runtime:v4.2.3}" # The tag the inner registry serves. Exported so the suite reads it once. @@ -74,13 +50,9 @@ cmd_up() { docker volume create "$DOCKER_VOLUME" >/dev/null log "starting the test host" - # Privileged because it runs a Docker daemon. The repo is mounted - # read-only so image builds inside can use it as a build context without - # any risk of a test writing to the working tree. - # The runtime and bootloader run with --network host INSIDE this - # container, so publishing here is what lets a browser on the developer's - # machine reach them -- which is how the editor and web UI get tested - # against a real device without one on the desk. + # Privileged for the inner Docker daemon. Repo mounted read-only. + # Ports published so a browser on the dev machine can reach the + # host-networked runtime and bootloader inside. docker run -d --name "$HOST_CONTAINER" --privileged \ --hostname "$DEVICE_HOSTNAME" \ -p 8443:8443 -p 8445:8445 \ @@ -116,14 +88,9 @@ transfer() { docker save "$image" | inner_stdin docker load >/dev/null } -# build_into_host builds an image from this repo and loads it into the inner -# daemon, ALWAYS fresh -- these are the artefacts under test, so a stale copy -# would quietly test the previous commit. -# -# buildx with `--output type=docker` rather than `docker build` + `docker save`: -# Docker 29 exports a buildx-built image as an OCI layout, and the inner -# daemon rejects that with "does not contain a manifest.json". This output type -# writes the legacy docker-archive both daemons agree on. +# build_into_host builds an image from this repo and loads it into the +# inner daemon, always fresh. buildx with --output type=docker emits +# the legacy archive both daemons accept. build_into_host() { local image="$1" context="$2" dockerfile="$3" shift 3 diff --git a/tests/integration/stubruntime/main.go b/tests/integration/stubruntime/main.go index 8242ea43..a90c2c5c 100644 --- a/tests/integration/stubruntime/main.go +++ b/tests/integration/stubruntime/main.go @@ -1,20 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -// Command stubruntime stands in for the OpenPLC runtime in integration tests. -// -// It serves the two endpoints the bootloader actually depends on -- an -// unauthenticated /api/version and a healthcheck -- and nothing else. The -// point is not to emulate the runtime; it is to make the runtime's FAILURE -// modes reproducible on demand, which the real image cannot be asked to do. -// A real runtime cannot be told "exit 1 during start-up" or "come up healthy -// then die three times", and those are exactly the paths where the -// bootloader's crash accounting and recovery transitions live. -// -// The real image is exercised separately in the same harness for the -// does-it-actually-come-up case, and hardware behaviour (SPI, GPIO, VPP -// plugins, real SCHED_FIFO latency) is validated on a device, which no -// container on a developer machine can stand in for. +// Command stubruntime stands in for the OpenPLC runtime in integration +// tests. Serves only /api/version and a healthcheck. Lets failure modes +// (exit 1 at start, die after N healthy scans) be reproduced on demand. package main import ( diff --git a/tests/integration/test_bootloader.py b/tests/integration/test_bootloader.py index a05a5bcb..d52688e7 100644 --- a/tests/integration/test_bootloader.py +++ b/tests/integration/test_bootloader.py @@ -2,7 +2,7 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® -"""End-to-end tests for the RTOP-283 bootloader, run inside the test host. +"""End-to-end tests for the openplc-bootloader, run inside the test host. These drive the real thing: a real Docker daemon, a real registry, real image pulls with real progress streaming, and the bootloader binary that ships. The @@ -47,11 +47,8 @@ DATA_DIR = "/var/lib/openplc-runtime" BOOTLOADER_URL = "https://127.0.0.1:8445" -# From the shared auth vector (bootloader/internal/runtimeauth/runtimeauth_test.go -# and tests/pytest/restapi/test_bootloader_auth_vector.py). Reusing it here -# means the data directory can be seeded with a genuine werkzeug hash without -# werkzeug being installed in the test host -- and it cross-checks the vector -# in a real integration setting rather than only in unit tests. +# Shared auth vector (also used by the Go and Python unit tests). Lets +# this harness seed a werkzeug hash without werkzeug installed. PEPPER = "a" * 64 JWT_SECRET = "b" * 64 USERNAME = "operator" @@ -200,8 +197,8 @@ def start_bootloader(extra_args: list[str] | None = None) -> None: args = [ "docker", "run", "-d", "--name", BOOTLOADER_NAME, "--restart", "always", "--network", "host", - # Mirrors install.sh, kept in step by hand. --uts=host is what makes - # the recovery-mode reply name the device (RTOP-292). + # Mirrors install.sh, kept in step by hand. --uts=host is what + # makes the recovery-mode reply name the device, not the container. "--uts=host", "-v", "/var/run/docker.sock:/var/run/docker.sock", "-v", f"{STATE_DIR}:{STATE_DIR}", @@ -211,13 +208,8 @@ def start_bootloader(extra_args: list[str] | None = None) -> None: args += extra_args or ["-log-level=debug"] sh(*args) - # Wait for THIS bootloader to answer before returning. - # - # Without it a case starts polling while nothing is listening on 8445 yet, - # and the failures read as "ConnectionRefused" or -- worse -- as progress - # belonging to a different case, because a poll can land on a bootloader - # that has not been replaced yet. Confirming a fresh, responsive process - # removes both by construction. + # Wait for THIS bootloader to answer before returning so polls do + # not land on a previous instance. def responsive() -> bool: state = container_state(BOOTLOADER_NAME) if not state.get("State", {}).get("Running"): @@ -425,8 +417,9 @@ def test_bootstrap_creates_and_supervises_the_runtime(): raise Failure("the runtime must be privileged for hardware parity") if host["NetworkMode"] != "host": raise Failure(f"want host networking, got {host['NetworkMode']}") - # RTOP-292: without this, discovery reports a container id. The flag only; - # the stub is FROM scratch, so what it SEES is checked on the real image. + # Without --uts=host, discovery reports a container id. Only the + # flag is checked here; what the runtime SEES is covered by the + # real-image test, since the stub is FROM scratch. if host["UTSMode"] != "host": raise Failure(f"want the host UTS namespace, got {host['UTSMode']!r}") if host["RestartPolicy"]["Name"] != "no": @@ -839,7 +832,7 @@ def _bake_pre_fix_runtime_container(image: str, *, running: bool) -> str: @case def test_a_runtime_container_with_a_private_uts_namespace_is_replaced(): - """The field-upgrade path for RTOP-292. + """Field-upgrade path for a runtime with a private UTS namespace. Both states ship: a normal pre-fix install leaves it RUNNING, an SLM-RP4 image leaves it STOPPED and never started, which skips the graceful stop @@ -882,8 +875,8 @@ def test_a_runtime_container_with_a_private_uts_namespace_is_replaced(): @case def test_recovery_mode_discovery_reports_the_device_hostname(): - """The bootloader answers discovery when the runtime is down, and that - reply has to name the board too (RTOP-292).""" + """The bootloader answers discovery when the runtime is down, and + that reply has to name the board, not the container.""" reset(version="v1.0.0", extra_env=["STUB_FAIL=exit"]) wait_for("recovery mode", lambda: bootloader_state() == "recovery", timeout=120) @@ -946,11 +939,9 @@ def test_the_real_runtime_image_comes_up_under_the_bootloader(): if not served.get("runtimeVersion"): raise Failure(f"the real runtime must report a version, got {served}") - # The data-directory bug, checked against the real runtime rather than a - # field the stub invents. config.py resolves the persistent directory by - # container detection, so without the env override the runtime writes a - # fresh .env inside the container and ignores the mounted one -- losing - # users, the stored project, retain data and licences on every swap. + # Verify OPENPLC_PERSISTENT_DATA_DIR is set: without it the runtime + # defaults the persistent dir inside the container and ignores the + # mount, discarding users and licences on every swap. env = container_state(RUNTIME_NAME)["Config"]["Env"] if f"OPENPLC_PERSISTENT_DATA_DIR={DATA_DIR}" not in env: raise Failure(f"the runtime was not pointed at the mounted data dir: {env}") @@ -964,8 +955,9 @@ def test_the_real_runtime_image_comes_up_under_the_bootloader(): if not os.path.exists(os.path.join(DATA_DIR, ".env")): raise Failure("the real runtime did not write .env into the mounted data dir") - # RTOP-292, at the syscall the responder actually calls. Asserted here - # rather than on the stub because this image has a userland to ask. + # Assert UTS passthrough at the syscall the responder actually + # calls. Done on the real image rather than the stub, which has + # no userland to ask. seen = sh("docker", "exec", RUNTIME_NAME, "hostname").strip() if seen != device_hostname(): raise Failure( diff --git a/tests/lifecycle/fakevpp_plugin.c b/tests/lifecycle/fakevpp_plugin.c index 579b94a7..0f4f543c 100644 --- a/tests/lifecycle/fakevpp_plugin.c +++ b/tests/lifecycle/fakevpp_plugin.c @@ -152,16 +152,9 @@ int init(void *args) return 0; } -/* The watcher lives between start_loop and stop_loop, and NOT a moment longer. - * - * init() is the tempting place for it -- a real mode switch has to be watched - * while the PLC is stopped too -- but the contract forbids threads there for a - * concrete reason: plugin_driver_update_config tears every slot down and - * re-dlopens it on each start, so a thread left running from init() ends up - * executing code that has been unmapped. That is a SIGSEGV, and this fixture - * earned one before being written this way. Everything the tests need still - * works, because the runtime stops plugins only AFTER joining the PLC thread: - * a flip during a stop is still seen. */ +/* Start the watcher here, not in init(): plugin_driver_update_config + * re-dlopens plugins per start, so a thread from init() runs code that + * has been unmapped (SIGSEGV). */ int start_loop(void *args) { (void)args; @@ -182,10 +175,8 @@ int stop_loop(void *args) { (void)args; - /* Optional slow teardown, BEFORE the watcher is joined, so the switch is - * still being watched while the stop is in flight. That is the only way to - * land a flip inside a stop transition on purpose: a stop is otherwise tens - * of milliseconds and there is nothing to aim at. */ + /* Optional slow teardown BEFORE joining the watcher, so a switch + * flip can land inside a stop transition. */ const long slow_ms = env_long("FAKEVPP_STOP_MS", 0); if (slow_ms > 0) { diff --git a/tests/lifecycle/harness.sh b/tests/lifecycle/harness.sh index ce4d34f3..45aeed60 100644 --- a/tests/lifecycle/harness.sh +++ b/tests/lifecycle/harness.sh @@ -39,11 +39,9 @@ ls build/libplc_*.so >/dev/null 2>&1 || { echo " program: $(ls build/libplc_*.so)" echo "### plugins.conf: the shipped set plus the fake VPP" -# The stock Python entries stay, disabled, because loading them is what -# initialises the interpreter -- and has_python_plugin && Py_IsInitialized() is -# the precondition for the Py_FinalizeEx() shutdown crash. A conf with only the -# native fixture in it quietly makes that whole class untestable. -# Native plugin lines whose .so is not built are dropped: they only add warnings. +# Keep the stock Python entries disabled; loading them initialises the +# interpreter, which is the precondition for the Py_FinalizeEx shutdown +# crash this suite covers. grep -v 'libs7comm_plugin\|libethercat_plugin' plugins_default.conf > plugins.conf # name,path,enabled,type,config_json,venv (type 1 = native) printf 'fakevpp,./build/plugins/libfakevpp_plugin.so,1,1,/tmp/fakevpp_config.json,\n' >> plugins.conf diff --git a/tests/pytest/compile/test_retain_capability_probe.py b/tests/pytest/compile/test_retain_capability_probe.py index 90903b2f..159ff699 100644 --- a/tests/pytest/compile/test_retain_capability_probe.py +++ b/tests/pytest/compile/test_retain_capability_probe.py @@ -54,11 +54,8 @@ "strucpp::debug::retain_layout_hash", ) -# --- header stubs ---------------------------------------------------------- -# -# Deliberately minimal. The probe includes exactly two headers, so these are all -# it can see, and hand-written stubs let the test state the SHAPE that matters -# without vendoring a copy of STruC++ into the runtime repo. +# Hand-written header stubs. The probe includes exactly two headers, so +# these are all it can see. _DEBUG_TABLE_COMMON = """ #pragma once @@ -296,16 +293,9 @@ def test_the_makefile_only_defines_the_gate_from_the_probe(): assert "$(CXX) $(SHIM_CXXFLAGS) -c $< -o $@" in makefile -# --------------------------------------------------------------------------- -# 4. The wiring: probe verdict -> compiler flag -# --------------------------------------------------------------------------- -# -# Everything above can pass while the build still does the wrong thing. It did: -# the first cut of the Makefile wrote the verdict with a line continuation, and -# GNU make counts the lone space that leaves behind as a NON-EMPTY $(if) -# condition -- so a legacy header set produced " " instead of "", the gate turned -# on for exactly the uploads it exists to protect, and the original error came -# straight back. Only asking make itself catches that class of bug. +# Asks `make` itself whether the probe verdict actually becomes the +# right compiler flag: $(if) treats a line-continuation space as truthy, +# which GNU make's own invocation is the only reliable detector of. _MAKE = shutil.which("make") or shutil.which("gmake") diff --git a/tests/pytest/modbus_slave/test_modbus_slave.py b/tests/pytest/modbus_slave/test_modbus_slave.py index 3d41116f..268c4ca1 100644 --- a/tests/pytest/modbus_slave/test_modbus_slave.py +++ b/tests/pytest/modbus_slave/test_modbus_slave.py @@ -45,10 +45,7 @@ def assert_block_zeroed(block, size): assert block.getValues(0, size) == [0] * size -# ----------------------------------------------------------------------- -# Fake SafeBufferAccess used to observe locking behavior. -# We patch simple_modbus.SafeBufferAccess to return this object inside blocks -# ----------------------------------------------------------------------- +# Fake SafeBufferAccess that observes locking behaviour. class ObservingSafeBufferAccess: """ Test double for SafeBufferAccess that matches the REAL method signatures used @@ -133,7 +130,6 @@ def read_int_output(self, index, thread_safe=True): return (int(self.args.analog_output[index]) & 0xFFFF, "Success") - # ----------------------------------------------------------------------- # Data Block tests (use ObservingSafeBufferAccess patched in) # ----------------------------------------------------------------------- diff --git a/tests/pytest/plugins/opcua/test_string_types.py b/tests/pytest/plugins/opcua/test_string_types.py index 9ae0e11e..a92fe7aa 100644 --- a/tests/pytest/plugins/opcua/test_string_types.py +++ b/tests/pytest/plugins/opcua/test_string_types.py @@ -125,10 +125,8 @@ def test_wstring_truncates_to_the_cap_in_units(self): assert len(enc) == 1 + DEBUG_STRING_CAP * 2 def test_wstring_drops_a_trailing_odd_byte(self): - # An odd length is malformed, and the encoder answers by dropping the - # trailing byte rather than refusing the value. Named for what it does: - # it used to be called "rejects_an_odd_byte_count" while asserting a - # successful one-unit encode, so the name argued against the assertion. + # Odd length is malformed; the encoder drops the trailing byte + # instead of refusing the value. enc = _encode_string("WSTRING", b"abc") assert enc[0] == 1 and len(enc) == 3 diff --git a/tests/pytest/plugins/test_vpp_anchor_cross_language.py b/tests/pytest/plugins/test_vpp_anchor_cross_language.py index fed24a3e..179d0348 100644 --- a/tests/pytest/plugins/test_vpp_anchor_cross_language.py +++ b/tests/pytest/plugins/test_vpp_anchor_cross_language.py @@ -106,10 +106,8 @@ def _find_packages_root(): def _read_source(rel_path: str) -> str: - # newline=None gives universal-newline translation; strip any stray \r on - # top of it, because a Windows checkout can hand us CRLF where the - # extraction patterns expect bare \n (tasks #50/#58). Normalizing line - # endings for matching does not change what the C does. + # Normalise newlines so a Windows CRLF checkout matches patterns + # that expect bare \n. path = os.path.join(_PACKAGES_ROOT, rel_path) with open(path, "r", encoding="utf-8", newline=None) as handle: return handle.read().replace("\r", "") @@ -164,15 +162,8 @@ def test_python_anchor_ceiling_matches_the_c_buffer(): ) -# --------------------------------------------------------------------------- -# 2. The behaviour, by executing the real C. Needs a compiler. -# -# license_platform.c ships its own test seam: LIC_LINUX_ANCHOR_PATH overrides -# the /proc/device-tree path at compile time (the packages host tests use the -# same seam). So the whole file compiles UNMODIFIED, pointed at a temp file, -# and a three-line main() prints what license_platform_anchor() returned -- -# no extraction, no transcription, the real translation unit end to end. -# --------------------------------------------------------------------------- +# Executes the real C via LIC_LINUX_ANCHOR_PATH, which overrides the +# /proc/device-tree path at compile time. _HARNESS_MAIN = """\ #include diff --git a/tests/pytest/plugins/test_vpp_license_debug.py b/tests/pytest/plugins/test_vpp_license_debug.py index fafdf797..84e5eb6a 100644 --- a/tests/pytest/plugins/test_vpp_license_debug.py +++ b/tests/pytest/plugins/test_vpp_license_debug.py @@ -27,14 +27,9 @@ def _hex(data: bytes) -> str: return " ".join(f"{b:02X}" for b in data) -# The real signed 98-byte license blob, copied verbatim from -# openplc-packages/license-core/test/license-golden-signed.json ("blobHex") -- -# the same vector license_core's host test and the editor/backend unit tests use -# (anchor 00b18ced -> deviceId 659a3520540f803625ddc34081e893d3, product -# 29a17c7c2486d355). Using it here means these tests assert the runtime against -# an artifact produced by ANOTHER implementation, not against bytes this file -# made up: if the magic or the crc32 range ever drifts on either side, this -# literal stops validating. +# Real 98-byte signed license blob from license-core's golden vector. +# If magic or crc32 range drifts on either side, this literal stops +# validating. _GOLDEN_BLOB_HEX = ( "4f504c430100659a3520540f803625ddc34081e893d329a17c7c2486d355" "fbff79f73b679ce59fa93304507867e82d7b41b93acd98274dc48531299e" @@ -116,21 +111,17 @@ def test_get_board_id_returns_raw_ascii_anchor(tmp_path, monkeypatch): def test_get_board_id_missing_anchor_is_empty_success(tmp_path, monkeypatch): - # No anchor -> LIC_UNSUPPORTED (review 2026-08-20, R2): on this medium 0x48 - # is ONLY the licensing anchor, and SUCCESS/0 made every anchor-less host - # derive the SAME deviceId -- a purchase bound to it never validated. + # No anchor -> LIC_UNSUPPORTED: on this medium 0x48 is only the + # licensing anchor, and SUCCESS/0 would make every anchor-less host + # derive the SAME deviceId, so purchases bound to it never validate. monkeypatch.setattr(lic, "ANCHOR_PATH", str(tmp_path / "nope")) assert lic.handle_license_command("48") == "48 85" def test_write_refuses_path_traversal(tmp_path, monkeypatch): - # A forged vpp_plugins.conf whose config_path escapes the runtime root must - # NOT let 0x49 write outside it (defense-in-depth; mirrors apply_vpp_plugin_conf). - # - # The blob is the VALID golden one on purpose: with blob validation in front - # of the path resolution, an invalid blob would be refused before the guard - # was ever reached, and this test could no longer tell a working guard from a - # missing one. + # 0x49 must refuse a config_path that escapes the runtime root even + # with a valid blob (blob validation runs first, so an invalid blob + # would never reach the path guard). cwd = tmp_path / "runtime" cwd.mkdir() monkeypatch.chdir(cwd) @@ -195,22 +186,6 @@ def test_write_without_installed_plugin_is_unsupported(tmp_path, monkeypatch): assert lic.handle_license_command(cmd) == "49 85" # LIC_UNSUPPORTED -# -------------------------------------------------------------------------- -# Blob integrity, both directions -# -# 0x4A used to test the LENGTH only. A 98-byte file that does not verify -- an -# SD card cloned from another Pi, corrupted flash, a torn write -- answered -# `4A 7E`, which the editor reads as "magic + crc32 verified", so it reported -# "Licensed" and returned BEFORE asking the backend for a fresh license. The one -# automatic repair path never ran precisely because the editor trusted the blob, -# while license_core refused it and the plugin dropped to demo. The same file on -# an ESP32 answers 0x83/0x84 and the editor recovers automatically. -# -# 0x49 wrote whatever it was handed, so 98 bytes of junk destroyed a valid -# license and answered SUCCESS. -# -------------------------------------------------------------------------- - - def test_read_reports_corrupt_when_the_crc_does_not_verify(tmp_path, monkeypatch): config_path = _install_plugin(tmp_path, monkeypatch) with open(config_path[:-5] + ".license", "wb") as handle: @@ -354,15 +329,6 @@ def _fake_open(path, *args, **kwargs): assert lic.handle_license_command("4A") == "4A 82" # IO_ERROR -# -------------------------------------------------------------------------- -# Anchor normalization (0x48) -# -# The C is canonical: rpi_plugin.c is the side that decides whether the license -# verifies. Byte-for-byte parity with the real C source is pinned separately, by -# test_vpp_anchor_cross_language.py; these are the wire-level consequences. -# -------------------------------------------------------------------------- - - def test_anchor_keeps_a_trailing_tab(tmp_path, monkeypatch): """TAB is NOT in the C's strip list, so it must not be in ours either. @@ -499,16 +465,6 @@ def emit(self, record): assert [r for r in records if r.levelno >= logging.WARNING] == [] -# -------------------------------------------------------------------------- -# Containment guard (is_inside_root) -# -# The pre-existing traversal test above uses a sibling that shares NO string -# prefix with the root, so the old buggy `startswith(root)` check rejected it -# too -- it could not tell the fixed guard from the broken one. These pin the -# two cases that actually distinguish them. -# -------------------------------------------------------------------------- - - def test_rejects_sibling_that_shares_the_root_as_a_string_prefix(tmp_path): """`/opt/runtime-evil/x` must not pass a root of `/opt/runtime`. diff --git a/tests/pytest/restapi/test_bootloader_auth_vector.py b/tests/pytest/restapi/test_bootloader_auth_vector.py index 5bae59f7..338ee89f 100644 --- a/tests/pytest/restapi/test_bootloader_auth_vector.py +++ b/tests/pytest/restapi/test_bootloader_auth_vector.py @@ -1,7 +1,7 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® -"""Python half of the bootloader's shared authentication vector (RTOP-283). +"""Python half of the bootloader's shared authentication vector. The bootloader is written in Go and reimplements two formats this codebase owns: Werkzeug's PBKDF2 password hash and Flask-JWT-Extended's HS256 access token. diff --git a/tests/pytest/restapi/test_capabilities.py b/tests/pytest/restapi/test_capabilities.py index b1d5372b..5ed489de 100644 --- a/tests/pytest/restapi/test_capabilities.py +++ b/tests/pytest/restapi/test_capabilities.py @@ -4,7 +4,7 @@ """Behavioural tests for GET /api/version and GET /api/capabilities. Both endpoints exist so an editor can decide, before login, whether it may -talk to this runtime at all (DOPE-448). The contract these tests pin down: +talk to this runtime at all. The contract these tests pin down: * both are reachable WITHOUT a token, even once users exist — an editor that cannot authenticate must still be able to tell why; diff --git a/tests/pytest/restapi/test_project_snapshot.py b/tests/pytest/restapi/test_project_snapshot.py index ad3c2611..332535f7 100644 --- a/tests/pytest/restapi/test_project_snapshot.py +++ b/tests/pytest/restapi/test_project_snapshot.py @@ -105,11 +105,8 @@ def test_clear_then_no_stage_erases_the_previous_project(): def test_a_stranded_staged_snapshot_is_never_promoted_by_a_later_build(): - # If an upload dies after staging but before the compile thread starts - # (a failed extract, say), nothing discards what it staged. The next - # upload's clear() has to be what removes it -- otherwise that build would - # promote a snapshot belonging to an upload that never landed, and the - # device would advertise a project it is definitely not running. + # The next upload's clear() must evict a stranded staged snapshot; + # otherwise it would be promoted by a build it did not belong to. project_snapshot.stage(b"stranded", _metadata(projectName="Never Landed")) project_snapshot.clear() # the next upload assert project_snapshot.promote() is False @@ -283,13 +280,8 @@ def test_capabilities_advertises_snapshot_support(client): assert client.get("/api/capabilities").get_json()["projectSnapshot"] is True -# --- the size guard at the boundary it defends --------------------------- -# -# `test_an_oversized_snapshot_is_refused` exercises `stage()` directly, which -# proves the cap fires once the bytes already exist. The point of the guard is -# that they never do: the route is authenticated but not admin-gated, so any -# account could otherwise have an arbitrarily large part spooled to disk and -# pulled into memory before anything refused it. +# Size-guard tests at the route boundary. The cap must fire before the +# bytes are spooled to disk, not only inside stage(). def _post_snapshot(client, admin_token, blob, *, metadata=None): @@ -398,11 +390,8 @@ def test_a_blob_left_without_metadata_says_so_in_the_log(caplog): assert any("no metadata beside it" in record.message for record in caplog.records) -# --- the advertised fields cache ----------------------------------------- -# -# The discovery responder reads these on every probe, so they are held in -# memory. A cache that can go stale would make the device advertise a project it -# is no longer running, which is the one thing this whole design refuses to do. +# Advertised-fields cache tests. The responder reads this per probe, so +# a stale cache would advertise a project no longer running. def test_what_the_device_advertises_follows_every_change_to_the_store(): @@ -422,11 +411,8 @@ def test_what_the_device_advertises_follows_every_change_to_the_store(): def test_a_staged_project_is_never_advertised(): - # By the time anything is staged, the upload has already passed its point of - # no return and cleared the old project -- the program it described is being - # replaced. So the device advertises nothing at all until a build succeeds, - # which is the honest answer: naming the staged project would name one the - # device is not running and may never run. + # Once staging starts, the old project is already cleared. The + # device advertises nothing until the build succeeds. _store(projectName="Running") project_snapshot.stage(b"new-archive", _metadata(projectName="Building")) diff --git a/tests/pytest/restapi/test_repair_missing_admin.py b/tests/pytest/restapi/test_repair_missing_admin.py index a4fdb64a..df9cb09d 100644 --- a/tests/pytest/restapi/test_repair_missing_admin.py +++ b/tests/pytest/restapi/test_repair_missing_admin.py @@ -52,15 +52,8 @@ def test_a_lone_non_admin_account_is_promoted(app): def test_a_null_role_counts_as_no_admin_rather_than_crashing(app): - # Rebuilt as the schema an affected device actually carries, copied from - # the hardware unit this was found on: - # - # role VARCHAR(20) -- nullable, no default - # - # The current model declares NOT NULL DEFAULT 'admin', so a NULL role - # cannot be created through it or even inserted into the table it builds. - # Reproducing the old shape is the only way to check the repair copes with - # what is out there rather than with what we would write today. + # Reproduces the legacy schema (role VARCHAR(20) nullable, no + # default) that the current model's NOT NULL would prevent. db.session.execute(sa_text("DROP TABLE users")) db.session.execute( sa_text( diff --git a/tests/support/debug_handler_mocks.c b/tests/support/debug_handler_mocks.c index 30732aee..878d2352 100644 --- a/tests/support/debug_handler_mocks.c +++ b/tests/support/debug_handler_mocks.c @@ -1,21 +1,10 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * debug_handler_mocks.c — implementation of the test-side debugger ABI - * fakes. See debug_handler_mocks.h for the contract. - * - * These functions are wired into the runtime by overwriting the - * ext_strucpp_debug_* function pointers (declared extern in - * image_tables.h, defined in image_tables.cpp) and the program-MD5 - * char* pointer (declared in utils.h, defined in utils.c). - * - * Because the runtime declares those externs in C++ (image_tables.cpp) - * and we install from C, the declarations are reproduced here under - * `extern "C"`-equivalent linkage. The installed function pointer - * signatures must match exactly — a mismatch silently corrupts the - * call frame. - */ +/* Implementation of the test-side debugger ABI fakes. The runtime's + * ext_strucpp_debug_* function pointers and the program-MD5 char* + * are overwritten to point here. Signatures must match exactly or + * the call frame is silently corrupted. */ #include "debug_handler_mocks.h" @@ -24,21 +13,9 @@ #include #include -/* The runtime's normal home for these is image_tables.cpp (function - * pointers) and utils.c (md5 char *). image_tables.cpp pulls in the - * full strucpp ABI and a lot of C++ infrastructure that's irrelevant - * to the debugger wire-protocol tests, so we provide the storage here - * in test-support land instead. Ceedling resolves the externs in - * debug_handler.c against these definitions and never compiles - * image_tables.cpp. - * - * scan_counter (referenced by debug_handler.c for the tick field of - * GET / GET_LIST responses) is owned by utils.c — that file is small - * and gets pulled in normally. - * - * If a future test ever wants the real image_tables.cpp definitions, - * gate this block with #ifndef MOCK_DEBUG_OWNS_EXTERNS or split it - * into a separate support file. */ +/* Storage for the externs in debug_handler.c, so Ceedling can link + * without pulling in image_tables.cpp (full strucpp ABI). + * scan_counter comes from utils.c. */ uint8_t (*ext_strucpp_debug_array_count)(void) = NULL; uint16_t (*ext_strucpp_debug_elem_count) (uint8_t) = NULL; uint16_t (*ext_strucpp_debug_size) (uint8_t, uint16_t) = NULL; diff --git a/tests/support/debug_handler_mocks.h b/tests/support/debug_handler_mocks.h index 16481ffc..c241fd4c 100644 --- a/tests/support/debug_handler_mocks.h +++ b/tests/support/debug_handler_mocks.h @@ -1,20 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * debug_handler_mocks.h — controllable fakes for the strucpp debugger ABI. - * - * The runtime resolves these function pointers from the loaded program .so - * at start time (image_tables.cpp:symbols_init). For tests, we install - * fakes whose behavior can be programmed per-test via the `mock_debug_*` - * setters: array layout, per-element bytes, write capture, and forced - * "out of range" returns from the .so side so we can verify the runtime - * gate doesn't depend on cooperation from the .so. - * - * Tests opt into the fakes via mock_debug_install() (typically in setUp). - * mock_debug_reset() returns the table to a canonical empty state without - * tearing down the function pointers, so each test sees a fresh slate. - */ +/* Fakes for the strucpp debugger ABI. install() replaces the + * ext_strucpp_debug_* pointers, reset() clears the fake table. + * Per-test setters program layout, bytes, writes and OOB returns. */ #ifndef TESTS_SUPPORT_DEBUG_HANDLER_MOCKS_H #define TESTS_SUPPORT_DEBUG_HANDLER_MOCKS_H @@ -72,10 +61,8 @@ const mock_debug_set_capture_t *mock_debug_last_set(void); /* Override the return value of the next debug_set() / debug_write() call. */ void mock_debug_program_set_status(uint8_t status); -/* Install (or clear) the program MD5 string. NULL clears the pointer - * entirely so tests can assert the "not loaded" branch. The - * `terminated` flag controls whether a trailing null byte is written — - * tests that exercise the unbounded-read mitigation set this to false.*/ +/* Install (NULL clears) the program MD5 string. `terminated` controls + * whether a trailing NUL is written. */ void mock_debug_set_md5(const char *md5_chars, size_t len, bool terminated); #ifdef __cplusplus diff --git a/tests/support/plugin_driver_stubs.c b/tests/support/plugin_driver_stubs.c index 40ec8ae1..78a9cc84 100644 --- a/tests/support/plugin_driver_stubs.c +++ b/tests/support/plugin_driver_stubs.c @@ -92,18 +92,9 @@ int journal_write_lint(journal_buffer_type_t type, uint16_t index, return 0; } -// The MatIEC-era flat-index API (get_var_list / get_var_size / -// get_var_count from plugin_utils.c) was removed alongside the rest of -// the MatIEC pipeline. Plugins now receive structured runtime args -// (plugin_runtime_args_t) constructed from the STruC++ debug map; the -// debugger ABI is exercised in test_debug_handler.c. - -// Stubs: plc_tasks_reader_lock / plc_tasks_reader_unlock (plc_state_manager.cpp). -// scan_cycle_manager.c calls these around format_timing_stats_response to -// keep the reader from racing the bootstrap thread freeing plc_tasks. The -// real lock lives in plc_state_manager.cpp; tests don't pull that .cpp in, -// so we provide no-op stubs. Tests that exercise the lifecycle (rather -// than just the per-tracker math) will need to link the real symbols. +// No-op stubs: scan_cycle_manager.c takes these around format_timing_ +// stats_response. Tests exercising the real lifecycle must link the +// real plc_state_manager.cpp symbols instead. void plc_tasks_reader_lock(void) {} void plc_tasks_reader_unlock(void) {} diff --git a/tests/test_debug_handler.c b/tests/test_debug_handler.c index ad41486f..b132a207 100644 --- a/tests/test_debug_handler.c +++ b/tests/test_debug_handler.c @@ -1,22 +1,10 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * test_debug_handler.c — wire-level tests for the new STruC++ debugger - * ABI (FC 0x41-0x45) at the `process_debug_data` boundary. - * - * Replaces the MatIEC-era flat-index API tests (get_var_list / - * get_var_size / get_var_count) deleted with the runtime cleanup. Goal - * is to lock the on-wire frame format (mirrored by the editor's debug - * client at src/frontend/utils/debug-parser.ts and the Arduino - * StrucppBaremetal/ModbusSlave.cpp), and to pin the defensive bounds - * checks the runtime added on top of the .so's internal validation. - * - * Tests use the mock debugger ABI in tests/support/debug_handler_mocks.* - * to drive controlled `arr_count` / `elem_count` / `read` / `set` - * behavior. process_debug_data is called directly with a constructed - * frame; the response is unpacked and compared against expected bytes. - */ +/* Wire-level tests for the STruC++ debugger ABI (FC 0x41-0x45) at the + * `process_debug_data` boundary. The mock debugger ABI in + * tests/support/debug_handler_mocks.* drives controlled responses so + * the on-wire frame format and runtime bounds checks can be pinned. */ #include "debug_handler.h" #include "debug_handler_mocks.h" @@ -172,12 +160,8 @@ void test_debug_set_unforce_clears_forcing_flag(void) void test_debug_set_rejects_oob_arr_at_runtime_gate(void) { - /* Configures the .so as having ONE array. The wire request asks - * to set arr=5 — way out of range. Without the runtime gate (the - * fix for review issue #17), this would call into the .so's - * debug_set with an OOB arr index and rely on the .so to validate. - * With the gate, the runtime returns OUT_OF_BOUNDS without ever - * dispatching. */ + /* One array configured; request asks arr=5. Runtime gate returns + * OUT_OF_BOUNDS without dispatching to the .so's debug_set. */ mock_debug_set_arr_count(1); mock_debug_set_elem_count(0, 4); diff --git a/tests/test_located_globals.c b/tests/test_located_globals.c index 3300919f..6db676ba 100644 --- a/tests/test_located_globals.c +++ b/tests/test_located_globals.c @@ -1,34 +1,9 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * test_located_globals.c — unit tests for located_globals_join_ex(), which - * resolves which locatedVars[] entries are CONFIGURATION VAR_GLOBAL ... AT by - * joining against the .so's locatedGlobals[] on pointer identity. - * - * Why this exists: the runtime used to derive the config-scope set from POSITION, - * assuming strucpp emitted [program-local ...][config globals ...] and treating - * the tail uncovered by any program's located_range() as the globals. strucpp - * emits config globals FIRST, so the program-local block is always the tail: - * `covered_end` reached locatedVarsCount as soon as any POU declared a located - * variable, the count collapsed to 0, and EVERY located configuration global - * (%MX, %QX, %MW alike) silently stopped being synced. Reported on the forum as - * "%MX locations now invalid - Runtime v4". - * - * The behaviour locked here: - * - the join follows pointer identity only, so the result is identical whether - * globals come first, last, or interleaved (ordering_* cases); - * - out_matched reports how many locatedGlobals[] entries found a located - * variable, so the caller can detect the two generated arrays disagreeing; - * - unbound (NULL) descriptors are skipped rather than guessed at; - * - an absent/empty globals array yields zero config-scope entries, which is - * the "older program" degradation path. - * - * Build (standalone, no runtime deps): - * cc -std=c11 -Wall -Wextra -I core/src/plc_app \ - * tests/test_located_globals.c core/src/plc_app/located_globals.c \ - * -o /tmp/test_located_globals && /tmp/test_located_globals - */ +/* Unit tests for located_globals_join_ex(). Pin: join is by pointer + * identity, order-independent; out_matched #include @@ -45,10 +20,6 @@ static int g_failures = 0; } \ } while (0) -/* --------------------------------------------------------------------------- - * Stand-in for locatedVars[]: an array of storage pointers. Distinct dummy - * objects give distinct addresses, exactly as real IEC storage does. - * ------------------------------------------------------------------------- */ #define MAX_VARS 8 static char g_storage[MAX_VARS]; /* distinct addresses */ static const void *g_lv[MAX_VARS]; /* the "locatedVars[i].pointer" values */ @@ -67,10 +38,6 @@ static void reset(void) for (uint32_t i = 0; i < MAX_VARS; ++i) g_lv[i] = NULL; } -/* --------------------------------------------------------------------------- - * The regression: config globals FIRST, one program-local var last. Under the - * old positional rule this produced zero config entries. - * ------------------------------------------------------------------------- */ static void test_globals_first(void) { printf("test_globals_first (the reported regression)\n"); @@ -123,10 +90,6 @@ static void test_ordering_independence(void) } } -/* --------------------------------------------------------------------------- - * Multi-program shape that also collapsed to zero under the old rule: - * 3 globals, then two programs with two located vars each. - * ------------------------------------------------------------------------- */ static void test_multi_program(void) { printf("test_multi_program\n"); @@ -204,11 +167,6 @@ static void test_degradation_and_edges(void) } } -/* --------------------------------------------------------------------------- - * Inconsistency detection: a locatedGlobals[] entry that matches no located - * variable means the two generated arrays disagree. The join must report it via - * out_matched rather than silently dropping it. - * ------------------------------------------------------------------------- */ static void test_inconsistency_detected(void) { printf("test_inconsistency_detected\n"); @@ -227,10 +185,6 @@ static void test_inconsistency_detected(void) matched); } -/* --------------------------------------------------------------------------- - * Unbound descriptors: a located variable whose pointer was never populated - * (e.g. a program never instantiated) must be skipped, not guessed at. - * ------------------------------------------------------------------------- */ static void test_unbound_skipped(void) { printf("test_unbound_skipped\n"); diff --git a/tests/test_plugin_config.c b/tests/test_plugin_config.c index 297fe509..5e87879d 100644 --- a/tests/test_plugin_config.c +++ b/tests/test_plugin_config.c @@ -6,13 +6,6 @@ #include "unity.h" #include -// Mock functions for standard library calls used in plugin_config.c -// Cmock will generate these automatically when we #include "mock_stdlib.h" or similar, -// but for direct functions like fopen, fgets, etc., we might need to create them manually -// or use a more generic mock approach if Cmock doesn't handle them out of the box. -// For simplicity, we'll assume Cmock can handle these or we'll create simple wrappers. -// Let's start by assuming Cmock handles them. If not, we'll adjust. - // Helper function to create a temporary config file for testing static void create_test_config_file(const char *filename, const char *content) { diff --git a/tests/test_plugin_driver.c b/tests/test_plugin_driver.c index 2fc7615f..a63cf5a5 100644 --- a/tests/test_plugin_driver.c +++ b/tests/test_plugin_driver.c @@ -189,17 +189,7 @@ void test_plugin_driver_data_structure_ShouldStorePluginInfo(void) } driver.plugin_count = config_count; - // In a complete implementation, you would mock python_plugin_get_symbols here - // For example: - // python_plugin_get_symbols_ExpectAndReturn(&driver.plugins[0], 0); // Success for py_plugin - // python_plugin_get_symbols_ExpectAndReturn(&driver.plugins[2], 0); // Success for - // py_plugin_venv - - // For this test, we're just testing the data structure population - // In a more complete test, you would mock plugin_driver_load_config entirely - // For now, we just test that our mock data was set up correctly - - // Assertions - testing the setup we created (simulating successful config loading) + // Assertions on the hand-wired driver (no load_config mock here). TEST_ASSERT_EQUAL_INT_MESSAGE(3, driver.plugin_count, "Driver plugin count should be 3"); // Validate plugin 1 (Python) @@ -218,10 +208,8 @@ void test_plugin_driver_data_structure_ShouldStorePluginInfo(void) // No cleanup needed for driver if it's stack allocated } -// Test Case 5: Test calling plugins that failed initialization -// This test focuses on the `plugin_driver_init` function and how it handles -// plugins where the `init` function (Python or Native) returns an error. -// plugins where the `init` function (Python or Native) returns an error. +// Test Case 5: plugin_driver_init must halt and return error when a +// plugin's init (Python or Native) returns non-zero. void test_plugin_driver_Init_WhenPluginInitFails_ShouldHaltAndReturnError(void) { // This test requires extensive mocking of Python C API and plugin structures. diff --git a/tests/test_scan_cycle_tracker.c b/tests/test_scan_cycle_tracker.c index f3fc0233..8d8fb3ea 100644 --- a/tests/test_scan_cycle_tracker.c +++ b/tests/test_scan_cycle_tracker.c @@ -1,31 +1,10 @@ // SPDX-License-Identifier: MIT // Copyright (c) 2026 Autonomy® -/* - * test_scan_cycle_tracker.c — unit tests for the per-task scan-cycle - * tracker introduced alongside the multi-task refactor. - * - * Replaces the previous global-stats path (single fastest-task counters - * shared across all threads). Each task now owns its own tracker and the - * STATS handler walks plc_tasks[] to emit one entry per task. - * - * The behaviour we lock here: - * - first call to scan_cycle_tracker_start() seeds anchors WITHOUT - * emitting cycle-time / latency stats (those need a baseline); - * - second and subsequent starts compute cycle_time = now - - * last_start, latency = now - expected_start; - * - scan_cycle_tracker_end() captures scan_time = now - last_start - * and increments `overruns` if we ran past the projected next-wakeup; - * - scan_cycle_tracker_snapshot() returns false until at least one - * scan completes, then returns the tracker's stats with the avg - * fields recovered from the EWMA sum/avg_window. - * - * The EWMA window is computed as EWMA_TARGET_WINDOW_US / interval_us - * (clamped to >= 1). Tests that pin specific `avg` values choose - * intervals that produce avg_window=1, so a single sample IS the average - * — that side-steps the cold-start ramp the reviewer flagged in #16, - * which is intended behaviour. - */ +/* Unit tests for the per-task scan_cycle_tracker. Pin: first start() + * only seeds anchors, later starts emit cycle/latency, end() records + * scan_time and bumps overruns. Tests pick intervals that make + * avg_window=1 so the EWMA cold-start ramp drops out. */ #include "scan_cycle_manager.h" #include "unity.h" @@ -116,11 +95,8 @@ void test_init_avg_window_matches_target_for_100ms_cycle(void) void test_first_start_only_seeds_no_stats_emitted(void) { - /* The first call to start() lays down anchors but cannot compute - * cycle_time or latency (no prior reference). After it returns, - * scan_count is 1 but stats aren't meaningful — snapshot returns - * true once scan_count > 0, but min fields stay at INT64_MAX - * until the second cycle observes something. */ + /* First start() seeds anchors only; cycle_time has no baseline, so + * min fields stay at INT64_MAX until the second cycle. */ scan_cycle_tracker_init(&tracker, 1000000); scan_cycle_tracker_start(&tracker); @@ -224,15 +200,8 @@ void test_no_overrun_when_scan_finishes_within_period(void) void test_avg_recovers_single_sample_when_avg_window_is_one(void) { - /* avg_window=1 makes the EWMA collapse to "the latest sample IS - * the average". interval_ns = EWMA_TARGET_WINDOW_US * 1000 makes - * the calculation interval_us / EWMA_TARGET_WINDOW_US = 1 sample. - * - * This sidesteps the cold-start ramp the reviewer flagged in #16 - * — at avg_window=1 there is no ramp. - * - * 2 s = 2_000_000 us → interval_ns = 2_000_000_000 (2 s cycle). - */ + /* avg_window=1 collapses the EWMA to the latest sample, which + * sidesteps the cold-start ramp. 2 s cycle = interval_ns 2e9. */ scan_cycle_tracker_init(&tracker, 2000000000LL); TEST_ASSERT_EQUAL_INT64(1, tracker.avg_window); diff --git a/webserver/app.py b/webserver/app.py index 33657fb2..a38f2247 100644 --- a/webserver/app.py +++ b/webserver/app.py @@ -64,13 +64,9 @@ app = flask.Flask(__name__) app.secret_key = str(os.urandom(16)) -# A backstop at the HTTP layer, under everything the routes do. -# -# Individual handlers check their own parts, but those checks run after Werkzeug -# has already parsed (and spooled to disk) the request. This bounds the whole -# body first, so an oversized upload is refused as 413 before any of it is -# stored. Sized to hold the largest legitimate request -- a program zip and a -# project snapshot together -- plus room for the multipart framing. +# HTTP body cap applied before Werkzeug spools the request to disk, so +# per-route checks never run on bytes that were already stored. Sized for +# a program zip plus a project snapshot plus multipart framing. app.config["MAX_CONTENT_LENGTH"] = ( MAX_FILE_SIZE + project_snapshot.MAX_SNAPSHOT_BYTES + (8 * 1024 * 1024) ) @@ -183,10 +179,8 @@ def handle_status(data: dict) -> dict: result: dict = {"status": response} - # Mode-switch position, so the editor can block a start before sending it - # rather than relying on the runtime's refusal alone. Additive: the existing - # `status` key is untouched, and an older editor simply ignores this field. - # A runtime with no switch-aware plugin always reports "run". + # Mode-switch position. Additive key: an older editor ignores it, and a + # runtime with no switch-aware plugin reports "run". switch_position = parse_switch_position(runtime_manager.switch_plc()) if switch_position is not None: result["switchPosition"] = switch_position @@ -302,18 +296,9 @@ def stage_project_snapshot() -> str: except project_snapshot.SnapshotError as e: return f"Snapshot ignored: {e}" - # Bounded read, before the bytes exist rather than after. - # - # `stage()` also enforces the cap, but only once the whole part is already - # in memory -- and this route is authenticated without an admin gate, so any - # account on the device could post an arbitrarily large `snapshot` field and - # have it spooled to disk and then pulled into RAM before anything refused - # it. On the hardware this runtime targets that is a disk-fill followed by - # an OOM. - # - # `content_length` on a multipart part is client-supplied and often absent, - # so it is a fast path and not the guard. Reading one byte past the cap and - # stopping is what actually bounds this, whatever the client claimed. + # Guard the size BEFORE the bytes are read into memory: stage() also caps, + # but only after the part is already in RAM. Read one byte past the cap + # to detect overshoot; declared content_length is a hint, not the guard. declared = snapshot_file.content_length if declared and declared > project_snapshot.MAX_SNAPSHOT_BYTES: return ( @@ -449,16 +434,10 @@ def _handle_upload_file(data: dict) -> dict: if was_running: build_state.log("[WARNING] The PLC was running; stopped it before the upload\n") - # Point of no return: past here the program on the device is being - # replaced, so the stored project snapshot must go with it. Clearing - # here rather than on arrival means a rejected upload (bad zip, too - # large, runtime busy) leaves the previous program AND its snapshot - # untouched, which is the pair that is actually still true. - # - # An upload carrying no snapshot therefore erases the stored one -- - # that is the point. Older editors, openplc-cli and any third-party - # client keep working, and the device stops advertising a project it - # is no longer running. + # Clear the stored snapshot together with the program it describes. + # Done here (not on arrival) so a rejected upload leaves program and + # snapshot both untouched. An upload with no snapshot therefore + # erases the stored one. project_snapshot.clear() if os.path.exists(extract_dir): @@ -469,19 +448,10 @@ def _handle_upload_file(data: dict) -> dict: # Apply VPP plugin conf from upload (copy if present, delete if not) apply_vpp_plugin_conf(extract_dir) - # Persistent storage settings, same present/absent contract as the VPP - # conf above: the project owns them, so an upload that carries - # retain.conf installs it and one that does not removes the device's - # copy. That absent case is what lets a target whose VPP owns retention - # switch the built-in file store off simply by not configuring it. - # - # Nothing clears retained VALUES here. The store itself decides, at - # program start, whether what it holds belongs to the program now - # running — it compares the program MD5 it stored against the one the - # runtime hands it. Doing it there rather than here is what makes the - # two platforms behave identically: baremetal has no webserver to - # observe an upload, and a device flashed or provisioned by any other - # route still reaches the right answer. + # Project owns retain.conf: an upload with it installs; without it + # removes the device's copy. Retained VALUES are not cleared here; + # the store compares program MD5 at start, which also works on + # baremetal where no webserver observes an upload. apply_retain_conf(extract_dir) # Update built-in plugin configurations based on extracted config files @@ -497,14 +467,9 @@ def _handle_upload_file(data: dict) -> dict: # don't pass this flag, so behaviour for them is unchanged. clean_build = flask.request.args.get("clean") == "1" - # Stage the snapshot only once the program itself is safely in place. - # run_compile's `finally` is what promotes or discards a staged - # snapshot, so staging before the extract would leave one stranded if - # the extract threw -- the compile thread never starts, nothing - # discards it, and the NEXT successful build would promote a snapshot - # belonging to an upload that never landed. The clear() above has - # already erased the old one either way, which is correct: the program - # it described is gone. + # Stage AFTER the extract succeeded: run_compile's finally promotes or + # discards. Staging earlier risks leaving a snapshot behind if the + # extract throws before the compile thread starts. snapshot_error = stage_project_snapshot() # Start compilation in a separate thread @@ -519,10 +484,9 @@ def _handle_upload_file(data: dict) -> dict: task_compile.start() - # The program upload itself succeeded. A snapshot that could not be - # stored is reported alongside rather than as a failure: the device is - # running the new program either way, and failing the upload over the - # optional half of it would be worse than losing retrievability. + # A snapshot error is reported alongside, not as a failure: the new + # program is live either way, and losing retrievability is less bad + # than refusing the upload over the optional half. return { "UploadFileFail": "", "CompilationStatus": build_state.status.name, @@ -602,12 +566,10 @@ def run_https(): # users.role column in place; no-op once present). apply_user_schema_migrations() db.session.commit() - # Rescue a device left with accounts but no admin. Unlike the - # schema migration above this is a DATA repair, and it has to run - # separately: the migration only fires when the role column is - # missing, so a database that already has one keeps whatever values - # it holds -- including none of them being 'admin'. Without an - # admin there is no API path back to having one. + # Data repair for a device with accounts but no admin. Runs + # separately from the schema migration (which only fires when the + # role column is missing). Without an admin there is no API path + # back to having one. repair_missing_admin() # logger.info("Database tables created successfully.") except Exception: diff --git a/webserver/config.py b/webserver/config.py index baffbe16..62c58fa7 100644 --- a/webserver/config.py +++ b/webserver/config.py @@ -107,18 +107,9 @@ def get_persistent_data_dir(): PERSISTENT_DATA_DIR = get_persistent_data_dir() ENV_PATH = PERSISTENT_DATA_DIR / ".env" DB_PATH = PERSISTENT_DATA_DIR / "restapi.db" -# VPP plugin configs + license blobs live here, OUTSIDE $OPENPLC_DIR/build, so a -# runtime version update (install.sh does ``rm -rf $OPENPLC_DIR/build``) can never -# delete a purchased license. The closed .so still reads them because the runtime -# writes this absolute path into vpp_plugins.conf's config_path field (see -# webserver/plcapp_management.py::apply_vpp_plugin_conf); the C loader passes -# config_path to the plugin verbatim, so only the .so binary itself must stay -# under build/vpp. -# -# Created on demand by apply_vpp_plugin_conf / _write_license_atomically, NOT at -# import: a bare module-scope mkdir is import-time filesystem work that turns a -# permission failure into a hard import crash (the very thing tests/pytest/ -# conftest.py exists to work around). +# VPP configs and license blobs live outside build/ so install.sh's wipe +# does not delete a purchased license. The runtime writes this path into +# vpp_plugins.conf.config_path; the C loader passes it to the .so verbatim. VPP_DATA_DIR = PERSISTENT_DATA_DIR / "vpp" BASE_DIR = os.path.abspath(os.path.dirname(__file__)) diff --git a/webserver/debug_websocket.py b/webserver/debug_websocket.py index 46cf876d..8bdafbe6 100644 --- a/webserver/debug_websocket.py +++ b/webserver/debug_websocket.py @@ -17,11 +17,8 @@ from webserver.logger import get_logger -# The debug socket is mounted on app_restapi and shares its JWT manager, so it -# also shares the two decisions that outlive a token: the logout blacklist and -# who the token's subject actually is. Importing the loaders rather than reaching -# into `jwt_blacklist` keeps one definition of each. Safe direction: restapi does -# not import this module. +# Share restapi's JWT manager (blacklist + subject lookup) by importing its +# callbacks. One-way: restapi never imports this module. from webserver.restapi import check_if_token_revoked, user_lookup_callback from webserver.vpp_license_debug import handle_license_command @@ -237,27 +234,16 @@ def handle_debug_command(data): emit("debug_response", {"success": False, "error": "Empty command"}) return - # Re-check EVERY command, not just the connect. See - # _reverify_session_token: a revoked token, or one whose account is - # gone, must stop working on a socket that is already open. Plain - # expiry does not, which is why this is a re-CHECK and not a full - # re-authentication. + # Re-CHECK on every command: a revoked token or a deleted account + # must stop working on a socket already open. Expiry alone would + # not catch that. if not _reverify_session_token(): emit("debug_response", {"success": False, "error": "Unauthorized"}) return - # The license FCs are open to any AUTHENTICATED role, not just admin - # (decision 2026-08-25). They were admin-gated on the theory that the - # anchor read (0x48) and the blob write (0x49) were a trust boundary, - # but that gate protected the wrong thing: the PURCHASE is authorized - # by the Edge account on the /buy page, never by the runtime role, so - # requiring admin here only stopped an operator from activating a - # licence they had already paid for. What stays open is low-risk: the - # anchor is the board's serial (baremetal exposes it with no auth at - # all), the blob is node-locked and useless on another device, and a - # bad write is recoverable (the entitlement lives in the backend; a - # refresh rewrites the correct blob). JWT re-verification above still - # applies, so "any role" means any logged-in user, never anonymous. + # License FCs (0x48/0x49/0x4A) are open to any authenticated role. + # Purchase authority lives on the Edge /buy page, not in the runtime + # role; the on-device entitlement is node-locked and recoverable. # License function codes (0x48/0x49/0x4A) operate on host files # (/proc anchor + conf/.license) and are resolved here in diff --git a/webserver/discovery/network_discovery.py b/webserver/discovery/network_discovery.py index 271afb7e..2079ea39 100644 --- a/webserver/discovery/network_discovery.py +++ b/webserver/discovery/network_discovery.py @@ -142,13 +142,9 @@ def _handle(self, data: bytes, addr: tuple[str, int]) -> None: ip: ts for ip, ts in self._last_seen.items() if ts >= cutoff } - # Name and timestamp of the stored source project, when there is one. - # This is what lets a client populate its "Retrieve Project from PLC" - # picker without logging in to every device on the LAN first. Absent - # keys mean no stored project, so there is no separate flag. - # - # Deliberately just these two: they are exactly what the picker shows. - # Everything else about the stored project needs authentication. + # Stored project name and timestamp are advertised unauthenticated so + # the picker can populate without probing every device. Absent keys + # mean no stored project; everything else requires auth. payload = json.dumps( { "service": "openplc-runtime", diff --git a/webserver/plcapp_management.py b/webserver/plcapp_management.py index 87df8270..2ead31eb 100644 --- a/webserver/plcapp_management.py +++ b/webserver/plcapp_management.py @@ -158,13 +158,9 @@ def safe_extract(zip_path, dest_dir, valid_files): out_path = os.path.join(dest_dir, filename) out_path = os.path.abspath(out_path) - # Ensure extraction stays inside destination. Same containment rule - # as the VPP config copy below: a bare prefix check accepts a - # sibling sharing dest_dir as a string prefix (dest_dir - # "core/generated" vs. an entry resolving to "core/generatedX/..."), - # and it ignores symlinks entirely. analyze_zip() already rejects - # entries containing ".." before we get here, so this is defence in - # depth -- but it is the same bug class, so it gets the same fix. + # Defence in depth: analyze_zip() already rejects ".." entries, + # but is_inside_root also catches prefix-sibling escapes and + # symlink targets a bare prefix check would miss. if not is_inside_root(out_path, dest_dir): # logger.warning("Skipping suspicious path: %s", filename) continue @@ -345,12 +341,9 @@ def against_root(candidate: str) -> str: if not p.path: return False, f"plugin '{p.name}' has an empty path" if os.path.isabs(p.path): - # The C loader (plugin_config.c, require_contained=1) rejects EVERY - # absolute path; tolerating a contained absolute here produced a - # conf accepted by the upload and silently dropped at parse time -- - # the VPP never loaded and nothing said why (review 2026-08-20, - # R5). The two guards must agree, and the editor only ever emits - # relative paths, so nothing legitimate breaks. + # Must match plugin_config.c (require_contained=1), which rejects + # every absolute path. Accepting one here would silently drop the + # VPP at C-loader parse time with no feedback. return False, ( f"plugin '{p.name}' path '{p.path}' is absolute -- plugin paths " f"must be relative to the runtime root (./build/vpp/...)" @@ -416,11 +409,9 @@ def apply_vpp_plugin_conf(generated_dir: str = "core/generated") -> None: shutil.copy2(uploaded_conf, VPP_CONF_DEST) build_state.log(f"[INFO] VPP: installed vpp_plugins.conf from upload\n") - # Copy each VPP plugin's config file into the persistent dir and rewrite - # its config_path to point there (see the loop below). config_path is the - # single source of truth for where the .so looks for its config at - # runtime, so relocating it there is what carries config+license out of - # the wipe-on-update build/ tree. + # Copy each VPP config into the persistent dir and rewrite config_path. + # config_path is the single source of truth for the .so, so this is + # what carries config+license out of the wipe-on-update build/ tree. conf_dir = os.path.join(generated_dir, "conf") vpp_conf_plugins = PluginsConfiguration.from_file(VPP_CONF_DEST) rewrote_paths = False @@ -432,19 +423,9 @@ def apply_vpp_plugin_conf(generated_dir: str = "core/generated") -> None: build_state.log(f"[WARNING] VPP: conf/{p.name}.json not found in upload, skipping\n") continue - # Relocate the config (and its license sibling) OUT of build/vpp and - # into PERSISTENT_DATA_DIR/vpp: install.sh does `rm -rf $OPENPLC_DIR/ - # build` on a runtime version update, which used to delete the - # purchased license with it. The .so still finds them because we - # rewrite config_path in vpp_plugins.conf below to this persistent - # absolute path -- the C loader passes config_path to the plugin - # verbatim (plugin_config.c only contains `path`, the .so itself, - # which stays under build/vpp). - # - # The destination is built from the plugin NAME (a basename), NEVER - # from the editor-supplied config_path, so a forged conf cannot steer - # the write outside the persistent dir. A name that is not a plain - # filename is refused rather than trusted. + # Destination is built from the plugin NAME (basename), never from + # editor-supplied config_path, so a forged conf cannot steer the + # write outside the persistent dir. A non-basename name is refused. if not p.name or os.path.basename(p.name) != p.name: build_state.log(f"[WARNING] VPP: suspicious plugin name '{p.name}', skipping\n") continue @@ -452,16 +433,9 @@ def apply_vpp_plugin_conf(generated_dir: str = "core/generated") -> None: if not is_inside_root(dest_config, str(VPP_DATA_DIR)): build_state.log(f"[WARNING] VPP: config dest '{dest_config}' escapes the persistent dir, skipping\n") continue - # The old build/vpp sibling of THIS plugin, so a device licensed - # before this change can be migrated below. Derived from the FIXED - # build/vpp location plus the (already basename-checked) plugin name - # -- NOT from config_path. config_path is only confined to the runtime - # root by validate_vpp_plugins_conf (not to build/vpp), so deriving - # the migration source from it would let a forged conf point the read - # at any .license under the root and have it copied where 0x4A reads - # it back. The old code always wrote the license next to a build/vpp - # config, so this is exactly where a pre-change license lives, and it - # cannot be steered anywhere else. + # Migration source derived from the FIXED build/vpp location plus + # the basename-checked plugin name, never from config_path. Using + # config_path would let a forged conf steer the read anywhere. old_license = os.path.join(runtime_root, VPP_BUILD_DIR, f"{p.name}.license") os.makedirs(os.path.dirname(dest_config), exist_ok=True) @@ -475,23 +449,16 @@ def apply_vpp_plugin_conf(generated_dir: str = "core/generated") -> None: rewrote_paths = True dest_license = derive_license_path(dest_config) - # Deliver the optional device license blob to the sibling of the - # persistent config (derive_license_path, shared with the 0x49 - # handler so both write the SAME file the .so reads). Present only - # for a licensed VPP whose device was activated; absent for free - # VPPs or demo devices. + # Deliver the optional license blob to the sibling of the persistent + # config (same path 0x49 writes and the .so reads). src_license = os.path.join(conf_dir, f"{p.name}.license") if os.path.exists(src_license): shutil.copy2(src_license, dest_license) build_state.log(f"[INFO] VPP: copied {p.name}.license to {dest_license}\n") elif old_license and os.path.exists(old_license) and not os.path.exists(dest_license): - # One-time migration: a device licensed before this change has - # its blob next to the OLD build/vpp config. Move it to the - # persistent sibling when the upload did not carry one, so the - # license is not orphaned in a directory install.sh wipes. - # Best-effort: a failure here just means the device re-activates - # from its existing entitlement on the next connect, as it does - # today when 0x4A reads EMPTY. + # One-time migration: move a pre-change license from + # build/vpp to the persistent sibling. Best-effort: a failure + # here reactivates from the backend on next connect. try: shutil.copy2(old_license, dest_license) build_state.log(f"[INFO] VPP: migrated {p.name}.license {old_license} -> {dest_license}\n") @@ -564,14 +531,9 @@ def apply_retain_conf(generated_dir: str = "core/generated") -> None: try: if cfg["enabled"]: - # Both only meaningful when the store is on. A disabled stanza - # carrying a path that does not exist, or a flush period outside the - # bounds, is not worth refusing an upload over -- neither is read - # while enabled=0, and refusing would delete the device's existing - # config over a field nothing consults. That bites for real when a - # later release tightens MAX_FLUSH_SECONDS: every project still - # carrying the old value would have its whole retain.conf refused, - # including the ones that had storage switched off anyway. + # Only validate when the store is on. A disabled stanza with a + # bad path or out-of-bounds flush period is harmless and refusing + # it would delete the device's existing config over an unread field. validate_retain_path(cfg["path"]) validate_flush_seconds(cfg["flushSeconds"]) except RetainConfigError as e: @@ -585,18 +547,10 @@ def apply_retain_conf(generated_dir: str = "core/generated") -> None: build_state.log("[INFO] Retain: removed previous retain.conf\n") return - # WRITTEN, not copied — and that is load-bearing, not stylistic. - # - # The editor emits `path=` to mean "use this device's default": it does not - # know the device's filesystem layout and should not guess at one. - # `read_retain_conf_file` substitutes this device's default for that empty - # value, so writing the PARSED stanza is what materialises it. Copying the - # upload byte-for-byte would ship the empty value to the core, which treats - # enabled-with-no-path as a misconfiguration and leaves the store off — so - # "use the default" would silently become "no retention at all". - # - # It also means anyone reading retain.conf on the device sees the real - # location rather than a blank. + # Write the PARSED stanza, not a byte copy of the upload: an empty + # `path=` from the editor means "use this device's default", which + # read_retain_conf_file substitutes. A byte copy would ship the empty + # value and the core would treat it as a misconfiguration. write_retain_conf_file( dest, enabled=cfg["enabled"], path_value=cfg["path"], flush_seconds=cfg["flushSeconds"] ) @@ -618,13 +572,9 @@ def run_compile(runtime_manager: RuntimeManager, cwd: str = "core/generated", cl script_path: str = "./scripts/compile.sh" def stream_output(pipe, prefix): - # If the drainer dies mid-stream (e.g. UnicodeDecodeError on a - # non-UTF8 byte from g++ stderr, transient I/O error), the - # subprocess eventually fills its pipe buffer and blocks on its - # next write — which means compile_proc.wait() never returns and - # the build is silently stuck in COMPILING forever. Catch here so - # the operator sees the drainer error, and the finally{} pipe - # close lets the child see EOF instead of blocking. + # If this drainer dies, the child's pipe fills and compile_proc.wait() + # blocks forever, pinning status at COMPILING. Catch to surface the + # error; the finally closes the pipe so the child sees EOF. try: for line in iter(pipe.readline, ''): msg = f"{prefix}{line}" @@ -647,20 +597,9 @@ def wait_step(proc: subprocess.Popen, step_name: str) -> bool: build_state.log(f"[ERROR] {step_name} failed (exit={exit_code})\n") return False - # Wrap the entire orchestration body in try/except so any unhandled - # exception flips status to FAILED instead of leaving it pinned at - # COMPILING. Without this guard, a raise inside run_compile (e.g. - # Popen FileNotFoundError on a missing script, OSError on a - # full disk, or the inner update_plugin_configurations catch - # re-raising) propagates out of the daemon thread that - # run_compile is invoked on. The thread dies silently and the - # /api/compilation-status endpoint keeps reporting COMPILING - # indefinitely; the editor only recovers via its TCP - # connection-timeout safety net several minutes later (after the - # runtime stops responding at the network layer for unrelated - # reasons). Catching here makes the runtime fail-closed: any crash - # transitions to a terminal status the editor can observe on its - # next poll. + # Fail-closed guard: without this, a raise inside run_compile dies in the + # daemon thread and status stays pinned at COMPILING, which the editor + # only recovers from via its minutes-long TCP timeout. try: build_state.status = BuildStatus.COMPILING build_state.log(f"[INFO] Starting compilation\n") @@ -718,15 +657,9 @@ def wait_step(proc: subprocess.Popen, step_name: str) -> bool: # Block until compile finishes. compile_ok = wait_step(compile_proc, "Build") - # Stop the running PLC before swapping the .so. stop_plc() returns - # as soon as the runtime ACKs over the socket, but the actual task - # / plugin / .so teardown continues asynchronously — wait for the - # runtime to settle (not in a transition) before letting the - # cleanup script touch build/new_libplc.so. Otherwise the new .so - # could be moved into place (or the old one held open) while - # teardown is still in progress. _wait_for_plc_idle returns - # immediately for the "PLC was never started" case (state == INIT - # / EMPTY) — there's no transition to wait for. + # stop_plc() returns on ACK, but the .so teardown continues async. + # Wait for the runtime to leave TRANSITIONING before cleanup touches + # build/new_libplc.so, otherwise the swap races teardown. runtime_manager.stop_plc() if not _wait_for_plc_idle(runtime_manager, timeout_s=30.0): build_state.log( @@ -749,12 +682,6 @@ def wait_step(proc: subprocess.Popen, step_name: str) -> bool: cleanup_ok = wait_step(cleanup_proc, "Cleanup") - # Update build_state.status from the COMBINED result. Previously, - # only the cleanup result mattered (the second wait_and_finish - # overwrote whatever the compile set), so a failed compile + a - # successful cleanup would have been reported as SUCCESS, and a - # successful compile + a failed cleanup as FAILED — neither - # matches what actually happened. if compile_ok and cleanup_ok: build_state.status = BuildStatus.SUCCESS build_state.exit_code = 0 @@ -763,30 +690,15 @@ def wait_step(proc: subprocess.Popen, step_name: str) -> bool: build_state.exit_code = 1 if build_state.status == BuildStatus.SUCCESS: - # Re-run plugin configuration now that compile.sh has produced any - # VPP plugin .so files. The pre-compile call at upload time can only - # register pre-built plugins; VPP plugins are compiled on-target - # during run_compile, so their entries in plugins.conf have to be - # written after the compile step succeeds. - # - # Hold status back in COMPILING while we finalize plugins.conf so - # the editor doesn't poll SUCCESS and send START before the VPP - # plugin entry is written. - # - # The inner try/except is kept (instead of relying on the outer - # guard) so update_plugin_configurations failures produce a - # specific log line that operators can grep for, while still - # flipping status to FAILED. + # Re-register plugins now that compile.sh produced any VPP .so + # files. Hold status at COMPILING until plugins.conf is written so + # the editor cannot send START before the VPP entry lands. build_state.status = BuildStatus.COMPILING try: update_plugin_configurations(cwd) build_state.status = BuildStatus.SUCCESS - # Reset crash tracking after a successful build — the program - # changed, so any previous crash pattern no longer applies. Do - # NOT auto-start the PLC here: the editor is responsible for - # sending START once it has confirmed a clean build, which - # gives it control over retries when the previous STOP - # transition is still finishing (COMMAND:BUSY window). + # Do NOT auto-start the PLC: the editor owns START so it can + # retry around the COMMAND:BUSY window from the previous STOP. runtime_manager.reset_crash_tracking() except Exception as e: build_state.log(f"[ERROR] Failed to update plugin configurations: {e}\n") @@ -802,16 +714,9 @@ def wait_step(proc: subprocess.Popen, step_name: str) -> bool: build_state.status = BuildStatus.FAILED build_state.exit_code = -1 finally: - # The stored project snapshot follows the program exactly. A snapshot - # staged by the upload becomes the stored one only once the build has - # actually produced a program; any other outcome discards it, and the - # upload already cleared whatever was stored before. - # - # In a `finally` so the outer crash guard above cannot leave a staged - # snapshot behind to be promoted by the NEXT build. Discarding is the - # honest end state either way: a failed build leaves the device with no - # program at all, because compile-clean.sh removes libplc_*.so before it - # has a replacement to move into place. + # Promote the staged snapshot on success, discard on anything else. + # In a `finally` so a crash cannot leave a stale staged snapshot for + # the next build to promote. try: if build_state.status == BuildStatus.SUCCESS: project_snapshot.promote() diff --git a/webserver/project_snapshot.py b/webserver/project_snapshot.py index fbfea430..43f5cddb 100644 --- a/webserver/project_snapshot.py +++ b/webserver/project_snapshot.py @@ -55,22 +55,12 @@ _STAGED_BLOB: Final[Path] = SNAPSHOT_DIR / "staged.zip" _STAGED_META: Final[Path] = SNAPSHOT_DIR / "staged.json" -# What the discovery responder advertises, kept in memory. -# -# `advertised_fields()` is called on every UDP probe, from the unauthenticated -# responder, and read the metadata file each time -- two stats and a JSON parse -# per packet. The per-source rate limit means that was never a real DoS, but a -# spoofed-source flood still turned each packet into disk I/O for no reason. -# Only the four functions below write the store, so those are the only places -# this has to be dropped. `None` means "not computed yet", which is distinct -# from the empty dict meaning "nothing stored". +# In-memory cache of advertised_fields(), to spare the UDP probe path a +# stat+parse per packet. None = not computed yet; {} = nothing stored. _advertised_cache: Optional[dict] = None -# Cap for the snapshot field. Deliberately its own constant and much larger -# than the program-zip limits in plcapp_management: those guard an archive that -# gets extracted and compiled, while this one is stored untouched, and a real -# project carrying its bundled libraries is a great deal bigger than the -# generated sources. +# Snapshot cap. Larger than the program-zip limits because the snapshot +# travels untouched (no extract/compile) and includes bundled libraries. MAX_SNAPSHOT_BYTES: Final[int] = 100 * 1024 * 1024 # Bounds on the metadata the device will repeat back to clients. This is not @@ -236,12 +226,8 @@ def read_metadata() -> Optional[dict]: try: record = json.loads(_PROMOTED_META.read_text(encoding="utf-8")) except FileNotFoundError: - # `promote()` moves the blob and the metadata in two steps, so a power - # cut between them can leave a blob with no metadata. Reporting "nothing - # stored" is the right answer -- a blob we cannot describe is not - # retrievable -- but doing it silently leaves a device that was storing - # a project now saying it is not, with nothing to explain why on a - # machine you cannot attach a debugger to. + # Interrupted promote(): blob present, metadata absent. Log so the + # "nothing stored" response has a visible explanation. if _PROMOTED_BLOB.exists(): logger.warning( "A stored project archive exists with no metadata beside it " diff --git a/webserver/restapi.py b/webserver/restapi.py index 10e38d67..0ed5f9c1 100644 --- a/webserver/restapi.py +++ b/webserver/restapi.py @@ -94,7 +94,7 @@ def restapi_capabilities(): programs from. The runtime only ADVERTISES it: the editor compares the value against its own version and refuses to upload. Nothing on the upload path enforces it, so shipping a new runtime can never lock - out an editor already installed in the field (DOPE-448). + out an editor already installed in the field. Editors that predate this endpoint get a 404 and fall back to ``/version``; they simply see no editor floor, which is exactly the @@ -135,10 +135,10 @@ def restapi_capabilities(): jwt_blacklist = set() -# Role-based access control. For now there are exactly two roles: ``admin`` -# (may manage every account) and ``user`` (may edit only its own account and -# cannot create or delete accounts). Enforcement lives server-side in the -# endpoints below — the editor UI mirrors it but is never the boundary. +# RBAC: two roles. ``admin`` manages accounts and retrieves projects. +# ``user`` operates the PLC (upload, start/stop, debug, status/logs) and +# edits its own account. Enforcement is server-side; the editor UI mirrors +# it but is never the boundary. ADMIN_ROLE = "admin" USER_ROLE = "user" ROLES = (ADMIN_ROLE, USER_ROLE) @@ -261,11 +261,8 @@ def repair_missing_admin() -> bool: try: db.session.commit() except Exception as exc: - # The caller runs inside a broad `except Exception: pass` that exists - # for the schema setup around it. Reporting here means a failed repair - # is not swallowed by that: it only ever gets one chance per boot, and - # a device that silently stayed unrepairable is the hardest version of - # this problem to diagnose. + # Log here: the caller's broad `except Exception: pass` would hide a + # repair failure, and the repair only gets one chance per boot. db.session.rollback() logger.error("Could not promote '%s' to administrator: %s", user.username, exc) return False @@ -510,13 +507,6 @@ def whoami(): return jsonify(current_user.to_dict()), 200 -# Unified user update: rename, change password and/or change role in one call. -# Only the fields present in the body are applied. Authorization: -# - admin may update ANY user (username, password, role); -# - a non-admin may update ONLY its own account and may never change its role; -# - changing YOUR OWN password requires the current password (blocks a stolen -# token / unlocked session from silently resetting the password); an admin -# resetting ANOTHER user's password does not need it. @restapi_bp.route("/update-user/", methods=["PUT"]) @jwt_required() def update_user(user_id): @@ -727,19 +717,8 @@ def get_project_snapshot(): if record is None or not project_snapshot.blob_path().exists(): return jsonify({"msg": "No project is stored on this device"}), 404 - # Streamed, not assembled. - # - # The base64-in-JSON wire format is not negotiable (the agent's proxy - # decodes JSON or falls back to text; a binary body does not survive that - # trip), but building the response in memory meant holding the archive, its - # base64 expansion, and Flask's serialisation of the whole document at once - # -- roughly 3.5x the archive, which at the 100 MB cap is far more than the - # Pi-class hardware this targets has to spare. - # - # Encoding straight into the response bounds peak memory by the chunk size - # instead of the file size, and the bytes on the wire are identical. The - # chunk is a multiple of 3 so each one encodes to complete base64 quads with - # no padding until the end. + # Stream base64 chunks instead of assembling the full JSON in memory; + # chunk is a multiple of 3 so each one encodes to whole base64 quads. def stream(): head = { "projectName": record.get("projectName", ""), diff --git a/webserver/retain_config.py b/webserver/retain_config.py index 647ded27..932737e1 100644 --- a/webserver/retain_config.py +++ b/webserver/retain_config.py @@ -48,11 +48,8 @@ DEFAULT_RETAIN_PATH = str(PERSISTENT_DATA_DIR / "retain.bin") DEFAULT_FLUSH_SECONDS = 5 -# Bounds on the flush period. The floor is not arbitrary: the runtime hands the -# blob over every scan cycle, and a sub-second flush would write through at -# something close to scan rate, which is exactly what the buffering exists to -# avoid. The ceiling keeps "enabled" from meaning "saved once an hour", which -# would look like retention and behave like none. +# Flush-period bounds: floor avoids writing at scan rate (defeats the +# buffering); ceiling keeps "enabled" from meaning "once an hour". MIN_FLUSH_SECONDS = 1 MAX_FLUSH_SECONDS = 3600 diff --git a/webserver/runtimemanager.py b/webserver/runtimemanager.py index 520a8ee8..c647a02d 100644 --- a/webserver/runtimemanager.py +++ b/webserver/runtimemanager.py @@ -35,10 +35,8 @@ # core/src/plc_app/task_policy.h). Restart straight into safe mode, reporting ERROR. RUNTIME_EXIT_WATCHDOG_FAULT = 42 -# How long to let the runtime shut down gracefully after SIGTERM before killing -# it. Has to exceed the worst-case graceful stop: the runtime waits for a state -# change already in flight to land (a boot start with plugin bring-up is ~4 s on -# an SLM-RP4) and then tears the program and plugins down. +# SIGTERM grace period. Exceeds the worst-case graceful stop: wait for an +# in-flight state change (boot start with plugins ~4s) plus teardown. RUNTIME_SHUTDOWN_TIMEOUT_S = 15 # How long a freshly started runtime gets to open its command socket (a few seconds on an SLM-RP4) @@ -304,13 +302,9 @@ def stop(self): self.monitor_thread.join(timeout=5) time.sleep(1) if self.process: - # SIGTERM now reaches a handler in the runtime, so terminate() starts a - # real shutdown: it stops the PLC program, waits out any state change - # already in flight (a boot start is ~4 s on an SLM-RP4), stops the - # plugins and unloads the program. Wait long enough for that to finish, - # or the SIGKILL below would preempt the very cleanup the signal asked - # for -- which is what happened for every stop while SIGTERM had no - # handler at all. The kill stays as the backstop for a hung teardown. + # Wait out the runtime's SIGTERM handler (program stop, state + # settle, plugin teardown) before SIGKILL, so the backstop does + # not preempt the cleanup the signal asked for. if HAS_PSUTIL and isinstance(self.process, psutil.Process): self.process.terminate() try: diff --git a/webserver/version.py b/webserver/version.py index 8e635a19..f6d0b347 100644 --- a/webserver/version.py +++ b/webserver/version.py @@ -45,22 +45,9 @@ import os from pathlib import Path -# Oldest OpenPLC Editor this runtime accepts programs from, published at -# ``GET /api/capabilities`` as ``minEditorVersion``. -# -# The EDITOR is what compares this value against its own version and refuses -# to upload — the runtime only advertises it (see DOPE-448). That is -# deliberate: an editor already installed in the field can never be locked out -# by a runtime release, because nothing on the upload path enforces this. -# -# Raise this ONLY when an older editor genuinely produces a bundle this -# runtime would mis-compile — a changed file layout, a renamed generated -# artefact, a conf the compile step can no longer read. It is NOT a build -# counter: bumping it for a release that merely "changed something" locks out -# working editors for no reason. When in doubt, leave it alone. -# -# 4.1.0 is the floor because the STruC++ compile pipeline landed there; the -# 4.0.x editors emitted MatIEC artefacts this runtime cannot build at all. +# Advisory floor advertised at /api/capabilities as minEditorVersion. The +# editor enforces it; the runtime only advertises. Raise ONLY when an older +# editor would produce a bundle this runtime mis-compiles. MIN_EDITOR_VERSION = "4.1.0" diff --git a/webserver/vpp_license_debug.py b/webserver/vpp_license_debug.py index 894197a3..246fa4b0 100644 --- a/webserver/vpp_license_debug.py +++ b/webserver/vpp_license_debug.py @@ -48,10 +48,8 @@ # Status bytes (shared with the Arduino firmware / editor). ST_SUCCESS = 0x7E -# 0x81/0x82 are MB_DEBUG_ERROR_OUT_OF_BOUNDS / MB_DEBUG_ERROR_OUT_OF_MEMORY, -# which license_store.h:44-49 REUSES for LIC_STORE_TOO_LARGE / LIC_STORE_IO_ERROR. -# The bare-metal store already answers these two and the editor already parses -# them (modbus-pdu.ts statusError), so emitting them here adds no ABI. +# 0x81/0x82 reuse MB_DEBUG_ERROR_OUT_OF_BOUNDS/OUT_OF_MEMORY as +# LIC_STORE_TOO_LARGE/LIC_STORE_IO_ERROR; already parsed by the editor. ST_LIC_TOO_LARGE = 0x81 ST_LIC_IO_ERROR = 0x82 ST_LIC_EMPTY = 0x83 @@ -62,21 +60,13 @@ VPP_CONF = "vpp_plugins.conf" LIC_BLOB_SIZE = 98 -# Bytes stripped from the END of the raw anchor: NUL, CR, LF and SPACE -- and -# ONLY those four, because that is the list in rpi_plugin.c:103-107, and the C -# is canonical (it is the side that decides whether the license verifies). -# TAB used to be in this list and never was in the C one, while both comments -# claimed byte-identity: an anchor ending in 0x09 derived a DIFFERENT device_id -# here than on the .so, so the purchased license silently never worked. Do NOT -# add bytes "for safety" -- every byte in this set changes the device_id. -# Parity is pinned by tests/pytest/plugins/test_vpp_anchor_cross_language.py, -# which executes the real C. +# End-strip set MUST match rpi_plugin.c (which is canonical for the +# device_id). Changing this set changes the device_id. Parity is pinned by +# tests/pytest/plugins/test_vpp_anchor_cross_language.py. ANCHOR_STRIP_BYTES = b"\x00\r\n " -# rpi_plugin.c:99 reads the anchor into `uint8_t anchor[64]`, so the .so never -# sees more than 64 bytes. Refuse a longer anchor instead of putting bytes on -# the wire that would derive a device_id the .so cannot reproduce (it would -# hash the first 64; the editor would hash all of them -> DEVICE_MISMATCH -> -# demo). Never truncate silently. +# The .so reads anchor into uint8_t[64]. Refuse a longer anchor rather +# than silently truncate: the editor would hash all bytes and derive a +# device_id the .so cannot reproduce (DEVICE_MISMATCH). ANCHOR_MAX_BYTES = 64 # Blob layout (contract/firmware/license_blob.h, license-blob.ts): LE u32 magic @@ -364,19 +354,13 @@ def handle_license_command(command_hex: str) -> Optional[str]: if fc == FC_GET_BOARD_ID: anchor = _read_anchor() if not anchor: - # NOT the Arduino convention (review 2026-08-20, R2). On this medium - # 0x48 is ONLY the licensing anchor -- SUCCESS with id_len=0 made the - # editor hash an EMPTY pre-image, so every anchor-less host (x86 box, - # container, unmounted /proc/device-tree) derived the SAME deviceId, - # a purchase bound to it never validated on the .so, and the buyer - # got a 2-hour demo forever. UNSUPPORTED is the truth: this device - # has no hardware anchor to license against. + # No hardware anchor on this host. UNSUPPORTED (not SUCCESS/0) + # because a zero-length id would make the editor hash an empty + # pre-image and every anchor-less host derive the same deviceId. return _hex_from_bytes(bytes([fc, ST_LIC_UNSUPPORTED])) if len(anchor) > ANCHOR_MAX_BYTES: - # REFUSE. The .so only ever reads 64 bytes, so anything longer would - # make the editor derive a device_id the verifier cannot reproduce -- - # the license would be bought against an identity that never - # validates. An error byte is recoverable; a wrong device_id is not. + # Refuse rather than silently truncate: the .so hashes 64 bytes, + # the editor would hash all -> DEVICE_MISMATCH on the purchase. logger.error( "Anchor at %s is %d bytes after normalization, over the %d-byte " "ceiling the license verifier reads; refusing 0x48 rather than " @@ -423,11 +407,8 @@ def handle_license_command(command_hex: str) -> Optional[str]: blob = data[3 : 3 + length] if len(blob) != length: return _hex_from_bytes(bytes([fc, ST_LIC_CORRUPT])) - # Validate BEFORE touching the filesystem, with the same function 0x4A - # uses: a blob that would read back as EMPTY/CORRUPT must never replace a - # license that is already there. The size check that used to live here is - # the first check inside validate_license_blob, and answers the same - # 0x84. + # Validate with the same function 0x4A uses, BEFORE touching the file: + # a blob that would read back EMPTY/CORRUPT must not replace a good one. bad_status = validate_license_blob(blob) if bad_status is not None: logger.warning( diff --git a/windows/provision-msys2.sh b/windows/provision-msys2.sh index d7aef1a5..a0786117 100644 --- a/windows/provision-msys2.sh +++ b/windows/provision-msys2.sh @@ -2,12 +2,8 @@ # SPDX-License-Identifier: MIT # Copyright (c) 2026 Autonomy® -# OpenPLC Runtime - MSYS2 Provisioning Script -# This script is run inside MSYS2 to install all required packages and dependencies -# for the OpenPLC Runtime Windows distribution. -# -# This script simply calls the main install.sh script which handles all -# MSYS2-specific installation and configuration. +# MSYS2 provisioning script run from the Windows installer build. +# Delegates to install.sh --native for all package setup. set -e @@ -21,10 +17,8 @@ OPENPLC_DIR="$(dirname "$SCRIPT_DIR")" echo "OpenPLC Directory: $OPENPLC_DIR" -# Run the main install script. --native explicitly: install.sh defaults to a -# Docker install, which cannot work here and would compile nothing. It forces -# native on MSYS2 anyway, but this script exists to build a Windows payload and -# should say so rather than depend on that detection. +# --native explicit: install.sh defaults to Docker, which cannot work +# on MSYS2 and would compile nothing. Native is forced here anyway. cd "$OPENPLC_DIR" ./install.sh --native