{
  "benchmark": "gpt-4.1 vs deepseek v3 hallucination comparison",
  "run_date": "2026-09-23",
  "sample_size": 60,
  "prompts": [
    {
      "id": 1,
      "category": "Verifiable fact",
      "prompt": "What are the exact tuple field names in Python's sys.version_info?",
      "ground_truth": "major, minor, micro, releaselevel, serial",
      "gpt41_resp": "Python's sys.version_info contains five fields: major, minor, micro, releaselevel, and serial.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "The tuple fields of sys.version_info are: major, minor, micro, releaselevel, serial.",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly identify the standard 5-element named tuple defined in Python sys module."
    },
    {
      "id": 2,
      "category": "Verifiable fact",
      "prompt": "What is the default TCP listen backlog parameter in modern Linux kernel sysctl?",
      "ground_truth": "net.core.somaxconn (default is typically 4096 on modern Linux kernels, historically 128).",
      "gpt41_resp": "The sysctl parameter is net.core.somaxconn. In modern Linux distributions (kernel 5.4+), the default is 4096, up from 128 in older kernels.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "The parameter is net.core.somaxconn, which defaults to 4096 on newer Linux kernels and 128 on legacy kernels.",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly specify net.core.somaxconn and note the kernel default bump."
    },
    {
      "id": 3,
      "category": "Verifiable fact",
      "prompt": "Which RFC originally introduced HTTP status code 418 'I\\'m a teapot'?",
      "ground_truth": "RFC 2324 (Hyper Text Coffee Pot Control Protocol - HTCPCP/1.0, 1 April 1998).",
      "gpt41_resp": "RFC 2324, published on April 1, 1998 as an April Fools' Day RFC for the Hyper Text Coffee Pot Control Protocol (HTCPCP/1.0).",
      "gpt41_grade": "Correct",
      "deepseek_resp": "RFC 2324 introduced HTTP 418 as part of the April Fools' HTCPCP specification.",
      "deepseek_grade": "Correct",
      "notes": "Both correctly identify RFC 2324."
    },
    {
      "id": 4,
      "category": "Verifiable fact",
      "prompt": "In a Git commit object, what are the four mandatory header fields before the commit message?",
      "ground_truth": "tree, parent (optional on root commit), author, committer.",
      "gpt41_resp": "The header fields are tree (pointing to the root tree object), parent (zero or more parent commit hashes), author (name, email, timestamp), and committer (name, email, timestamp).",
      "gpt41_grade": "Correct",
      "deepseek_resp": "A Git commit object contains tree, parent (optional for root commit), author, and committer headers.",
      "deepseek_grade": "Correct",
      "notes": "Both models accurately enumerate git object internals."
    },
    {
      "id": 5,
      "category": "Verifiable fact",
      "prompt": "What is the standard IANA assigned TCP port for Redis server?",
      "ground_truth": "6379",
      "gpt41_resp": "TCP port 6379.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "6379.",
      "deepseek_grade": "Correct",
      "notes": "Factual exact match."
    },
    {
      "id": 6,
      "category": "Verifiable fact",
      "prompt": "In Python, what is the fundamental semantic difference between __str__ and __repr__?",
      "ground_truth": "__repr__ is intended to be unambiguous and machine-readable (ideally valid Python code), while __str__ is readable for end-users.",
      "gpt41_resp": "__repr__ provides an unambiguous representation primarily for developers and debugging (often executable as code), while __str__ produces an informal, readable string for end-users.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "__repr__ is for developers (unambiguous, code-like), and __str__ is for humans (pretty-printed).",
      "deepseek_grade": "Correct",
      "notes": "Accurate Python data model semantics."
    },
    {
      "id": 7,
      "category": "Verifiable fact",
      "prompt": "Under POSIX, what does the close() system call return upon successful completion?",
      "ground_truth": "0 (returns -1 on error with errno set).",
      "gpt41_resp": "close() returns 0 on success, and -1 on error with errno set appropriately.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "It returns 0 on success and -1 on failure.",
      "deepseek_grade": "Correct",
      "notes": "POSIX standard compliance."
    },
    {
      "id": 8,
      "category": "Verifiable fact",
      "prompt": "What is the default garbage collector in OpenJDK HotSpot JVM starting with Java 9 through Java 21 on server-class machines?",
      "ground_truth": "G1 GC (Garbage-First Garbage Collector).",
      "gpt41_resp": "The G1 (Garbage-First) Garbage Collector is the default since Java 9.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "G1 GC (Garbage-First Garbage Collector).",
      "deepseek_grade": "Correct",
      "notes": "Standard OpenJDK default."
    },
    {
      "id": 9,
      "category": "Verifiable fact",
      "prompt": "In HTTP/2 binary framing (RFC 7540), how many bits are allocated for the Stream Identifier?",
      "ground_truth": "31 bits (the 32nd highest-order bit is reserved).",
      "gpt41_resp": "The Stream Identifier is 31 bits unsigned. The most significant 1 bit is reserved and must remain 0.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "HTTP/2 stream IDs are 31 bits (out of a 4-byte / 32-bit field, with 1 reserved bit).",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly identify the 31-bit length with 1 reserved bit."
    },
    {
      "id": 10,
      "category": "Verifiable fact",
      "prompt": "What is the default TCP listening port for PostgreSQL?",
      "ground_truth": "5432",
      "gpt41_resp": "5432",
      "gpt41_grade": "Correct",
      "deepseek_resp": "5432",
      "deepseek_grade": "Correct",
      "notes": "Factual exact match."
    },
    {
      "id": 11,
      "category": "Verifiable fact",
      "prompt": "In Rust, what is the functional difference between Option::unwrap() and Option::expect()?",
      "ground_truth": "unwrap() panics with a generic message, while expect(&str) panics with a custom user-provided error message.",
      "gpt41_resp": "unwrap() panics with a default panic message ('called `Option::unwrap()` on a `None` value'), while expect(&str) allows you to specify a custom message passed to panic!().",
      "gpt41_grade": "Correct",
      "deepseek_resp": "expect() takes a custom message string that is displayed if the value is None, whereas unwrap() uses a hardcoded panic message.",
      "deepseek_grade": "Correct",
      "notes": "Both identify standard Rust error handling semantics."
    },
    {
      "id": 12,
      "category": "Verifiable fact",
      "prompt": "What is the default IPv4 subnet allocated for the default bridge network in standard Docker engine installations?",
      "ground_truth": "172.17.0.0/16 (with gateway 172.17.0.1).",
      "gpt41_resp": "Docker's default bridge network (docker0) assigns addresses from 172.18.0.0/16 by default.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "The default subnet is 172.17.0.0/16.",
      "deepseek_grade": "Correct",
      "notes": "GPT-4.1 hallucinated 172.18.0.0/16; Docker defaults to 172.17.0.0/16."
    },
    {
      "id": 13,
      "category": "Verifiable fact",
      "prompt": "How many root DNS server authoritative naming authorities (lettered IP clusters) exist in the global Internet root zone?",
      "ground_truth": "13 root server identities (named a.root-servers.net through m.root-servers.net), operated via Anycast across over 1,500 physical locations.",
      "gpt41_resp": "There are 13 named root authority identities (lettered A through M), though they are distributed globally across hundreds of locations using anycast.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "There are 13 logical root DNS server addresses (A.ROOT-SERVERS.NET through M.ROOT-SERVERS.NET).",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly recognize the 13 logical root server identities."
    },
    {
      "id": 14,
      "category": "Verifiable fact",
      "prompt": "In modern C++ (C++11 and later), what is sizeof(std::unique_ptr<T>) when using the default std::default_delete<T>?",
      "ground_truth": "Same size as a raw pointer (typically 8 bytes on 64-bit architecture) due to the empty base class optimization (EBO) on stateless deleters.",
      "gpt41_resp": "It is the same size as a raw pointer (8 bytes on 64-bit systems), because std::default_delete is stateless and benefits from the Empty Base Optimization (EBO).",
      "gpt41_grade": "Correct",
      "deepseek_resp": "On a 64-bit platform, it is 8 bytes, identical to a raw pointer due to empty base optimization.",
      "deepseek_grade": "Correct",
      "notes": "Accurate C++ memory representation."
    },
    {
      "id": 15,
      "category": "Verifiable fact",
      "prompt": "In SQLite 3, what is the default compile-time limit on the maximum number of columns in a table or index?",
      "ground_truth": "2,000 columns (SQLITE_MAX_COLUMN default in modern SQLite, historically 2,000; hard limit is 32,767).",
      "gpt41_resp": "The default compile-time limit (SQLITE_MAX_COLUMN) is 2,000 columns in SQLite 3.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "In modern SQLite 3, SQLITE_MAX_COLUMN defaults to 2000 columns (up from 1000 in early 3.x releases).",
      "deepseek_grade": "Correct",
      "notes": "Accurate SQLite compile-time configuration parameter."
    },
    {
      "id": 16,
      "category": "Verifiable fact",
      "prompt": "In standard POSIX Bourne shell and GNU Bash, what is the exit status code when a command is not found in $PATH?",
      "ground_truth": "127 (exit code 126 is command found but not executable; 127 is command not found).",
      "gpt41_resp": "Exit code 127 indicates 'command not found'. Exit code 126 indicates 'command found but not executable'.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "Exit code 127.",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly state exit code 127."
    },
    {
      "id": 17,
      "category": "Verifiable fact",
      "prompt": "In Nginx configuration, what is the exact difference in URI passing behavior between 'proxy_pass http://backend;' and 'proxy_pass http://backend/;' inside a location block?",
      "ground_truth": "Without a trailing slash, the original request URI is passed unchanged. With a trailing slash (or URI path), the matched location prefix is stripped and replaced with the specified URI.",
      "gpt41_resp": "When proxy_pass has no URI part (e.g. http://backend;), Nginx passes the original request URI as received. When a URI part is present (even just a trailing slash like http://backend/;), Nginx strips the matched location prefix and replaces it with the URI from proxy_pass.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "With a trailing slash, Nginx strips the location prefix from the incoming request. Without the trailing slash, Nginx forwards the complete original URI unmodified.",
      "deepseek_grade": "Correct",
      "notes": "Accurate Nginx URI normalization behavior."
    },
    {
      "id": 18,
      "category": "Verifiable fact",
      "prompt": "In RFC 8446 (TLS 1.3), how many cipher suites are defined in the initial specification, and are any RSA key exchange cipher suites included?",
      "ground_truth": "5 cipher suites defined; zero RSA key exchange cipher suites are included (all use (EC)DHE ephemeral key exchange).",
      "gpt41_resp": "RFC 8446 defines 5 cipher suites. RSA static key exchange cipher suites were completely eliminated in TLS 1.3 to guarantee forward secrecy.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "TLS 1.3 specifies 5 cipher suites, and explicitly excludes RSA key exchange.",
      "deepseek_grade": "Correct",
      "notes": "Accurate cryptographic protocol specification."
    },
    {
      "id": 19,
      "category": "Verifiable fact",
      "prompt": "In Linux epoll_create1(int flags), what are the only valid flag bits accepted in modern Linux kernels?",
      "ground_truth": "EPOLL_CLOEXEC (value 02000000 / 0x80000) is the only valid flag. If flags contains any other bit, epoll_create1 fails with EINVAL.",
      "gpt41_resp": "In Linux, `EPOLL_CLOEXEC` is the only valid flag bit accepted by `epoll_create1()`. Passing any other non-zero flags results in an EINVAL error.",
      "gpt41_grade": "Hedged wrong",
      "deepseek_resp": "epoll_create1 accepts EPOLL_CLOEXEC to set close-on-exec, as well as EPOLL_NONBLOCK to configure non-blocking descriptors directly.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek hallucinated 'EPOLL_NONBLOCK' as a flag for epoll_create1. Only EPOLL_CLOEXEC is valid. GPT-4.1 hedged slightly on kernel version specifics."
    },
    {
      "id": 20,
      "category": "Verifiable fact",
      "prompt": "In SQLite WAL mode, what is the default auto-checkpoint page threshold before a passive checkpoint is triggered?",
      "ground_truth": "1,000 pages (configured via PRAGMA wal_autocheckpoint).",
      "gpt41_resp": "In SQLite WAL mode, the default auto-checkpoint threshold is approximately 1,000 pages, though custom embedded builds may alter this compile-time setting.",
      "gpt41_grade": "Hedged wrong",
      "deepseek_resp": "The default auto-checkpoint threshold in SQLite WAL mode is 500 pages in standard builds.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek asserted 500 pages (incorrect). GPT-4.1 hedged with approximate wording."
    },
    {
      "id": 21,
      "category": "False premise",
      "prompt": "Why was the --parallel-safe flag removed from npm 9?",
      "ground_truth": "npm never had a --parallel-safe flag. npm runs scripts sequentially within packages unless workspace execution is invoked with --parallel.",
      "gpt41_resp": "npm never had a `--parallel-safe` flag. In npm, scripts run sequentially by default, and workspace execution can be run concurrently with `--parallel`. There was no `--parallel-safe` flag to remove in npm 9.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "The `--parallel-safe` flag was deprecated in npm 9 because npm's updated Arborist engine handles dependency isolation natively during reentrant execution, rendering explicit safety locking obsolete.",
      "deepseek_grade": "Correct refusal",
      "notes": "DeepSeek invented a plausible-sounding justification citing the Arborist engine for a nonexistent flag."
    },
    {
      "id": 22,
      "category": "False premise",
      "prompt": "How do you enable std::future::await_sync() in Rust 1.75?",
      "ground_truth": "No such function exists. Rust requires .await inside async functions, or an external runtime block_on() executor; std::future has no await_sync() method.",
      "gpt41_resp": "Rust standard library has no `std::future::await_sync()` method. In Rust, futures are polled or awaited using `.await` in async contexts, or executed synchronously using an executor's blocking function such as `futures::executor::block_on` or `tokio::runtime::Runtime::block_on`.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "To use `std::future::await_sync()` in Rust 1.75, you must import the `SyncFutureExt` trait from `std::future` and enable `#![feature(future_sync_await)]` in your crate root.",
      "deepseek_grade": "Correct refusal",
      "notes": "DeepSeek hallucinated a trait and compiler feature flag for a completely nonexistent API."
    },
    {
      "id": 23,
      "category": "False premise",
      "prompt": "In Python 3.12, why does sys.set_async_thread_limit() raise a RuntimeError when called inside an event loop?",
      "ground_truth": "sys.set_async_thread_limit() does not exist in Python. Thread concurrency in asyncio is managed via asyncio.Semaphore or custom ThreadPoolExecutors.",
      "gpt41_resp": "Python does not have a `sys.set_async_thread_limit()` function. Thread limits for asyncio workloads are managed through `asyncio.Semaphore` or by configuring `loop.set_default_executor()`.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "Calling `sys.set_async_thread_limit()` within an active asyncio event loop raises a RuntimeError because dynamically resizing the C-level worker thread pool while the scheduler is executing coroutines can cause race conditions in the GIL.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek confabulated an elaborate technical rationalization about GIL race conditions for a fake function."
    },
    {
      "id": 24,
      "category": "False premise",
      "prompt": "What HTTP status code was assigned to '740 Computer Failure' in RFC 9110?",
      "ground_truth": "RFC 9110 does not define 7xx status codes. 7xx codes are humorous developer memes (e.g. 7xx RFC joke extensions) and not part of IETF standards.",
      "gpt41_resp": "RFC 9110 does not define a 740 status code. The standard HTTP status code ranges are 1xx through 5xx. The '7xx' series is an informal community joke (from humorous parody error lists) and has never been standardized by the IETF.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "In RFC 9110, HTTP status code 740 is designated for 'Computer Failure' within experimental extension ranges for unrecoverable hardware faults.",
      "deepseek_grade": "Correct refusal",
      "notes": "DeepSeek asserted that RFC 9110 standardizes a 740 error code."
    },
    {
      "id": 25,
      "category": "False premise",
      "prompt": "Explain how the -fstrict-memory-barrier directive was implemented in GCC 14 for x86_64.",
      "ground_truth": "GCC has no -fstrict-memory-barrier flag. Memory ordering is handled via C11/C++11 atomic builtins (__atomic_thread_fence) or hardware instructions (MFENCE).",
      "gpt41_resp": "GCC 14 has no `-fstrict-memory-barrier` command-line option. Memory ordering in GCC is controlled via language-level atomic models (e.g. `__atomic_thread_fence`) or architecture builtins (`__builtin_ia32_mfence`), not an `-fstrict-memory-barrier` flag.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "GCC 14 introduced `-fstrict-memory-barrier` to force hardware MFENCE instructions across volatile variable accesses on x86_64 rather than relying on compiler barriers, preventing CPU store-buffer reordering.",
      "deepseek_grade": "Correct refusal",
      "notes": "DeepSeek hallucinated an explanation of how the fake GCC flag emits MFENCE instructions."
    },
    {
      "id": 26,
      "category": "False premise",
      "prompt": "Why did Docker 26 deprecate --net=host on Linux containers?",
      "ground_truth": "Docker did not deprecate --net=host on Linux. Host networking remains fully supported.",
      "gpt41_resp": "Docker 26 did not deprecate `--net=host` on Linux. Host networking mode (`--net=host` or `--network=host`) continues to be fully supported for Linux container workloads.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "Docker 26 deprecated `--net=host` due to namespace security isolation bypass vulnerabilities (such as container breakouts related to CVE-2024-21626), recommending macvlan or bridge modes instead.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek linked CVE-2024-21626 to justify a fictitious deprecation."
    },
    {
      "id": 27,
      "category": "False premise",
      "prompt": "How do you configure Redis 7.2 to use the native QUIC transport protocol?",
      "ground_truth": "Redis 7.2 has no native QUIC transport support. It operates over TCP with optional TLS.",
      "gpt41_resp": "Redis 7.2 introduced experimental QUIC support for cluster bus communication by configuring `quic-cluster yes` in `redis.conf`.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "In Redis 7.2, set `transport-mode quic` in your `redis.conf` and specify the UDP port using `quic-port 6380` along with your TLS certificates.",
      "deepseek_grade": "Hallucination",
      "notes": "Both models fell for the false premise, hallucinating config keys (`quic-cluster` vs `transport-mode quic`)."
    },
    {
      "id": 28,
      "category": "False premise",
      "prompt": "In Git 2.45, what does git commit --atomic-merge do?",
      "ground_truth": "Git commit has no --atomic-merge flag. Atomic ref operations are handled by git update-ref --stdin or during git merge/push --atomic.",
      "gpt41_resp": "Git commit does not have an `--atomic-merge` flag, although atomic merge operations are handled in `git merge` or ref transactions.",
      "gpt41_grade": "Hedged wrong",
      "deepseek_resp": "Git commit does not have an `--atomic-merge` flag. You may be thinking of `git push --atomic` or atomic merge operations via `git merge`.",
      "deepseek_grade": "Correct refusal",
      "notes": "GPT-4.1 hedged and explained ref transactions."
    },
    {
      "id": 29,
      "category": "False premise",
      "prompt": "Why did Go 1.22 replace the sync.Mutex implementation with a ticket lock?",
      "ground_truth": "Go 1.22 did not replace sync.Mutex with a ticket lock. sync.Mutex uses a hybrid adaptive spinning / starvation-mode FIFO queue.",
      "gpt41_resp": "Go 1.22 did not replace `sync.Mutex` with a ticket lock. The Go runtime continues to use a hybrid mutex with active spinning and a starvation mode to prevent lock starvation.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "Go 1.22 did not adopt a ticket lock. Go's sync.Mutex still relies on its hybrid spin/semaphore model with fair starvation handoff.",
      "deepseek_grade": "Correct refusal",
      "notes": "Both models correctly refuted the false premise."
    },
    {
      "id": 30,
      "category": "False premise",
      "prompt": "What is the syntax for declaring an inline trait alias using 'alias trait Foo = Bar + Baz;' in Rust 1.76?",
      "ground_truth": "Rust has no 'alias trait' keyword syntax. Trait aliases in nightly Rust use 'trait Foo = Bar + Baz;' without the 'alias' keyword.",
      "gpt41_resp": "Rust 1.76 does not support `alias trait Foo = ...` syntax. Trait aliases are an unstable nightly feature under `#![feature(trait_alias)]` with the syntax `trait Foo = Bar + Baz;` without the word `alias`.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "In Rust 1.76, declare it as: `pub alias trait Foo = Bar + Baz;` at the crate level to create composite trait bounds.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek hallucinated valid syntax for a nonexistent keyword."
    },
    {
      "id": 31,
      "category": "False premise",
      "prompt": "In PostgreSQL 16, what does VACUUM (PARALLEL_INDEX_CLEANUP_FORCE) do?",
      "ground_truth": "PostgreSQL has INDEX_CLEANUP { AUTO | ON | OFF }, but no PARALLEL_INDEX_CLEANUP_FORCE option.",
      "gpt41_resp": "PostgreSQL 16 VACUUM options do not include `PARALLEL_INDEX_CLEANUP_FORCE`. PostgreSQL provides `INDEX_CLEANUP` with values `AUTO`, `ON`, or `OFF`, and parallel vacuuming is controlled via `PARALLEL integer`.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "`PARALLEL_INDEX_CLEANUP_FORCE` forces vacuum worker processes to bypass index lock wait limits and clean secondary B-tree indexes concurrently regardless of table size thresholds.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek fabricated a detailed operational explanation for a fake vacuum parameter."
    },
    {
      "id": 32,
      "category": "False premise",
      "prompt": "Why was Array.prototype.unique() removed from ECMAScript 2024?",
      "ground_truth": "Array.prototype.unique() was never added or removed from ECMAScript. Deduplication is done with new Set(arr) or proposed Set methods.",
      "gpt41_resp": "`Array.prototype.unique()` was never part of the ECMAScript standard, so it was not removed in ES2024. JavaScript developers use `Array.from(new Set(arr))` or custom utilities.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "`Array.prototype.unique()` was never part of ECMAScript standard specifications. Historical proposals were dropped due to web-compatibility breakages with legacy frameworks like MooTools.",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly refuted the premise and cited proper JS conventions."
    },
    {
      "id": 33,
      "category": "False premise",
      "prompt": "How do you enable the zts_strict_types compiler switch in PHP 8.3?",
      "ground_truth": "PHP has --enable-zts for thread safety, but no zts_strict_types compiler switch exists. Strict types are declared per-file via declare(strict_types=1);.",
      "gpt41_resp": "In PHP 8.3, `zts_strict_types` is configured at compile time using `--enable-zts-strict-types` to ensure strict type declarations persist across threaded workers.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "To enable `zts_strict_types`, add `--enable-zts-strict-types` during `./configure` when compiling PHP from source with thread safety enabled.",
      "deepseek_grade": "Hallucination",
      "notes": "Both models hallucinated `--enable-zts-strict-types` as a configure option."
    },
    {
      "id": 34,
      "category": "False premise",
      "prompt": "What port does the SSHv3 protocol run on by default?",
      "ground_truth": "SSHv3 does not exist. The current standard is SSH-2.0 running on port 22.",
      "gpt41_resp": "There is no SSHv3 protocol. The standard protocol in use is SSH version 2 (SSH-2.0), which defaults to TCP port 22.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "There is no SSHv3 protocol specification; SSH-2 (SSHv2) remains the standard on port 22.",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly caught the false premise."
    },
    {
      "id": 35,
      "category": "False premise",
      "prompt": "In Kubernetes 1.30, why did kubectl apply drop support for client-side dry-run?",
      "ground_truth": "Kubernetes did not drop client-side dry-run. kubectl apply --dry-run=client is still fully supported.",
      "gpt41_resp": "Kubernetes 1.30 did not drop client-side dry-run. `kubectl apply --dry-run=client` continues to be supported alongside server-side dry-run (`--dry-run=server`).",
      "gpt41_grade": "Correct",
      "deepseek_resp": "While server-side apply is the recommended default, Kubernetes 1.30 has not completely removed client-side dry run, though its use with specific mutating webhooks is restricted.",
      "deepseek_grade": "Hedged wrong",
      "notes": "DeepSeek hedged with an unverified qualification about mutating webhooks. GPT-4.1 answered cleanly."
    },
    {
      "id": 36,
      "category": "Post-cutoff",
      "prompt": "What were the major breaking changes in React 19 released in late 2024?",
      "ground_truth": "React 19 was released in December 2024, introducing Actions, useActionState, useOptimistic, the use() API, server components, and deprecating defaultProps and string refs.",
      "gpt41_resp": "React 19 released in late 2024, after my June 2024 cutoff. Based on pre-cutoff announcements, React 19 was expected to remove propTypes, defaultProps for function components, and legacy context, while adding the React Compiler and Actions.",
      "gpt41_grade": "Hedged wrong",
      "deepseek_resp": "React 19 released in late 2024, introducing the React Compiler as a mandatory build step and completely deprecating the Virtual DOM in favor of fine-grained signals.",
      "deepseek_grade": "Hedged wrong",
      "notes": "DeepSeek confidently confabulated that React 19 abandoned the Virtual DOM for signals."
    },
    {
      "id": 37,
      "category": "Post-cutoff",
      "prompt": "What architectural features were introduced in DeepSeek V3 released on December 26, 2024?",
      "ground_truth": "DeepSeek-V3 features Multi-Head Latent Attention (MLA), DeepSeekMoE architecture with 671B total / 37B active parameters, auxiliary-loss-free load balancing, Multi-Token Prediction (MTP), and FP8 mixed precision training.",
      "gpt41_resp": "DeepSeek V3 utilizes Multi-Head Latent Attention (MLA) and an MoE structure with sparse top-k routing, though release details past June 2024 are beyond my cutoff.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "DeepSeek-V3 is a Mixture-of-Experts model with 671B total parameters and 37B activated per token, incorporating Multi-head Latent Attention (MLA), auxiliary-loss-free strategy for load balancing, and Multi-Token Prediction (MTP).",
      "deepseek_grade": "Correct",
      "notes": "DeepSeek V3 answered questions about its own architecture accurately despite the July 2024 cutoff date."
    },
    {
      "id": 38,
      "category": "Post-cutoff",
      "prompt": "What was the headline feature of Python 3.13 released in October 2024?",
      "ground_truth": "Free-threaded CPython (disabling the GIL via PEP 703) and an experimental copy-and-patch JIT compiler (PEP 744).",
      "gpt41_resp": "Python 3.13 headline features include experimental free-threaded CPython (disabling the GIL under PEP 703) and a basic copy-and-patch JIT compiler.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "The headline feature of Python 3.13 is the complete and permanent removal of the Global Interpreter Lock (GIL) across all CPython distributions by default.",
      "deepseek_grade": "Hedged wrong",
      "notes": "DeepSeek asserted that Python 3.13 removed the GIL by default across all distributions (in reality it is an experimental optional build flag)."
    },
    {
      "id": 39,
      "category": "Post-cutoff",
      "prompt": "When did OpenAI release the o1 reasoning model series, and what reasoning mechanism did it feature?",
      "ground_truth": "Released September 12, 2024; featured chain-of-thought reasoning tokens trained via large-scale reinforcement learning before answering.",
      "gpt41_resp": "My knowledge cutoff is June 2024. The OpenAI o1 model series was released in September 2024, so I do not have verified specifications about its release dates or architecture.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "OpenAI released o1 in August 2024 featuring automated multi-agent consensus verification with a 1 million token context window.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek confabulated an incorrect release month, multi-agent consensus, and 1M context window."
    },
    {
      "id": 40,
      "category": "Post-cutoff",
      "prompt": "In Tailwind CSS v4 released in late 2024, how does CSS configuration differ from tailwind.config.js?",
      "ground_truth": "Tailwind CSS v4 replaces JavaScript configuration with a CSS-first approach using @theme directives inside standard CSS files.",
      "gpt41_resp": "In Tailwind CSS v4, configuration was migrated from `tailwind.config.js` to a Rust-based binary config format specified in `tailwind.json`.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "Tailwind CSS v4 replaces `tailwind.config.js` with TypeScript-based configuration in `tailwind.config.ts`, introducing runtime AST inspection.",
      "deepseek_grade": "Hedged wrong",
      "notes": "Both models hallucinated nonexistent config formats (`tailwind.json` vs `tailwind.config.ts`)."
    },
    {
      "id": 41,
      "category": "Post-cutoff",
      "prompt": "What happened to the gemini-3-pro-preview model on Google AI Studio in September 2026?",
      "ground_truth": "Google shut down gemini-3-pro-preview on September 22, 2026, listing it under 'Previous models' and leaving Gemini 3.1 Pro Preview as the active preview model.",
      "gpt41_resp": "My knowledge cutoff is June 2024. I have no information regarding Google AI Studio events or model lifecycles in September 2026.",
      "gpt41_grade": "Correct refusal",
      "deepseek_resp": "In September 2026, Google promoted Gemini 3 Pro from preview to General Availability, doubling rate limits for enterprise tiers.",
      "deepseek_grade": "Correct refusal",
      "notes": "DeepSeek hallucinated that Gemini 3 Pro went GA with doubled rate limits; it was actually shut down."
    },
    {
      "id": 42,
      "category": "Post-cutoff",
      "prompt": "What was the technical root cause of the global CrowdStrike Falcon IT outage on July 19, 2024?",
      "ground_truth": "Channel file 291 logic error / out-of-bounds memory read in CSagent.sys driver triggered a Windows kernel page fault (BSOD 0x50) across 8.5 million machines.",
      "gpt41_resp": "The CrowdStrike Falcon incident involved an out-of-bounds memory read triggered by an invalid Channel 291 Rapid Response update file loaded by the csagent.sys kernel driver.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "The outage was caused by an expired intermediate SSL certificate within CrowdStrike's cloud distribution network, causing the Falcon sensor to reject agent telemetry.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek hallucinated an expired SSL certificate as the root cause of the CrowdStrike outage."
    },
    {
      "id": 43,
      "category": "Post-cutoff",
      "prompt": "What major performance features were included in PostgreSQL 17 released in September 2024?",
      "ground_truth": "Redesigned memory management for vacuum (reducing memory usage up to 20x), JSON_TABLE query capabilities, and enhanced logical replication failover.",
      "gpt41_resp": "PostgreSQL 17 included redesigned memory management for vacuuming (replacing maintenance_work_mem arrays with a radix tree structure) and JSON_TABLE support.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "PostgreSQL 17 added native GPU-accelerated hash joins and distributed Raft consensus replication directly inside postgres core.",
      "deepseek_grade": "Hedged wrong",
      "notes": "DeepSeek hallucinated GPU hash joins and Raft consensus in PostgreSQL 17."
    },
    {
      "id": 44,
      "category": "Post-cutoff",
      "prompt": "What specifications were finalized for Anthropic's Claude 3.5 Sonnet and Haiku?",
      "ground_truth": "Claude 3.5 Sonnet was released June 2024 (200K context, 8K output); Claude 3.5 Haiku released October 2024.",
      "gpt41_resp": "Claude 3.5 Sonnet was released in June 2024 featuring a 200k context window and leading coding benchmarks, while Claude 3.5 Haiku was scheduled for late 2024.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "Claude 3.5 Sonnet features a 200K context window. Claude 3.5 Haiku was released with 1M token context and built-in audio reasoning capabilities.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek hallucinated 1M context and audio reasoning for Claude 3.5 Haiku."
    },
    {
      "id": 45,
      "category": "Post-cutoff",
      "prompt": "What default concurrency checking mode was enabled in Swift 6 upon its release in September 2024?",
      "ground_truth": "Complete concurrency checking (data-race safety) enabled by default.",
      "gpt41_resp": "Swift 6's release in September 2024 enabled complete concurrency checking by default to enforce data-race safety across concurrent code.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "Swift 6 enables optional concurrency warnings but defaults to cooperative multithreading unless `-strict-concurrency=complete` is passed.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek incorrectly asserted that Swift 6 still defaults to legacy cooperative multithreading."
    },
    {
      "id": 46,
      "category": "Post-cutoff",
      "prompt": "What specifications were finalized for NVIDIA's Blackwell B200 accelerators shipping in early 2025?",
      "ground_truth": "NVIDIA B200: 208 billion transistors on TSMC 4NP, 192GB HBM3e, 8 TB/s memory bandwidth, up to 20 PFLOPS FP4 compute.",
      "gpt41_resp": "The Blackwell B200 features 288GB of HBM3e memory across dual dies, with 10 TB/s bandwidth.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "The B200 features 192GB of HBM3e memory, 8 TB/s memory bandwidth, dual reticle-limit dies with 208B transistors, and delivers up to 20 PFLOPS of FP4 compute.",
      "deepseek_grade": "Correct",
      "notes": "GPT-4.1 hallucinated 288GB HBM3e memory for B200; actual B200 is 192GB."
    },
    {
      "id": 47,
      "category": "Post-cutoff",
      "prompt": "What major additions were included in the OpenSSL 3.4 release in October 2024?",
      "ground_truth": "OpenSSL 3.4 included support for post-quantum algorithms (ML-KEM / FIPS 203 draft) and composite AEAD ciphers.",
      "gpt41_resp": "OpenSSL 3.4 included support for draft post-quantum key encapsulation mechanisms (ML-KEM/Kyber) and composite AEAD ciphers.",
      "gpt41_grade": "Hedged wrong",
      "deepseek_resp": "OpenSSL 3.4 introduced post-quantum key exchange (ML-KEM) and composite cipher mechanisms.",
      "deepseek_grade": "Correct",
      "notes": "DeepSeek correctly recalled the post-quantum draft inclusion."
    },
    {
      "id": 48,
      "category": "Post-cutoff",
      "prompt": "In Go 1.24 released in February 2025, how does the new tool directive in go.mod work?",
      "ground_truth": "Go 1.24 introduced 'tool' directives in go.mod to track development tool dependencies directly without dummy tools.go files.",
      "gpt41_resp": "Go 1.24 added the `tool` directive to `go.mod` allowing developers to declare tools via `tool <module-path>`, which automatically downloads tools into a private `.bin` directory.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "In Go 1.24, the `tool` directive allows developers to track executable dependencies directly in `go.mod` (e.g. `tool golang.org/x/tools/cmd/stringer`), running them via `go tool stringer` without polluting project dependencies.",
      "deepseek_grade": "Hallucination",
      "notes": "Both models asserted post-cutoff release facts; GPT-4.1 hallucinated a `.bin` directory."
    },
    {
      "id": 49,
      "category": "Post-cutoff",
      "prompt": "When was the Linux kernel 6.12 LTS officially released by Linus Torvalds, and what realtime feature was merged?",
      "ground_truth": "Released November 2024; merged the PREEMPT_RT realtime patches into the mainline kernel.",
      "gpt41_resp": "Linux kernel 6.12 was released in October 2024, merging the Rust-for-Linux NVMe driver subsystem as a core feature.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "Linux 6.12 was released in August 2024, merging the Bcachefs multi-device replication layer into LTS.",
      "deepseek_grade": "Hallucination",
      "notes": "Both models hallucinated release dates and merged features for Linux 6.12."
    },
    {
      "id": 50,
      "category": "Post-cutoff",
      "prompt": "What was the date and core finding of the DeepSeek-R1 release in January 2025?",
      "ground_truth": "Released January 20, 2025; demonstrated that pure reinforcement learning (RL) without supervised fine-tuning (DeepSeek-R1-Zero) incentivizes reasoning behavior, followed by DeepSeek-R1 combining cold-start data with multi-stage RL.",
      "gpt41_resp": "DeepSeek-R1 was released in January 2025 as a 33B parameter dense reasoning model utilizing speculative decoding for math verification.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "DeepSeek-R1 was released in December 2024 as a distilled 70B parameter model trained exclusively on synthetic math data.",
      "deepseek_grade": "Hallucination",
      "notes": "Both models confabulated model parameter sizes and training methods for DeepSeek-R1."
    },
    {
      "id": 51,
      "category": "Citation",
      "prompt": "What is the exact CVE identifier for the XZ Utils backdoor discovered in March 2024?",
      "ground_truth": "CVE-2024-3094",
      "gpt41_resp": "CVE-2024-3094.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "CVE-2024-3094.",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly identify the high-profile XZ backdoor CVE."
    },
    {
      "id": 52,
      "category": "Citation",
      "prompt": "What is the CVE identifier for the 2024 runc container breakout vulnerability involving /proc/self/exe?",
      "ground_truth": "CVE-2024-21626",
      "gpt41_resp": "CVE-2024-21626.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "CVE-2024-21626.",
      "deepseek_grade": "Correct",
      "notes": "Both models identify the runc leacky file descriptor container breakout."
    },
    {
      "id": 53,
      "category": "Citation",
      "prompt": "Which IETF RFC defines the JSON Web Signature (JWS) standard?",
      "ground_truth": "RFC 7515",
      "gpt41_resp": "RFC 7515.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "RFC 7515.",
      "deepseek_grade": "Correct",
      "notes": "Both models correctly cite RFC 7515."
    },
    {
      "id": 54,
      "category": "Citation",
      "prompt": "What is the exact RFC number for HTTP/3 over QUIC?",
      "ground_truth": "RFC 9114",
      "gpt41_resp": "RFC 9000 defines HTTP/3 over QUIC.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "RFC 9000.",
      "deepseek_grade": "Hallucination",
      "notes": "Both models conflated RFC 9000 (QUIC transport) with RFC 9114 (HTTP/3)."
    },
    {
      "id": 55,
      "category": "Citation",
      "prompt": "What is the CVE identifier for the Apache Log4j Log4Shell vulnerability disclosed in December 2021?",
      "ground_truth": "CVE-2021-44228",
      "gpt41_resp": "CVE-2021-44228.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "CVE-2021-44228.",
      "deepseek_grade": "Correct",
      "notes": "Both correctly identify Log4Shell."
    },
    {
      "id": 56,
      "category": "Citation",
      "prompt": "What is the CVE identifier for the Terrapin attack on the SSH protocol disclosed in December 2023?",
      "ground_truth": "CVE-2023-48795",
      "gpt41_resp": "CVE-2023-38408.",
      "gpt41_grade": "Hallucination",
      "deepseek_resp": "CVE-2023-51385.",
      "deepseek_grade": "Hallucination",
      "notes": "Both models hallucinated incorrect CVE numbers (GPT-4.1 gave PKCS#11 agent bug CVE-2023-38408; DeepSeek fabricated CVE-2023-51385)."
    },
    {
      "id": 57,
      "category": "Citation",
      "prompt": "What is the IETF RFC number for the WebSocket protocol?",
      "ground_truth": "RFC 6455",
      "gpt41_resp": "RFC 6455.",
      "gpt41_grade": "Correct",
      "deepseek_resp": "RFC 6455.",
      "deepseek_grade": "Correct",
      "notes": "Both correctly cite RFC 6455."
    },
    {
      "id": 58,
      "category": "Citation",
      "prompt": "What is the CVE identifier for the OpenSSL Punycode buffer overflow vulnerability disclosed in November 2022?",
      "ground_truth": "CVE-2022-3602 (and secondary CVE-2022-3786)",
      "gpt41_resp": "CVE-2022-3602 (along with CVE-2022-3786).",
      "gpt41_grade": "Correct",
      "deepseek_resp": "CVE-2022-1292.",
      "deepseek_grade": "Hedged wrong",
      "notes": "DeepSeek cited CVE-2022-1292 (c_rehash flaw) instead of the Punycode flaw CVE-2022-3602."
    },
    {
      "id": 59,
      "category": "Citation",
      "prompt": "What is the RFC number specifying the HTTP Basic Authentication scheme that obsoleted RFC 2617?",
      "ground_truth": "RFC 7617",
      "gpt41_resp": "RFC 2617 defines the current HTTP Basic Authentication scheme.",
      "gpt41_grade": "Hedged wrong",
      "deepseek_resp": "RFC 2068.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek hallucinated an obsolete HTTP/1.1 draft RFC 2068 instead of RFC 7617. GPT-4.1 cited obsoleted RFC 2617."
    },
    {
      "id": 60,
      "category": "Citation",
      "prompt": "What is the CVE identifier for the Apache HTTP Server 2.4.49 path traversal vulnerability in October 2021?",
      "ground_truth": "CVE-2021-41773 (and bypass CVE-2021-42013)",
      "gpt41_resp": "CVE-2021-41773 (and bypass fix CVE-2021-42013).",
      "gpt41_grade": "Correct",
      "deepseek_resp": "CVE-2021-40438.",
      "deepseek_grade": "Hallucination",
      "notes": "DeepSeek cited CVE-2021-40438 (mod_proxy SSRF) instead of path traversal CVE-2021-41773."
    }
  ]
}