#!/usr/bin/env python3 """ make_joke_gguf.py Crafts a GGUF file whose tensor metadata declares an absurd number of parameters (by default, INT64_MAX = 9,223,372,036,854,775,807 elements, which Hugging Face's param-count formatter renders as roughly "9223372T params") while the actual file on disk is small — padded out to whatever real size you ask for (default ~50MB). How it works: GGUF files store tensor *shape* declarations (name, dims, dtype, offset) separately from the raw tensor data blob. Hub UIs that show a "params" badge read the declared shapes from the header/tensor-info section and multiply them out — they don't verify that the data blob actually contains that many bytes. So we can declare a single 1-D tensor with ne[0] = 9223372036854775807 and then just pad the real file with zero bytes to hit a normal-looking file size. Note: this makes an intentionally non-functional / joke model. It will show a huge param count on the Hub but will NOT load in llama.cpp or any other real inference stack, since the declared tensor size and the actual byte count on disk don't match. Usage: python make_joke_gguf.py [output.gguf] [--size-mb 50] [--params N] """ import argparse import struct GGUF_MAGIC = b"GGUF" GGUF_VERSION = 3 ALIGNMENT = 32 # GGUF metadata value type IDs UINT8, INT8, UINT16, INT16, UINT32, INT32 = 0, 1, 2, 3, 4, 5 FLOAT32, BOOL, STRING, ARRAY, UINT64, INT64, FLOAT64 = 6, 7, 8, 9, 10, 11, 12 # ggml tensor type IDs (0 = F32, harmless placeholder since the data is fake) GGML_TYPE_F32 = 0 def gguf_str(s: str) -> bytes: b = s.encode("utf-8") return struct.pack(" bytes: return gguf_str(key) + struct.pack(" bytes: return gguf_str(key) + struct.pack(" None: # --- metadata key/value section ------------------------------------- metadata = [ kv_string("general.architecture", "joke"), kv_string("general.name", f"Absurd-{fake_param_count}-Params"), kv_uint32("general.quantization_version", 2), kv_uint32("general.alignment", ALIGNMENT), ] # --- single fake tensor declaration ---------------------------------- tensor_name = "dummy.weight" dims = [fake_param_count] # 1-D tensor, one huge dimension tensor_info = gguf_str(tensor_name) tensor_info += struct.pack(" 0: n = min(len(chunk), remaining) f.write(chunk[:n]) remaining -= n print(f"Wrote {path} ({total_size_bytes / (1024*1024):.1f} MB on disk, " f"declared {fake_param_count:,} params)") def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("output", nargs="?", default="joke_model.gguf") parser.add_argument("--size-mb", type=float, default=50.0, help="real file size on disk, in MB (default: 50)") parser.add_argument("--params", type=int, default=(2**63) - 1, help="fake declared parameter count " "(default: 9223372036854775807, i.e. INT64_MAX, " "which the Hub renders as ~9223372T params)") args = parser.parse_args() build_gguf(args.output, int(args.size_mb * 1024 * 1024), args.params) if __name__ == "__main__": main()