alexwengg commited on
Commit
0deb31c
·
verified ·
1 Parent(s): fc4b561

Publish validated Jeff GLiFormer Large L128 FP16 Core ML classifier

Browse files
JeffDecision-L128-FP16.mlpackage/Data/com.apple.CoreML/model.mlmodel ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:87c0960087a6b39d6aac2e0c2511340f0197f19d471b77b2e894ca265204c1d3
3
+ size 423429
JeffDecision-L128-FP16.mlpackage/Data/com.apple.CoreML/weights/weight.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25988597bb4cc8b488970f7c6a15671dcf5f422c88c01d244a35a99a6aa45ba3
3
+ size 922387968
JeffDecision-L128-FP16.mlpackage/Manifest.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "fileFormatVersion": "1.0.0",
3
+ "itemInfoEntries": {
4
+ "A8DBF8BD-E4B2-4DC7-9A54-336D47C02F68": {
5
+ "author": "com.apple.CoreML",
6
+ "description": "CoreML Model Weights",
7
+ "name": "weights",
8
+ "path": "com.apple.CoreML/weights"
9
+ },
10
+ "AB3C25D4-080D-4C1A-9C41-B16757CD95F1": {
11
+ "author": "com.apple.CoreML",
12
+ "description": "CoreML Model Specification",
13
+ "name": "model.mlmodel",
14
+ "path": "com.apple.CoreML/model.mlmodel"
15
+ }
16
+ },
17
+ "rootModelIdentifier": "AB3C25D4-080D-4C1A-9C41-B16757CD95F1"
18
+ }
README.md ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ library_name: coreml
4
+ tags:
5
+ - coreml
6
+ - text-classification
7
+ - jeff
8
+ - gliformer
9
+ - on-device
10
+ ---
11
+
12
+ # Jeff / GLiFormer Large classification for Core ML
13
+
14
+ This is a fixed L128, batch-1 Core ML conversion of the **trained classification path** in [knowledgator/gliformer-large-v1](https://huggingface.co/knowledgator/gliformer-large-v1), as used by [Jeff](https://github.com/logan-markewich/jeff). It includes the 24-layer DeBERTa encoder, learned prompt-marker embeddings, and trained linear classification head in one FP16 ML Program. The original checkpoint contains 575,637,510 parameters; this Core ML package is about 880 MB on disk. The source checkpoint, tokenizer and decision-engine revisions are pinned in `assets.lock.json`.
15
+
16
+ The model scores one classification group with 1–8 labels and up to 128 tokenized input tokens. Choice, yes/no and ordinal labels use the same trained classification head; the probabilities are independent sigmoid scores, not a normalized softmax. Only the first `len(labels)` logits are returned. Other GLiFormer tasks such as NER, layout extraction, vision, audio and structuring are **not** exported here. This is not a Decision Index benchmark result.
17
+
18
+ The original model's classifier uses CLS pooling, a parent prompt anchor, linear parent+category fusion and dot scoring. Its word-level RNN is evaluated upstream but cannot affect these classification logits under this checkpoint's settings. `jeff_decision.py` checks those settings and carries the trained encoder and head; `trace_compat.py` makes fixed-shape DeBERTa attention convertible without changing trained weights. The trace-only finite mask substitutes -10000 for fp32-min before FP16 conversion; patched PyTorch logits match native logits on the checked cases.
19
+
20
+ ## Local inference
21
+
22
+ On macOS 15 or later with Python 3.12:
23
+
24
+ ```bash
25
+ uv sync --frozen
26
+ uv run python - <<'PY'
27
+ from runtime import JeffCoreML
28
+
29
+ model = JeffCoreML(".", "JeffDecision-L128-FP16.mlpackage")
30
+ print(model.score(
31
+ "The invoice was charged twice and the customer asks for a refund.",
32
+ ["billing: invoice or payment issue", "support: technical product issue"],
33
+ name="Choose the correct support queue",
34
+ ))
35
+ PY
36
+ ```
37
+
38
+ `JeffCoreML` uses the bundled tokenizer/config and GLiFormer processor, and loads **no original PyTorch weights**. Inputs exceeding 128 tokens or eight labels fail explicitly. Build from the pinned source checkpoint with `uv run python export.py --convert --precision fp16`; the source checkpoint must be in the local Hugging Face cache. `uv run python verify.py --precision fp16` compares Core ML logits to native PyTorch.
39
+
40
+ ## Validation
41
+
42
+ Four real source-checkpoint fixtures cover billing, technical support, a three-label intent choice and yes/no classification. The mathematical decision wrapper matched native logits within **3.82e-6**. Tracing-only patches matched native logits exactly on those fixtures. The exported FP16 Core ML model preserved all four chosen labels with maximum absolute logit error **0.1139**. FP32 Core ML is a diagnostic control with four-of-four agreement and maximum logit error **3.44e-5**; the FP32 package is not included because it is much larger. Full Decision Index quality, additional input lengths, other task heads and broad latency/ANE performance have not been evaluated. See `native-parity.json` and `coreml-parity-fp16.json`.
43
+
44
+ This conversion is derived from the Apache-2.0 GLiFormer weights and [Transformers](https://github.com/huggingface/transformers) DeBERTa implementation. Jeff's decision adapter code is MIT licensed; the helper source here retains attribution. The model card makes no claim that this conversion is faster or more accurate than another model on a benchmark.
assets.lock.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "decision_engine": {
3
+ "repo": "https://github.com/logan-markewich/jeff",
4
+ "revision": "34b32f99a727c47b679adde33f4702a001e02979",
5
+ "license": "MIT"
6
+ },
7
+ "checkpoint": {
8
+ "repo": "knowledgator/gliformer-large-v1",
9
+ "revision": "d0a4e53d09cebe6bc963dd9be319d4279084bb2d",
10
+ "weights_file": "pytorch_model.bin",
11
+ "weights_bytes": 2302735855,
12
+ "weights_sha256": "f80b29199d66f878669f283703e4dba9fd726755dcc20aba1ed0d24fce4a23f1",
13
+ "license": "Apache-2.0"
14
+ },
15
+ "scope": "Jeff typed-decision classification path; not all GLiFormer extraction heads"
16
+ }
coreml-parity-fp16.json ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "name": "billing",
4
+ "labels": [
5
+ "billing: invoice or payment issue",
6
+ "support: technical product issue"
7
+ ],
8
+ "native_logits": [
9
+ 10.187458038330078,
10
+ -6.71131706237793
11
+ ],
12
+ "coreml_logits": [
13
+ 10.125,
14
+ -6.65625
15
+ ],
16
+ "max_logit_error": 0.062458038330078125,
17
+ "top_label_agreement": true,
18
+ "coreml_wall_ms": 165.4915830004029
19
+ },
20
+ {
21
+ "name": "technical",
22
+ "labels": [
23
+ "billing: invoice or payment issue",
24
+ "support: technical product issue"
25
+ ],
26
+ "native_logits": [
27
+ -12.322779655456543,
28
+ 13.379047393798828
29
+ ],
30
+ "coreml_logits": [
31
+ -12.296875,
32
+ 13.4140625
33
+ ],
34
+ "max_logit_error": 0.035015106201171875,
35
+ "top_label_agreement": true,
36
+ "coreml_wall_ms": 82.55887497216463
37
+ },
38
+ {
39
+ "name": "three_way",
40
+ "labels": [
41
+ "schedule: appointment request",
42
+ "billing: payment issue",
43
+ "support: technical issue"
44
+ ],
45
+ "native_logits": [
46
+ 5.41853141784668,
47
+ -12.076669692993164,
48
+ -9.080745697021484
49
+ ],
50
+ "coreml_logits": [
51
+ 5.40234375,
52
+ -12.09375,
53
+ -9.1171875
54
+ ],
55
+ "max_logit_error": 0.036441802978515625,
56
+ "top_label_agreement": true,
57
+ "coreml_wall_ms": 53.515166975557804
58
+ },
59
+ {
60
+ "name": "boolean",
61
+ "labels": [
62
+ "yes",
63
+ "no"
64
+ ],
65
+ "native_logits": [
66
+ 1.0098832845687866,
67
+ -1.0589547157287598
68
+ ],
69
+ "coreml_logits": [
70
+ 0.89599609375,
71
+ -1.001953125
72
+ ],
73
+ "max_logit_error": 0.11388719081878662,
74
+ "top_label_agreement": true,
75
+ "coreml_wall_ms": 53.68766700848937
76
+ }
77
+ ]
export.py ADDED
@@ -0,0 +1,166 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Validate Jeff's trained classifier graph and export its complete decision path."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import warnings
8
+ from pathlib import Path
9
+
10
+ import numpy as np
11
+ import torch
12
+ import torch.nn.functional as F
13
+ from huggingface_hub import snapshot_download
14
+ from jeff.backends.torch_backend import TorchBackend
15
+ from jeff.core.backend import Group
16
+
17
+ from jeff_decision import JeffDecision, marker_positions
18
+ from trace_compat import finite_fp16_mask, install_trace_compatibility
19
+
20
+ SOURCE = "knowledgator/gliformer-large-v1"
21
+ REVISION = "d0a4e53d09cebe6bc963dd9be319d4279084bb2d"
22
+ BUCKET = 128
23
+ MAX_CATEGORIES = 8
24
+
25
+ FIXTURES = (
26
+ (
27
+ "billing",
28
+ "The invoice was charged twice and the customer asks for a refund.",
29
+ Group(key="route", labels=("billing: invoice or payment issue", "support: technical product issue"),
30
+ name="Choose the correct support queue"),
31
+ ),
32
+ (
33
+ "technical",
34
+ "The app crashes when I save my project. Please help me recover the file.",
35
+ Group(key="route", labels=("billing: invoice or payment issue", "support: technical product issue"),
36
+ name="Choose the correct support queue"),
37
+ ),
38
+ (
39
+ "three_way",
40
+ "Tomorrow at 9 a.m. works well for the appointment.",
41
+ Group(key="intent", labels=("schedule: appointment request", "billing: payment issue",
42
+ "support: technical issue"), name="Classify the user intent"),
43
+ ),
44
+ (
45
+ "boolean",
46
+ "I cannot sign in after resetting my password.",
47
+ Group(key="answer", labels=("yes", "no"), name="Is this a technical support request?"),
48
+ ),
49
+ )
50
+
51
+
52
+ def make_batch(backend: TorchBackend, text: str, group: Group) -> dict:
53
+ tokens, _, _ = backend.model.prepare_inputs([text])
54
+ return backend._collator([{
55
+ "tokenized_text": tokens[0],
56
+ "classification": [{
57
+ "name": group.name,
58
+ "description": group.description,
59
+ "all_labels": list(group.labels),
60
+ "true_labels": [],
61
+ }],
62
+ }])
63
+
64
+
65
+ def model_inputs(batch: dict, config) -> tuple[torch.Tensor, ...]:
66
+ ids = batch["input_ids"]
67
+ mask = batch["attention_mask"]
68
+ if ids.shape[1] > BUCKET:
69
+ raise ValueError(f"input has {ids.shape[1]} tokens; L{BUCKET} cannot serve it")
70
+ parent, children, count = marker_positions(ids, config, MAX_CATEGORIES)
71
+ if count != len(batch["classes_mapping"].cat_mapping[0].cat_class_to_id[0].class_to_id):
72
+ raise ValueError("collator category mapping does not match marker count")
73
+ pad = BUCKET - ids.shape[1]
74
+ return (
75
+ F.pad(ids.to(torch.int32), (0, pad)),
76
+ F.pad(mask.to(torch.int32), (0, pad)),
77
+ parent,
78
+ children,
79
+ )
80
+
81
+
82
+ @torch.inference_mode()
83
+ def native_report(backend: TorchBackend, decision: JeffDecision) -> tuple[list[dict], tuple[torch.Tensor, ...]]:
84
+ records = []
85
+ first = None
86
+ for name, text, group in FIXTURES:
87
+ batch = make_batch(backend, text, group)
88
+ tensors = model_inputs(batch, backend.model.config)
89
+ native = backend.model.model(**batch, include_media=False).cat_logits.detach().float().numpy()[0]
90
+ converted = decision(*tensors).detach().float().numpy()[0, :len(group.labels)]
91
+ error = float(np.max(np.abs(native - converted)))
92
+ records.append({
93
+ "name": name,
94
+ "token_count": int(batch["attention_mask"].sum()),
95
+ "labels": list(group.labels),
96
+ "native_logits": native.tolist(),
97
+ "decision_logits": converted.tolist(),
98
+ "max_logit_error": error,
99
+ })
100
+ if error > 1e-3:
101
+ raise AssertionError(f"{name}: trained graph differs from native logits by {error}")
102
+ if first is None:
103
+ first = tensors
104
+ assert first is not None
105
+ return records, first
106
+
107
+
108
+ def main() -> None:
109
+ parser = argparse.ArgumentParser()
110
+ parser.add_argument("--convert", action="store_true")
111
+ parser.add_argument("--precision", choices=("fp16", "fp32"), default="fp16")
112
+ args = parser.parse_args()
113
+ torch.set_num_threads(2)
114
+ checkpoint = snapshot_download(SOURCE, revision=REVISION, local_files_only=True)
115
+ with warnings.catch_warnings():
116
+ warnings.filterwarnings("ignore", message=r"attn_kernel=.*flashdeberta")
117
+ backend = TorchBackend(checkpoint, device="cpu", dtype="float32", attn_kernel="eager", batch_size=1)
118
+ decision = JeffDecision(backend.model.model).eval()
119
+ report, first = native_report(backend, decision)
120
+ out = Path("build")
121
+ out.mkdir(exist_ok=True)
122
+ (out / "native-parity.json").write_text(json.dumps(report, indent=2) + "\n")
123
+ print(json.dumps({"native_parity": report}, indent=2), flush=True)
124
+ if not args.convert:
125
+ return
126
+
127
+ import coremltools as ct
128
+ install_trace_compatibility()
129
+ with finite_fp16_mask():
130
+ patched_report, _ = native_report(backend, decision)
131
+ patched_error = max(
132
+ abs(before - after)
133
+ for baseline, patched in zip(report, patched_report)
134
+ for before, after in zip(baseline["decision_logits"], patched["decision_logits"])
135
+ )
136
+ if patched_error > 1e-3:
137
+ raise AssertionError(f"tracing-only mask/attention scale changes native logits by {patched_error}")
138
+ print(f"Patched mask/attention max native logit error: {patched_error:.8f}", flush=True)
139
+ traced = torch.jit.trace(decision, first, check_trace=False).eval()
140
+ traced.save(str(out / "jeff-decision-L128.pt"))
141
+ with torch.inference_mode():
142
+ expected = decision(*first).detach().numpy()
143
+ actual = traced(*first).detach().numpy()
144
+ trace_error = float(np.max(np.abs(expected - actual)))
145
+ if trace_error > 1e-3:
146
+ raise AssertionError(f"TorchScript trace mismatch: {trace_error}")
147
+ print(f"TorchScript trace max logit error: {trace_error:.8f}", flush=True)
148
+ mlmodel = ct.convert(
149
+ traced,
150
+ convert_to="mlprogram",
151
+ minimum_deployment_target=ct.target.macOS15,
152
+ compute_precision=ct.precision.FLOAT16 if args.precision == "fp16" else ct.precision.FLOAT32,
153
+ inputs=[
154
+ ct.TensorType(name="input_ids", shape=(1, BUCKET), dtype=np.int32),
155
+ ct.TensorType(name="attention_mask", shape=(1, BUCKET), dtype=np.int32),
156
+ ct.TensorType(name="parent_position", shape=(1, 1), dtype=np.int32),
157
+ ct.TensorType(name="category_positions", shape=(1, MAX_CATEGORIES), dtype=np.int32),
158
+ ],
159
+ )
160
+ package = out / f"JeffDecision-L128-{args.precision.upper()}.mlpackage"
161
+ mlmodel.save(str(package))
162
+ print(f"Saved {package}", flush=True)
163
+
164
+
165
+ if __name__ == "__main__":
166
+ main()
gliner_config.json ADDED
@@ -0,0 +1,441 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "adjacency_loss_coef": 1.0,
3
+ "anchor_num_heads": 4,
4
+ "anchor_num_layers": 2,
5
+ "backbone_type": "deberta_2d",
6
+ "bos_token_id": 1,
7
+ "cat_loss_coef": 1.0,
8
+ "cat_parent_token": "[SCHEMA]",
9
+ "cat_token": "[CLASS]",
10
+ "cat_token_index": 128003,
11
+ "child_token": "[FIELD]",
12
+ "child_token_index": 128005,
13
+ "class_token_index": 128001,
14
+ "classification_config": {
15
+ "anchor_context_gate_init": 0.1,
16
+ "anchor_context_gate_trainable": true,
17
+ "anchor_cross_attention_bias": null,
18
+ "anchor_layer": null,
19
+ "anchor_memory_position": null,
20
+ "anchor_memory_position_usage": null,
21
+ "anchor_mode": "parent",
22
+ "anchor_modeling": "linear",
23
+ "anchor_normalization": "none",
24
+ "anchor_query_position": null,
25
+ "anchor_refine_heads": 8,
26
+ "anchor_refine_layer_scale_init": null,
27
+ "anchor_refine_layers": 0,
28
+ "anchor_refine_norm": "post_norm",
29
+ "anchor_refinement": null,
30
+ "anchor_self_attention_bias": null,
31
+ "cat_token_index": 128003,
32
+ "embed_cat_token": true,
33
+ "embed_parent_token": true,
34
+ "feature_anchor_mlp": false,
35
+ "feature_anchor_mlp_hidden_multiplier": 1,
36
+ "focal_loss_alpha": 0.8,
37
+ "focal_loss_gamma": -1.0,
38
+ "focal_loss_prob_margin": 0.0,
39
+ "loss_coef": 1.0,
40
+ "parent_token_index": 128002,
41
+ "pooling_type": "cls",
42
+ "scorer_type": "dot"
43
+ },
44
+ "classifier_layer": null,
45
+ "count_config": null,
46
+ "count_layer": null,
47
+ "count_loss_coef": 1.0,
48
+ "count_mode": "regression",
49
+ "dropout": 0.25,
50
+ "embed_cat_token": true,
51
+ "embed_child_token": true,
52
+ "embed_ent_token": true,
53
+ "embed_parent_token": true,
54
+ "embed_rel_token": true,
55
+ "embedding_config": {
56
+ "encoder_dropout": 0.0,
57
+ "loss_coef": 1.0,
58
+ "loss_fn": "mse",
59
+ "margin": 0.0,
60
+ "pooling_type": "cls",
61
+ "projection_dim": 1024,
62
+ "projection_dropout": 0.0,
63
+ "similarity_fn": "cosine"
64
+ },
65
+ "embedding_loss_coef": 1.0,
66
+ "encoder_config": {
67
+ "_name_or_path": "knowledgator/DeBERTa-large-joint-3000",
68
+ "architectures": [
69
+ "FlashDebertaV2ForContextPredictionTraining"
70
+ ],
71
+ "attention_probs_dropout_prob": 0.0,
72
+ "attn_kernel": "flash",
73
+ "bos_token_id": null,
74
+ "chunk_size_feed_forward": 0,
75
+ "coordinate_size": 128,
76
+ "dtype": "bfloat16",
77
+ "eos_token_id": null,
78
+ "hidden_act": "gelu",
79
+ "hidden_dropout_prob": 0.0,
80
+ "hidden_size": 1024,
81
+ "id2label": {
82
+ "0": "LABEL_0",
83
+ "1": "LABEL_1"
84
+ },
85
+ "initializer_range": 0.02,
86
+ "intermediate_size": 4096,
87
+ "is_encoder_decoder": false,
88
+ "label2id": {
89
+ "LABEL_0": 0,
90
+ "LABEL_1": 1
91
+ },
92
+ "layer_norm_eps": 1e-07,
93
+ "layout_bias_propagation": "all_layers",
94
+ "layout_embedding_type": "absolute",
95
+ "layout_max_relative_positions": 1024,
96
+ "layout_position_buckets": 32,
97
+ "layout_relative_attention": false,
98
+ "legacy": true,
99
+ "max_2d_position_embeddings": 1024,
100
+ "max_page_embeddings": 1024,
101
+ "max_position_embeddings": 512,
102
+ "max_relative_positions": -1,
103
+ "model_type": "layout-deberta",
104
+ "norm_rel_ebd": "layer_norm",
105
+ "num_attention_heads": 16,
106
+ "num_hidden_layers": 24,
107
+ "output_attentions": false,
108
+ "output_hidden_states": false,
109
+ "pad_token_id": 0,
110
+ "pooler_dropout": 0.0,
111
+ "pooler_hidden_act": "gelu",
112
+ "pooler_hidden_size": 1024,
113
+ "pos_att_type": [
114
+ "p2c",
115
+ "c2p"
116
+ ],
117
+ "position_biased_input": false,
118
+ "position_buckets": 256,
119
+ "problem_type": null,
120
+ "relative_attention": true,
121
+ "return_dict": true,
122
+ "shape_size": 64,
123
+ "share_att_key": true,
124
+ "spatial_embedding_type": "absolute",
125
+ "tie_word_embeddings": true,
126
+ "type_vocab_size": 0,
127
+ "vocab_size": 128008
128
+ },
129
+ "ent_token": "[ENTITY]",
130
+ "eos_token_id": 2,
131
+ "fine_tune": true,
132
+ "fuse_layers": false,
133
+ "groups_layer": "fixed",
134
+ "groups_loss_coef": 1.0,
135
+ "hidden_size": 1024,
136
+ "image_size": 224,
137
+ "joint_relex_config": {
138
+ "adjacency_loss_coef": 1.0,
139
+ "anchor_context_gate_init": 0.1,
140
+ "anchor_context_gate_trainable": true,
141
+ "anchor_cross_attention_bias": null,
142
+ "anchor_layer": null,
143
+ "anchor_memory_position": null,
144
+ "anchor_memory_position_usage": null,
145
+ "anchor_mode": "fixed",
146
+ "anchor_modeling": "linear",
147
+ "anchor_normalization": "none",
148
+ "anchor_num_heads": 4,
149
+ "anchor_num_layers": 2,
150
+ "anchor_query_position": null,
151
+ "anchor_refine_heads": 8,
152
+ "anchor_refine_layer_scale_init": null,
153
+ "anchor_refine_layers": 0,
154
+ "anchor_refine_norm": "post_norm",
155
+ "anchor_refinement": null,
156
+ "anchor_self_attention_bias": null,
157
+ "embed_parent_token": true,
158
+ "embed_rel_token": true,
159
+ "feature_anchor_mlp": false,
160
+ "feature_anchor_mlp_hidden_multiplier": 1,
161
+ "focal_loss_alpha": 0.8,
162
+ "focal_loss_gamma": -1.0,
163
+ "focal_loss_prob_margin": 0.0,
164
+ "loss_coef": 1.0,
165
+ "max_count": 20,
166
+ "max_relation_entities": null,
167
+ "max_relation_span_width": null,
168
+ "neg_spans_ratio": 1.0,
169
+ "num_fixed_slots": 10,
170
+ "parent_token_index": -1,
171
+ "rel_token_index": 128004,
172
+ "relation_focal_loss_alpha": 0.8,
173
+ "relation_focal_loss_gamma": 2,
174
+ "relation_focal_loss_prob_margin": 0.0,
175
+ "relation_loss_coef": 1.0,
176
+ "relation_loss_reduction": "mean",
177
+ "relation_neighbor_chunk_size": 512,
178
+ "relation_span_nms": false,
179
+ "relation_top_k_neighbors": null,
180
+ "relations_layer": null,
181
+ "represent_spans": false,
182
+ "span_loss_coef": 1.0,
183
+ "span_loss_reduction": "mean",
184
+ "triples_layer": null
185
+ },
186
+ "labels_encoder": null,
187
+ "labels_encoder_config": null,
188
+ "layout_image_tokens": true,
189
+ "max_count": 20,
190
+ "max_len": 16384,
191
+ "max_neg_type_ratio": 1,
192
+ "max_page_embeddings": 1024,
193
+ "max_types": 128,
194
+ "max_width": 12,
195
+ "media_parent_embedding_source": "fixed",
196
+ "model_name": "knowledgator/DeBERTa-large-joint-3000",
197
+ "model_type": "gliformer-layout",
198
+ "model_variant": "layout",
199
+ "multimodal_fusion": "uni-encoder",
200
+ "name": "gliformer",
201
+ "neg_spans_ratio": 1.0,
202
+ "ner_config": {
203
+ "anchor_context_gate_init": 0.1,
204
+ "anchor_context_gate_trainable": true,
205
+ "anchor_cross_attention_bias": null,
206
+ "anchor_layer": null,
207
+ "anchor_memory_position": null,
208
+ "anchor_memory_position_usage": null,
209
+ "anchor_mode": "parent",
210
+ "anchor_modeling": "linear",
211
+ "anchor_normalization": "none",
212
+ "anchor_query_position": null,
213
+ "anchor_refine_heads": 8,
214
+ "anchor_refine_layer_scale_init": null,
215
+ "anchor_refine_layers": 0,
216
+ "anchor_refine_norm": "post_norm",
217
+ "anchor_refinement": null,
218
+ "anchor_self_attention_bias": null,
219
+ "embed_parent_token": true,
220
+ "feature_anchor_mlp": false,
221
+ "feature_anchor_mlp_hidden_multiplier": 1,
222
+ "focal_loss_alpha": 0.8,
223
+ "focal_loss_gamma": -1,
224
+ "focal_loss_prob_margin": 0.0,
225
+ "loss_coef": 1.0,
226
+ "neg_spans_ratio": 1.0,
227
+ "parent_token_index": 128002,
228
+ "represent_spans": false,
229
+ "span_loss_coef": 1.0,
230
+ "span_loss_reduction": "mean"
231
+ },
232
+ "ner_loss_coef": 1.0,
233
+ "ner_parent_token": "[SCHEMA]",
234
+ "num_post_fusion_layers": 1,
235
+ "num_rnn_layers": 1,
236
+ "obj_token": "[OBJECT]",
237
+ "open_rel_parent_token": "[SCHEMA]",
238
+ "open_relex_config": null,
239
+ "pad_token_id": 0,
240
+ "parent_token": "[SCHEMA]",
241
+ "parent_token_index": 128002,
242
+ "per_task_parents": false,
243
+ "post_fusion_schema": "",
244
+ "projector_hidden_act": "gelu",
245
+ "rel_mode": "adjacency",
246
+ "rel_token": "[RELATION]",
247
+ "rel_token_index": 128004,
248
+ "relation_loss_coef": 1.0,
249
+ "relations_config": {
250
+ "adjacency_loss_coef": 1.0,
251
+ "anchor_context_gate_init": 0.1,
252
+ "anchor_context_gate_trainable": true,
253
+ "anchor_cross_attention_bias": null,
254
+ "anchor_layer": null,
255
+ "anchor_memory_position": null,
256
+ "anchor_memory_position_usage": null,
257
+ "anchor_mode": "fixed",
258
+ "anchor_modeling": "linear",
259
+ "anchor_normalization": "none",
260
+ "anchor_num_heads": 4,
261
+ "anchor_num_layers": 2,
262
+ "anchor_query_position": null,
263
+ "anchor_refine_heads": 8,
264
+ "anchor_refine_layer_scale_init": null,
265
+ "anchor_refine_layers": 0,
266
+ "anchor_refine_norm": "post_norm",
267
+ "anchor_refinement": null,
268
+ "anchor_self_attention_bias": null,
269
+ "embed_parent_token": true,
270
+ "embed_rel_token": true,
271
+ "feature_anchor_mlp": false,
272
+ "feature_anchor_mlp_hidden_multiplier": 1,
273
+ "focal_loss_alpha": 0.8,
274
+ "focal_loss_gamma": -1.0,
275
+ "focal_loss_prob_margin": 0.0,
276
+ "loss_coef": 1.0,
277
+ "max_count": 20,
278
+ "max_relation_entities": null,
279
+ "max_relation_span_width": null,
280
+ "neg_spans_ratio": 1.0,
281
+ "num_fixed_slots": 10,
282
+ "parent_token_index": -1,
283
+ "rel_token_index": 128004,
284
+ "relation_focal_loss_alpha": 0.8,
285
+ "relation_focal_loss_gamma": 2,
286
+ "relation_focal_loss_prob_margin": 0.0,
287
+ "relation_loss_coef": 1.0,
288
+ "relation_loss_reduction": "mean",
289
+ "relation_neighbor_chunk_size": 512,
290
+ "relation_span_nms": false,
291
+ "relation_top_k_neighbors": null,
292
+ "relations_layer": null,
293
+ "represent_spans": false,
294
+ "span_loss_coef": 1.0,
295
+ "span_loss_reduction": "mean",
296
+ "triples_layer": null
297
+ },
298
+ "relations_layer": null,
299
+ "represent_spans": false,
300
+ "sep_token": "[SEP]",
301
+ "seq_token": "[SEQ]",
302
+ "shared_anchor_modeling": null,
303
+ "shared_anchor_refine_heads": 8,
304
+ "shared_anchor_refine_layers": 0,
305
+ "shared_anchor_refinement": null,
306
+ "span_loss_coef": 1.0,
307
+ "span_mode": "markerV0",
308
+ "struct_parent_token": "[SCHEMA]",
309
+ "structuring_child_token": "<<CHILD>>",
310
+ "structuring_config": {
311
+ "anchor_context_gate_init": 0.1,
312
+ "anchor_context_gate_trainable": true,
313
+ "anchor_cross_attention_bias": {
314
+ "params": {
315
+ "sigma": 0.5,
316
+ "units": "query_steps",
317
+ "weight": 1.0
318
+ },
319
+ "type": "gaussian_distance"
320
+ },
321
+ "anchor_layer": {
322
+ "params": {
323
+ "num_slots": 100
324
+ },
325
+ "type": "fixed"
326
+ },
327
+ "anchor_memory_position": {
328
+ "params": {},
329
+ "type": "fourier"
330
+ },
331
+ "anchor_memory_position_usage": "keys_and_values",
332
+ "anchor_mode": "fixed_transformer",
333
+ "anchor_modeling": "linear",
334
+ "anchor_normalization": "center_rms",
335
+ "anchor_num_heads": 4,
336
+ "anchor_num_layers": 2,
337
+ "anchor_objectness": true,
338
+ "anchor_objectness_loss_coef": 1.0,
339
+ "anchor_objectness_threshold": 0.5,
340
+ "anchor_query_position": {
341
+ "params": {},
342
+ "type": "fourier"
343
+ },
344
+ "anchor_refine_heads": 8,
345
+ "anchor_refine_layer_scale_init": null,
346
+ "anchor_refine_layers": 0,
347
+ "anchor_refine_norm": "post_norm",
348
+ "anchor_refinement": {
349
+ "params": {
350
+ "layer_scale_init": 0.1,
351
+ "norm_style": "pre_norm",
352
+ "num_heads": 8,
353
+ "num_layers": 2
354
+ },
355
+ "type": "cross_attention"
356
+ },
357
+ "anchor_relations_focal_loss_alpha": 0.75,
358
+ "anchor_relations_focal_loss_gamma": null,
359
+ "anchor_relations_focal_loss_prob_margin": null,
360
+ "anchor_relations_layer": "mlp",
361
+ "anchor_relations_loss_coef": 1.0,
362
+ "anchor_relations_threshold": 0.5,
363
+ "anchor_self_attention_bias": null,
364
+ "assignment_loss_coef": 1.0,
365
+ "bio_loss_reduction": "mean",
366
+ "child_token_index": 128005,
367
+ "embed_child_token": true,
368
+ "embed_parent_token": true,
369
+ "entity_loss_coef": 1.0,
370
+ "feature_anchor_mlp": false,
371
+ "feature_anchor_mlp_hidden_multiplier": 1,
372
+ "focal_loss_alpha": 0.75,
373
+ "focal_loss_gamma": -1,
374
+ "focal_loss_prob_margin": 0.0,
375
+ "head_type": "structuring",
376
+ "log_loss_stats": false,
377
+ "log_loss_stats_every": 50,
378
+ "loss_coef": 1.0,
379
+ "masking": "none",
380
+ "matcher_dice_cost": 1.0,
381
+ "matcher_membership_cost": 1.0,
382
+ "matcher_membership_temperature": 1.0,
383
+ "matcher_objectness_cost": 0.5,
384
+ "matcher_objectness_temperature": 1.0,
385
+ "matching_focal_loss_alpha": 0.99,
386
+ "matching_focal_loss_gamma": null,
387
+ "matching_focal_loss_prob_margin": null,
388
+ "max_count": 100,
389
+ "memory_position_embedding_kwargs": {},
390
+ "memory_position_embedding_type": "none",
391
+ "memory_position_in_values": false,
392
+ "multi_level": true,
393
+ "neg_spans_ratio": 1.0,
394
+ "negatives": 1.0,
395
+ "ner_focal_loss_alpha": 0.75,
396
+ "ner_focal_loss_gamma": null,
397
+ "ner_focal_loss_prob_margin": null,
398
+ "num_fixed_slots": 10,
399
+ "objectness_focal_loss_alpha": 0.5,
400
+ "objectness_focal_loss_gamma": null,
401
+ "objectness_focal_loss_prob_margin": null,
402
+ "parent_token_index": 128002,
403
+ "position_bucket_attention_bias_type": "none",
404
+ "position_bucket_attention_bias_weight": 1.0,
405
+ "position_bucket_attention_sigma": 0.5,
406
+ "position_bucket_normalization": "none",
407
+ "query_position_embedding_kwargs": {},
408
+ "query_position_embedding_type": "none",
409
+ "represent_spans": true,
410
+ "reuse_ner_head": true,
411
+ "span_loss_coef": 1.0,
412
+ "span_loss_reduction": "mean",
413
+ "structure_mode": {
414
+ "decoder": null,
415
+ "decoder_options": {},
416
+ "params": {},
417
+ "processor": null,
418
+ "processor_options": {},
419
+ "type": "multi_level"
420
+ },
421
+ "use_anchor_matching": true
422
+ },
423
+ "structuring_end_token": "<<END>>",
424
+ "structuring_loss_coef": 1.0,
425
+ "subtoken_pooling": "first",
426
+ "token_loss_coef": 1.0,
427
+ "transformers_version": "5.16.1",
428
+ "triples_layer": null,
429
+ "use_layout": true,
430
+ "vision_center_crop_size": null,
431
+ "vision_do_normalize": false,
432
+ "vision_do_rescale": true,
433
+ "vision_image_mean": null,
434
+ "vision_image_std": null,
435
+ "vision_interpolation": "bilinear",
436
+ "vision_processor_name": null,
437
+ "vision_processor_type": "custom",
438
+ "vision_resize_size": null,
439
+ "vocab_size": 128008,
440
+ "words_splitter_type": "whitespace"
441
+ }
jeff_decision.py ADDED
@@ -0,0 +1,79 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """The trained GLiFormer Large classification path used by Jeff.
2
+
3
+ This checkpoint's classification config uses CLS pooling, parent anchors,
4
+ no anchor refinement/normalization, linear anchor modeling, and dot scoring.
5
+ Under those exact settings the word-level RNN is computed upstream but cannot
6
+ influence classification logits. We still assert the settings at construction.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import torch
12
+ from torch import nn
13
+
14
+
15
+ class JeffDecision(nn.Module):
16
+ def __init__(self, model: nn.Module):
17
+ super().__init__()
18
+ config = model.config.classification_config
19
+ expected = {
20
+ "pooling_type": "cls",
21
+ "anchor_mode": "parent",
22
+ "anchor_modeling": "linear",
23
+ "anchor_normalization": "none",
24
+ "scorer_type": "dot",
25
+ "anchor_refine_layers": 0,
26
+ }
27
+ for name, value in expected.items():
28
+ actual = getattr(config, name)
29
+ if actual != value:
30
+ raise ValueError(f"unsupported classification config {name}={actual!r}; expected {value!r}")
31
+ if not config.embed_parent_token or not config.embed_cat_token:
32
+ raise ValueError("this path requires embeddings at the parent and category marker tokens")
33
+ if model.config.hidden_size != 1024:
34
+ raise ValueError("this fixed classifier requires the pinned 1024-wide checkpoint")
35
+ head = model.heads["classification"]
36
+ if hasattr(head, "anchor_refine"):
37
+ raise ValueError("classification anchor refinement cannot be omitted")
38
+ if type(head.anchor_layer).__name__ != "ParentAnchorLayer":
39
+ raise ValueError("unsupported anchor layer")
40
+ if type(head.anchor_modeling).__name__ != "LinearAnchorModeling":
41
+ raise ValueError("unsupported anchor model")
42
+ self.encoder = model.token_rep_layer
43
+ self.projection = head.anchor_modeling.proj
44
+
45
+ def forward(
46
+ self,
47
+ input_ids: torch.Tensor,
48
+ attention_mask: torch.Tensor,
49
+ parent_position: torch.Tensor,
50
+ category_positions: torch.Tensor,
51
+ ) -> torch.Tensor:
52
+ """Return unnormalized logits for one classification group, padded to C=8."""
53
+ encoded = self.encoder(input_ids.long(), attention_mask.long())
54
+ cls = encoded[:, 0, :]
55
+ parent_index = parent_position.long().unsqueeze(-1).expand(1, 1, 1024)
56
+ parent = torch.gather(encoded, 1, parent_index)
57
+ child_index = category_positions.long().unsqueeze(-1).expand(1, 8, 1024)
58
+ children = torch.gather(encoded, 1, child_index)
59
+ combined = torch.cat((parent.expand(1, 8, 1024), children), dim=-1)
60
+ fused = self.projection(combined)
61
+ return (cls.unsqueeze(1) * fused).sum(dim=-1)
62
+
63
+
64
+ def marker_positions(
65
+ input_ids: torch.Tensor, config, max_categories: int = 8
66
+ ) -> tuple[torch.Tensor, torch.Tensor, int]:
67
+ """Find the actual learned prompt markers in a one-row Jeff collator batch."""
68
+ if input_ids.shape[0] != 1:
69
+ raise ValueError("one classification group per call is required")
70
+ parent = torch.nonzero(input_ids[0] == config.classification_config.parent_token_index).flatten()
71
+ children = torch.nonzero(input_ids[0] == config.classification_config.cat_token_index).flatten()
72
+ if parent.numel() != 1 or not (1 <= children.numel() <= max_categories):
73
+ raise ValueError(
74
+ f"expected 1 parent and 1..{max_categories} categories; got {parent.numel()}, {children.numel()}"
75
+ )
76
+ count = int(children.numel())
77
+ category_positions = torch.zeros((1, max_categories), dtype=torch.int32)
78
+ category_positions[0, :count] = children.to(torch.int32)
79
+ return parent.to(torch.int32).view(1, 1), category_positions, count
native-parity.json ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "name": "billing",
4
+ "token_count": 39,
5
+ "labels": [
6
+ "billing: invoice or payment issue",
7
+ "support: technical product issue"
8
+ ],
9
+ "native_logits": [
10
+ 10.187458038330078,
11
+ -6.71131706237793
12
+ ],
13
+ "decision_logits": [
14
+ 10.187459945678711,
15
+ -6.71131706237793
16
+ ],
17
+ "max_logit_error": 1.9073486328125e-06
18
+ },
19
+ {
20
+ "name": "technical",
21
+ "token_count": 42,
22
+ "labels": [
23
+ "billing: invoice or payment issue",
24
+ "support: technical product issue"
25
+ ],
26
+ "native_logits": [
27
+ -12.322779655456543,
28
+ 13.379047393798828
29
+ ],
30
+ "decision_logits": [
31
+ -12.322783470153809,
32
+ 13.379046440124512
33
+ ],
34
+ "max_logit_error": 3.814697265625e-06
35
+ },
36
+ {
37
+ "name": "three_way",
38
+ "token_count": 41,
39
+ "labels": [
40
+ "schedule: appointment request",
41
+ "billing: payment issue",
42
+ "support: technical issue"
43
+ ],
44
+ "native_logits": [
45
+ 5.41853141784668,
46
+ -12.076669692993164,
47
+ -9.080745697021484
48
+ ],
49
+ "decision_logits": [
50
+ 5.418530464172363,
51
+ -12.076668739318848,
52
+ -9.080747604370117
53
+ ],
54
+ "max_logit_error": 1.9073486328125e-06
55
+ },
56
+ {
57
+ "name": "boolean",
58
+ "token_count": 29,
59
+ "labels": [
60
+ "yes",
61
+ "no"
62
+ ],
63
+ "native_logits": [
64
+ 1.0098832845687866,
65
+ -1.0589547157287598
66
+ ],
67
+ "decision_logits": [
68
+ 1.009883165359497,
69
+ -1.0589548349380493
70
+ ],
71
+ "max_logit_error": 1.1920928955078125e-07
72
+ }
73
+ ]
probe-native.py ADDED
@@ -0,0 +1,65 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Inspect Jeff's pinned, trained GLiFormer decision boundary using a real request."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from pathlib import Path
7
+
8
+ import torch
9
+ from huggingface_hub import snapshot_download
10
+ from jeff.backends.torch_backend import TorchBackend
11
+ from jeff.core.backend import Group
12
+
13
+ SOURCE_REPO = "knowledgator/gliformer-large-v1"
14
+ SOURCE_REVISION = "d0a4e53d09cebe6bc963dd9be319d4279084bb2d"
15
+
16
+
17
+ def main() -> None:
18
+ torch.set_num_threads(2)
19
+ checkpoint = snapshot_download(SOURCE_REPO, revision=SOURCE_REVISION)
20
+ backend = TorchBackend(checkpoint, device="cpu", dtype="float32", attn_kernel="eager", batch_size=1)
21
+ text = "The invoice was charged twice and the customer asks for a refund."
22
+ group = Group(
23
+ key="route",
24
+ labels=("billing: invoice or payment issue", "support: technical product issue"),
25
+ name="Choose the correct support queue",
26
+ )
27
+ native = backend.score([text], [[group]])[0]
28
+ tokens, _, _ = backend.model.prepare_inputs([text])
29
+ batch = backend._collator(
30
+ [
31
+ {
32
+ "tokenized_text": tokens[0],
33
+ "classification": [
34
+ {
35
+ "name": group.name,
36
+ "description": group.description,
37
+ "all_labels": list(group.labels),
38
+ "true_labels": [],
39
+ }
40
+ ],
41
+ }
42
+ ]
43
+ )
44
+ report = {
45
+ "source_repo": SOURCE_REPO,
46
+ "source_revision": SOURCE_REVISION,
47
+ "backend": backend.info(),
48
+ "scores": native.scores,
49
+ "input_tokens": native.input_tokens,
50
+ "model_parameters": sum(parameter.numel() for parameter in backend.model.model.parameters()),
51
+ "batch_tensors": {
52
+ name: {"shape": list(value.shape), "dtype": str(value.dtype)}
53
+ for name, value in batch.items()
54
+ if isinstance(value, torch.Tensor)
55
+ },
56
+ "batch_other": {name: str(type(value)) for name, value in batch.items() if not isinstance(value, torch.Tensor)},
57
+ }
58
+ output = Path("build/native-probe.json")
59
+ output.parent.mkdir(parents=True, exist_ok=True)
60
+ output.write_text(json.dumps(report, indent=2) + "\n")
61
+ print(json.dumps(report, indent=2), flush=True)
62
+
63
+
64
+ if __name__ == "__main__":
65
+ main()
pyproject.toml ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [project]
2
+ name = "jeff-coreml"
3
+ version = "0.1.0"
4
+ requires-python = ">=3.12,<3.13"
5
+ dependencies = [
6
+ "coremltools==9.0",
7
+ "gliformer==0.1.2",
8
+ "jeff @ git+https://github.com/logan-markewich/jeff.git@34b32f99a727c47b679adde33f4702a001e02979",
9
+ "numpy>=2.5.3",
10
+ "torch==2.7.0",
11
+ ]
12
+
13
+ [dependency-groups]
14
+ dev = ["pytest>=9.1", "ruff>=0.13"]
15
+
16
+ [tool.pytest.ini_options]
17
+ testpaths = ["tests"]
18
+ pythonpath = ["."]
19
+
20
+ [tool.ruff]
21
+ line-length = 120
22
+ target-version = "py312"
23
+
24
+ [tool.ruff.lint]
25
+ select = ["E", "F", "I"]
runtime.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Standalone Jeff classification inference from the Core ML package.
2
+
3
+ No original PyTorch checkpoint is loaded. Tokenization and the classification
4
+ prompt formatting use the pinned GLiFormer processor and tokenizer files.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ from pathlib import Path
11
+
12
+ import coremltools as ct
13
+ import numpy as np
14
+ from gliformer.config import GLiFormerConfig
15
+ from gliformer.processing.collator import resolve_gliformer_collator_class
16
+ from gliformer.processing.processor import resolve_gliformer_processor_class
17
+ from gliner.data_processing.tokenizer import WordsSplitter
18
+ from transformers import AutoTokenizer
19
+
20
+ from jeff_decision import marker_positions
21
+
22
+
23
+ class JeffCoreML:
24
+ """One-group, one-text Jeff classifier for 1–8 labels and at most 128 tokens."""
25
+
26
+ def __init__(self, asset_dir: str | Path, package: str | Path, compute_units=ct.ComputeUnit.ALL):
27
+ asset_dir = Path(asset_dir)
28
+ config = GLiFormerConfig(**json.loads((asset_dir / "gliner_config.json").read_text()))
29
+ tokenizer = AutoTokenizer.from_pretrained(asset_dir, local_files_only=True)
30
+ splitter = WordsSplitter(config.words_splitter_type)
31
+ processor_cls = resolve_gliformer_processor_class(config)
32
+ processor = processor_cls(config, tokenizer, splitter)
33
+ collator_cls = resolve_gliformer_collator_class(config)
34
+ self.collator = collator_cls(config, data_processor=processor, return_tokens=True, prepare_labels=False)
35
+ self.splitter = splitter
36
+ self.config = config
37
+ self.model = ct.models.MLModel(str(package), compute_units=compute_units)
38
+ self.output_name = self.model.get_spec().description.output[0].name
39
+
40
+ def score(self, text: str, labels: list[str], name: str = "", description: str = "") -> list[float]:
41
+ if not 1 <= len(labels) <= 8:
42
+ raise ValueError("JeffCoreML accepts 1 to 8 labels")
43
+ tokens = [word for word, _, _ in self.splitter(text)]
44
+ batch = self.collator([{
45
+ "tokenized_text": tokens,
46
+ "classification": [{
47
+ "name": name,
48
+ "description": description,
49
+ "all_labels": labels,
50
+ "true_labels": [],
51
+ }],
52
+ }])
53
+ ids = batch["input_ids"]
54
+ mask = batch["attention_mask"]
55
+ if ids.shape[1] > 128:
56
+ raise ValueError(f"JeffCoreML L128 bucket cannot fit {ids.shape[1]} tokens")
57
+ parent, children, count = marker_positions(ids, self.config, 8)
58
+ if count != len(labels):
59
+ raise ValueError("tokenizer label markers disagree with the supplied label count")
60
+ inputs = {
61
+ "input_ids": np.pad(ids.numpy().astype(np.int32), ((0, 0), (0, 128 - ids.shape[1]))),
62
+ "attention_mask": np.pad(mask.numpy().astype(np.int32), ((0, 0), (0, 128 - mask.shape[1]))),
63
+ "parent_position": parent.numpy(),
64
+ "category_positions": children.numpy(),
65
+ }
66
+ logits = self.model.predict(inputs)[self.output_name][0, :count].astype(np.float32)
67
+ probabilities = 1.0 / (1.0 + np.exp(-logits))
68
+ return probabilities.tolist()
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": true,
3
+ "backend": "tokenizers",
4
+ "bos_token": "[CLS]",
5
+ "clean_up_tokenization_spaces": false,
6
+ "cls_token": "[CLS]",
7
+ "do_lower_case": false,
8
+ "eos_token": "[SEP]",
9
+ "is_local": true,
10
+ "local_files_only": false,
11
+ "mask_token": "[MASK]",
12
+ "max_length": 1024,
13
+ "model_max_length": 1000000000000000019884624838656,
14
+ "pad_to_multiple_of": null,
15
+ "pad_token": "[PAD]",
16
+ "pad_token_type_id": 0,
17
+ "padding_side": "right",
18
+ "sep_token": "[SEP]",
19
+ "sp_model_kwargs": {},
20
+ "split_by_punct": false,
21
+ "stride": 0,
22
+ "tokenizer_class": "DebertaV2Tokenizer",
23
+ "truncation_side": "right",
24
+ "truncation_strategy": "longest_first",
25
+ "unk_id": 3,
26
+ "unk_token": "[UNK]",
27
+ "vocab_type": "spm"
28
+ }
trace_compat.py ADDED
@@ -0,0 +1,138 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Tracing-only DeBERTa relative attention for fixed batch 1.
2
+
3
+ This is adapted from Hugging Face Transformers' Apache-2.0
4
+ ``DisentangledSelfAttention.disentangled_attention_bias``. The trained weights
5
+ are untouched. The only semantic change is a literal repeat count of one for
6
+ the fixed B=1 Core ML export, avoiding an aten::Int conversion failure.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import math
12
+ from contextlib import contextmanager
13
+
14
+ import torch
15
+ from transformers.models.deberta_v2.modeling_deberta_v2 import build_relative_position
16
+
17
+
18
+ def constant_attention_scale(query_layer: torch.Tensor, scale_factor: int) -> torch.Tensor:
19
+ """Static head-width sqrt, equivalent to Transformers' fp32 calculation."""
20
+ return torch.tensor(
21
+ math.sqrt(query_layer.shape[-1] * scale_factor),
22
+ dtype=torch.float32,
23
+ device=query_layer.device,
24
+ )
25
+
26
+
27
+ def batch_one_disentangled_attention_bias(
28
+ self, query_layer, key_layer, relative_pos, rel_embeddings, scale_factor
29
+ ):
30
+ if query_layer.shape[0] != self.num_attention_heads:
31
+ raise ValueError("this tracing path requires batch size one")
32
+ if relative_pos is None:
33
+ relative_pos = build_relative_position(
34
+ query_layer,
35
+ key_layer,
36
+ bucket_size=self.position_buckets,
37
+ max_position=self.max_relative_positions,
38
+ )
39
+ if relative_pos.dim() == 2:
40
+ relative_pos = relative_pos.unsqueeze(0).unsqueeze(0)
41
+ elif relative_pos.dim() == 3:
42
+ relative_pos = relative_pos.unsqueeze(1)
43
+ elif relative_pos.dim() != 4:
44
+ raise ValueError(f"relative position ids must have 2, 3 or 4 dims; got {relative_pos.dim()}")
45
+
46
+ att_span = self.pos_ebd_size
47
+ relative_pos = relative_pos.to(device=query_layer.device, dtype=torch.long)
48
+ rel_embeddings = rel_embeddings[: att_span * 2, :].unsqueeze(0)
49
+ if self.share_att_key:
50
+ pos_query_layer = self.transpose_for_scores(
51
+ self.query_proj(rel_embeddings), self.num_attention_heads
52
+ ).repeat(1, 1, 1)
53
+ pos_key_layer = self.transpose_for_scores(
54
+ self.key_proj(rel_embeddings), self.num_attention_heads
55
+ ).repeat(1, 1, 1)
56
+ else:
57
+ if "c2p" in self.pos_att_type:
58
+ pos_key_layer = self.transpose_for_scores(
59
+ self.pos_key_proj(rel_embeddings), self.num_attention_heads
60
+ ).repeat(1, 1, 1)
61
+ if "p2c" in self.pos_att_type:
62
+ pos_query_layer = self.transpose_for_scores(
63
+ self.pos_query_proj(rel_embeddings), self.num_attention_heads
64
+ ).repeat(1, 1, 1)
65
+
66
+ score = 0
67
+ if "c2p" in self.pos_att_type:
68
+ scale = constant_attention_scale(pos_key_layer, scale_factor)
69
+ c2p_att = torch.bmm(query_layer, pos_key_layer.transpose(-1, -2))
70
+ c2p_pos = torch.clamp(relative_pos + att_span, 0, att_span * 2 - 1)
71
+ c2p_att = torch.gather(
72
+ c2p_att,
73
+ dim=-1,
74
+ index=c2p_pos.squeeze(0).expand(
75
+ [query_layer.size(0), query_layer.size(1), relative_pos.size(-1)]
76
+ ),
77
+ )
78
+ score += c2p_att / scale.to(dtype=c2p_att.dtype)
79
+
80
+ if "p2c" in self.pos_att_type:
81
+ scale = constant_attention_scale(pos_query_layer, scale_factor)
82
+ if query_layer.shape[-2] != key_layer.shape[-2]:
83
+ raise ValueError("fixed classification encoder requires equal query and key lengths")
84
+ r_pos = relative_pos
85
+ p2c_pos = torch.clamp(-r_pos + att_span, 0, att_span * 2 - 1)
86
+ p2c_att = torch.bmm(key_layer, pos_query_layer.transpose(-1, -2))
87
+ p2c_att = torch.gather(
88
+ p2c_att,
89
+ dim=-1,
90
+ index=p2c_pos.squeeze(0).expand(
91
+ [query_layer.size(0), key_layer.size(-2), key_layer.size(-2)]
92
+ ),
93
+ ).transpose(-1, -2)
94
+ score += p2c_att / scale.to(dtype=p2c_att.dtype)
95
+ return score
96
+
97
+
98
+ def install_trace_compatibility() -> None:
99
+ """Install fixed-shape tracing helpers; call only after native baseline."""
100
+ import gliformer.backbones.deberta_2d as gliformer_deberta
101
+ import transformers.models.deberta_v2.modeling_deberta_v2 as hf_deberta
102
+
103
+ gliformer_deberta.scaled_size_sqrt = constant_attention_scale
104
+ hf_deberta.scaled_size_sqrt = constant_attention_scale
105
+ gliformer_deberta.LayoutDisentangledSelfAttention.disentangled_attention_bias = (
106
+ batch_one_disentangled_attention_bias
107
+ )
108
+
109
+
110
+ @contextmanager
111
+ def finite_fp16_mask():
112
+ """Trace finite attention-mask fill instead of fp32 minimum overflowing to FP16 -inf.
113
+
114
+ Valid attention rows retain the same softmax in fp32. The unmasked native
115
+ baseline and patched model are compared on every parity fixture.
116
+ """
117
+ original = torch.finfo
118
+
119
+ class FiniteFinfo:
120
+ def __init__(self, real):
121
+ self.real = real
122
+
123
+ def __getattr__(self, name):
124
+ return getattr(self.real, name)
125
+
126
+ @property
127
+ def min(self):
128
+ return -10000.0
129
+
130
+ def patched(dtype):
131
+ info = original(dtype)
132
+ return FiniteFinfo(info) if dtype.is_floating_point else info
133
+
134
+ torch.finfo = patched
135
+ try:
136
+ yield
137
+ finally:
138
+ torch.finfo = original
uv.lock ADDED
The diff for this file is too large to render. See raw diff
 
verify.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Compare the exported Jeff FP16 Core ML package with its trained native model."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import time
8
+ from pathlib import Path
9
+
10
+ import coremltools as ct
11
+ import numpy as np
12
+ import torch
13
+ from huggingface_hub import snapshot_download
14
+ from jeff.backends.torch_backend import TorchBackend
15
+
16
+ from export import FIXTURES, REVISION, SOURCE, make_batch, model_inputs
17
+
18
+
19
+ def main() -> None:
20
+ parser = argparse.ArgumentParser()
21
+ parser.add_argument("--precision", choices=("fp16", "fp32"), default="fp16")
22
+ args = parser.parse_args()
23
+ torch.set_num_threads(2)
24
+ checkpoint = snapshot_download(SOURCE, revision=REVISION, local_files_only=True)
25
+ backend = TorchBackend(checkpoint, device="cpu", dtype="float32", attn_kernel="eager", batch_size=1)
26
+ package = Path(f"build/JeffDecision-L128-{args.precision.upper()}.mlpackage")
27
+ model = ct.models.MLModel(str(package), compute_units=ct.ComputeUnit.CPU_ONLY)
28
+ output_name = model.get_spec().description.output[0].name
29
+ results = []
30
+ for name, text, group in FIXTURES:
31
+ batch = make_batch(backend, text, group)
32
+ native = backend.model.model(**batch, include_media=False).cat_logits.detach().float().numpy()[0]
33
+ tensor_inputs = model_inputs(batch, backend.model.config)
34
+ inputs = {
35
+ key: tensor.numpy()
36
+ for key, tensor in zip(
37
+ ("input_ids", "attention_mask", "parent_position", "category_positions"),
38
+ tensor_inputs,
39
+ )
40
+ }
41
+ start = time.perf_counter()
42
+ converted = model.predict(inputs)[output_name][0, :len(group.labels)].astype(np.float32)
43
+ elapsed_ms = (time.perf_counter() - start) * 1000
44
+ error = float(np.max(np.abs(native - converted))) if np.isfinite(converted).all() else float("inf")
45
+ agreement = int(np.argmax(native) == np.argmax(converted))
46
+ results.append({
47
+ "name": name,
48
+ "labels": list(group.labels),
49
+ "native_logits": native.tolist(),
50
+ "coreml_logits": converted.tolist(),
51
+ "max_logit_error": error,
52
+ "top_label_agreement": bool(agreement),
53
+ "coreml_wall_ms": elapsed_ms,
54
+ })
55
+ print(
56
+ f"{name}: logits={converted.tolist()}, error={error:.6f}, "
57
+ f"top_label={bool(agreement)}, wall_ms={elapsed_ms:.1f}",
58
+ flush=True,
59
+ )
60
+ Path(f"build/coreml-parity-{args.precision}.json").write_text(json.dumps(results, indent=2) + "\n")
61
+ if not all(row["top_label_agreement"] for row in results):
62
+ raise AssertionError("Core ML changed the chosen label on a parity fixture")
63
+ if max(row["max_logit_error"] for row in results) > 0.25:
64
+ raise AssertionError("Core ML logit error exceeds the 0.25 tolerance")
65
+
66
+
67
+ if __name__ == "__main__":
68
+ main()