Extract and profile STINIT stage records
This commit is contained in:
@@ -1,7 +1,7 @@
|
|||||||
<!-- DO NOT EDIT -- generated from vm-map/globals.toml by tools/globals_build.py --build -->
|
<!-- DO NOT EDIT -- generated from vm-map/globals.toml by tools/globals_build.py --build -->
|
||||||
# Global Variable Reference (generated)
|
# Global Variable Reference (generated)
|
||||||
|
|
||||||
5023 globals (148 curated, 4875 auto shape-inferred). Source of truth: `vm-map/globals.toml`.
|
5026 globals (155 curated, 4871 auto shape-inferred). Source of truth: `vm-map/globals.toml`.
|
||||||
|
|
||||||
## choice-output
|
## choice-output
|
||||||
|
|
||||||
@@ -101,6 +101,9 @@
|
|||||||
| `0xaa8f2` | skill_proc_chance_percent | high | investigation | Probability for 14 passive skills. CALCDMG compares random-modulo 100 against this value; examples include Re-action 20, Double Action 100, Counter 10, and Resurrection 50. |
|
| `0xaa8f2` | skill_proc_chance_percent | high | investigation | Probability for 14 passive skills. CALCDMG compares random-modulo 100 against this value; examples include Re-action 20, Double Action 100, Counter 10, and Resurrection 50. |
|
||||||
| `0xaaa1e` | skill_battle_animation_id | high | investigation | Populated for 101 combat skills. BTL and CALCDMG place this value in the battle-animation selector before calling BTANINIT; most skills reuse their own id, while related skills deliberately share an animation and passive reactions use ids 801..808. |
|
| `0xaaa1e` | skill_battle_animation_id | high | investigation | Populated for 101 combat skills. BTL and CALCDMG place this value in the battle-animation selector before calling BTANINIT; most skills reuse their own id, while related skills deliberately share an animation and passive reactions use ids 801..808. |
|
||||||
| `0xaab4a` | skill_handler_script_id | high | investigation | SKINIT field for all 131 skills. CHMENU and INFOIT look it up and pass it directly to call-script; packed id 0x31ca resolves to SKMES.BIN, the shared per-skill text/behavior dispatcher. |
|
| `0xaab4a` | skill_handler_script_id | high | investigation | SKINIT field for all 131 skills. CHMENU and INFOIT look it up and pass it directly to call-script; packed id 0x31ca resolves to SKMES.BIN, the shared per-skill text/behavior dispatcher. |
|
||||||
|
| `0xe7302` | stage_bgm_id | high | investigation | STINIT's per-stage scalar loaded for all 74 records. FIELD passes the value directly to play-bgm when starting the stage. |
|
||||||
|
| `0xe730c` | stage_turn_limit | high | investigation | STINIT's per-stage turn limit. DRAWCHP presents the value in the stage information, while FIELD compares the current turn against it when checking stage completion. |
|
||||||
|
| `0xe730d` | stage_turn_limit_outcome | high | investigation | STINIT mode paired with stage_turn_limit. Stage 1 stores 0 and describes 50-turn expiry as defeat; stage 2 stores 1 and explicitly describes 15-turn expiry as a forced-retreat clear. |
|
||||||
| `0x5` | — | low | auto-shape | array |
|
| `0x5` | — | low | auto-shape | array |
|
||||||
| `0xd2` | — | low | auto-shape | array |
|
| `0xd2` | — | low | auto-shape | array |
|
||||||
| `0xd7` | — | low | auto-shape | array |
|
| `0xd7` | — | low | auto-shape | array |
|
||||||
@@ -412,6 +415,10 @@
|
|||||||
|
|
||||||
| address | name | conf | source | usage |
|
| address | name | conf | source | usage |
|
||||||
|---|---|---|---|---|
|
|---|---|---|---|---|
|
||||||
|
| `0x27b9` | stage_victory_condition_1 | high | investigation | STINIT writes one value for each of its 74 stage records. AIM renders this line first in the victory-condition section, and FIELD copies it into the current mission-condition display. |
|
||||||
|
| `0x27ba` | stage_victory_condition_2 | high | investigation | STINIT writes one value for each of its 74 stage records. AIM renders nonempty values after stage_victory_condition_1, and FIELD copies the slot into the current mission-condition display. |
|
||||||
|
| `0x27bb` | stage_defeat_condition_1 | high | investigation | STINIT writes one value for each of its 74 stage records. AIM renders this line first in the defeat-condition section, and FIELD copies it into the current mission-condition display. |
|
||||||
|
| `0x27bc` | stage_defeat_condition_2 | high | investigation | STINIT writes one value for each of its 74 stage records. AIM renders nonempty values after stage_defeat_condition_1, and FIELD copies the slot into the current mission-condition display. |
|
||||||
| `0x276` | — | med | auto-shape | string-table (written by SC0130) |
|
| `0x276` | — | med | auto-shape | string-table (written by SC0130) |
|
||||||
| `0x277` | — | med | auto-shape | string-table (written by SC0130) |
|
| `0x277` | — | med | auto-shape | string-table (written by SC0130) |
|
||||||
| `0x278` | — | med | auto-shape | string-table (written by FIELD) |
|
| `0x278` | — | med | auto-shape | string-table (written by FIELD) |
|
||||||
@@ -2624,10 +2631,6 @@
|
|||||||
| `0x277a` | — | med | auto-shape | string-table (written by OBINIT) |
|
| `0x277a` | — | med | auto-shape | string-table (written by OBINIT) |
|
||||||
| `0x2782` | — | med | auto-shape | string-table (written by OBINIT) |
|
| `0x2782` | — | med | auto-shape | string-table (written by OBINIT) |
|
||||||
| `0x2783` | — | med | auto-shape | string-table (written by OBINIT) |
|
| `0x2783` | — | med | auto-shape | string-table (written by OBINIT) |
|
||||||
| `0x27b9` | — | med | auto-shape | string-table (written by STINIT) |
|
|
||||||
| `0x27ba` | — | med | auto-shape | string-table (written by STINIT) |
|
|
||||||
| `0x27bb` | — | med | auto-shape | string-table (written by STINIT) |
|
|
||||||
| `0x27bc` | — | med | auto-shape | string-table (written by STINIT) |
|
|
||||||
| `0x27be` | — | med | auto-shape | string-table (written by STINIT2) |
|
| `0x27be` | — | med | auto-shape | string-table (written by STINIT2) |
|
||||||
| `0x27bf` | — | med | auto-shape | string-table (written by STINIT2) |
|
| `0x27bf` | — | med | auto-shape | string-table (written by STINIT2) |
|
||||||
| `0x27c0` | — | med | auto-shape | string-table (written by STINIT2) |
|
| `0x27c0` | — | med | auto-shape | string-table (written by STINIT2) |
|
||||||
|
|||||||
@@ -239,6 +239,29 @@ fallbacks are deliberately limited to the still-unresolved voice reactions and S
|
|||||||
three fields remain wholly anonymous (`0x7843e` and two suspicious sparse writes into runtime table
|
three fields remain wholly anonymous (`0x7843e` and two suspicious sparse writes into runtime table
|
||||||
`0x4e693/300`).
|
`0x4e693/300`).
|
||||||
|
|
||||||
|
### STINIT mixed stage records (2026-07-22)
|
||||||
|
|
||||||
|
STINIT is not a name table. Its preamble allocates 29 fixed global work buffers, then 74 sparse branches
|
||||||
|
compare `scjump_progress_a` with stage ids 1 through 170. Each selected branch populates the same current-
|
||||||
|
stage buffer with four strings, six scalar globals, sparse cells inside the fixed buffers, and
|
||||||
|
length-prefixed arrays copied from the script footer. `extract_init.py` now detects this shape as `mixed`,
|
||||||
|
evaluates preamble length arithmetic, attaches writes to their containing buffer, and preserves the branch
|
||||||
|
offset and footer offset as provenance. The extraction accounts for all 296 string writes and all 1,396
|
||||||
|
`copy-local-array` operations. Six buffers also inherit exact strides from independent
|
||||||
|
`lookup-array-2d` consumers.
|
||||||
|
|
||||||
|
`init_table_profile.py STINIT --build` profiles the four string slots, six scalars, 932 distinct buffer
|
||||||
|
cell destinations, and 37 footer-array destinations across the 74 records. The record label falls back to
|
||||||
|
the first nonempty victory-condition string, making consumer/value correlations readable without inventing
|
||||||
|
a stage-name field.
|
||||||
|
|
||||||
|
The strongest consumer-backed meanings are curated in `globals.toml`: `0x27b9..0x27bc` are the two
|
||||||
|
victory and two defeat-condition lines rendered by AIM/FIELD; `0xe7302` is passed by FIELD to `play-bgm`;
|
||||||
|
`0xe730c` is the turn limit displayed by DRAWCHP and checked by FIELD; and `0xe730d` selects defeat versus
|
||||||
|
forced-retreat clear when that limit expires. The remaining stage scalars and buffer families stay raw until
|
||||||
|
their FIELD/SETEN/SETOBJ/DRAWMAP consumers support stable names. In particular, repeated asset-like values
|
||||||
|
inside `0xe7311` are not promoted merely because they resolve through SYS4INI.
|
||||||
|
|
||||||
### The curated registry — `vm-map/globals.toml` (2026-07-07)
|
### The curated registry — `vm-map/globals.toml` (2026-07-07)
|
||||||
|
|
||||||
The v1 auto map (`build/global-var-map.json`) infers *shapes* but cannot recover branch-flag
|
The v1 auto map (`build/global-var-map.json`) infers *shapes* but cannot recover branch-flag
|
||||||
@@ -283,11 +306,12 @@ are *not* story flags — the miner over-tags them; they are recategorized `unkn
|
|||||||
|
|
||||||
The v1 map labels *shapes and tables*; the next increments add *meaning*, cheapest first:
|
The v1 map labels *shapes and tables*; the next increments add *meaning*, cheapest first:
|
||||||
|
|
||||||
1. **Continue INIT semantics by evidence density.** ITINIT/SKINIT and the confirmed EBINIT row layouts now
|
1. **Continue INIT semantics by evidence density.** ITINIT/SKINIT, the confirmed EBINIT row layouts, and
|
||||||
have machine-readable meanings. Next isolate EBINIT's voice-slot roles from their battle/map selection
|
STINIT's mixed stage records now have machine-readable investigation surfaces. Next trace STINIT's
|
||||||
branches and SALLY's action/event slots from their routed scenes; investigate the unread enum and boss-class
|
highest-coverage object/enemy/map buffers through FIELD, SETOBJ, SETEN, and DRAWMAP; isolate EBINIT's
|
||||||
sign only when consumer or native evidence appears. Preserve explicit item → skill and
|
remaining voice/action slots only when their routed consumers make them distinguishable. Investigate
|
||||||
unit → attack/skill/equipment/drop joins. Do not infer meaning from column position alone.
|
unread enums and signed classes only when consumer or native evidence appears. Preserve explicit joins
|
||||||
|
and do not infer meaning from column position alone.
|
||||||
2. **Extend message-table joins beyond the completed ITMES/SKMES pair** (`VIMES`, other id dispatchers, …)
|
2. **Extend message-table joins beyond the completed ITMES/SKMES pair** (`VIMES`, other id dispatchers, …)
|
||||||
and fold in other `set-string`/`copy-to-global` writers not covered by the `*INIT` set.
|
and fold in other `set-string`/`copy-to-global` writers not covered by the `*INIT` set.
|
||||||
3. **Label 2D record tables by their readers** — cross-reference which scripts read each
|
3. **Label 2D record tables by their readers** — cross-reference which scripts read each
|
||||||
|
|||||||
@@ -746,8 +746,10 @@ skill-information visibility. The remaining item/skill work is refinement rather
|
|||||||
schema.
|
schema.
|
||||||
|
|
||||||
The remaining EBINIT unknowns are now the unread `0x7843e` enum and the signed meaning within boss classes;
|
The remaining EBINIT unknowns are now the unread `0x7843e` enum and the signed meaning within boss classes;
|
||||||
enemy AI appears to live outside the static EBINIT schema. STINIT's bespoke parser remains a separate
|
enemy AI appears to live outside the static EBINIT schema. STINIT's separate mixed parser is now complete:
|
||||||
extraction task.
|
all 74 sparse stage ids retain their victory/defeat strings, six scalars, fixed-buffer cells, and 1,396
|
||||||
|
footer-array copies. AIM/FIELD/DRAWCHP consumers establish the four condition slots, stage BGM, turn limit,
|
||||||
|
and turn-limit outcome; the object/enemy/map buffer families remain an evidence-driven follow-up.
|
||||||
|
|
||||||
Once the natural spine and first gameplay loop are trustworthy, broaden in independent tracks:
|
Once the natural spine and first gameplay loop are trustworthy, broaden in independent tracks:
|
||||||
|
|
||||||
@@ -757,7 +759,8 @@ Once the natural spine and first gameplay loop are trustworthy, broaden in indep
|
|||||||
- Unit statistics, equipment, inventory, skills, magic, heroine forms, and progression.
|
- Unit statistics, equipment, inventory, skills, magic, heroine forms, and progression.
|
||||||
- Combat resolution presentation, enemy turns/AI services, and win/loss transitions.
|
- Combat resolution presentation, enemy turns/AI services, and win/loss transitions.
|
||||||
- A full chapter of ADV and the scene types encountered between gameplay segments.
|
- A full chapter of ADV and the scene types encountered between gameplay segments.
|
||||||
- `STINIT` and other data schemas when their runtime consumers make them necessary.
|
- Remaining STINIT object/enemy/map buffer semantics and other data schemas when their runtime consumers make
|
||||||
|
them necessary.
|
||||||
|
|
||||||
The high-level Phase B direction remains canonical in `docs/remake-architecture-and-roadmap.md`; this file
|
The high-level Phase B direction remains canonical in `docs/remake-architecture-and-roadmap.md`; this file
|
||||||
only provides the execution framework.
|
only provides the execution framework.
|
||||||
|
|||||||
@@ -67,6 +67,11 @@ preserves the scripts' zero-minus-immediate negative writes, which carry item pe
|
|||||||
skill SP costs. Generated records keep linked cells under `record_fields[base/stride/column]`; `fields`
|
skill SP costs. Generated records keep linked cells under `record_fields[base/stride/column]`; `fields`
|
||||||
contains only genuine parallel arrays.
|
contains only genuine parallel arrays.
|
||||||
|
|
||||||
|
STINIT uses the separate mixed shape: 74 sparse `scjump_progress_a` branches load one current-stage work
|
||||||
|
buffer rather than parallel per-id arrays. Its generated records preserve four mission-condition strings,
|
||||||
|
six scalars, cells in 29 preallocated buffers, and all 1,396 length-prefixed footer-array copies. Confirmed
|
||||||
|
stage field meanings and the evidence workflow live in `docs/name-resolution.md`.
|
||||||
|
|
||||||
### Message/string tables (`*MES`)
|
### Message/string tables (`*MES`)
|
||||||
`ITMES` (64 KB — item text), `VIMES` (43 KB), `EIMES` (37 KB), `SKMES` (31 KB — skill
|
`ITMES` (64 KB — item text), `VIMES` (43 KB), `EIMES` (37 KB), `SKMES` (31 KB — skill
|
||||||
text), `CIMES` (15 KB), `MAMES`, `INFOMES`, `MES` — where most translatable text
|
text), `CIMES` (15 KB), `MAMES`, `INFOMES`, `MES` — where most translatable text
|
||||||
|
|||||||
@@ -55,9 +55,9 @@ All opcode knowledge (ABI, semantics, provenance, `depends_on`) is hand-edited *
|
|||||||
|---|---|---|---|
|
|---|---|---|---|
|
||||||
| `extract_phase2.py` | Batch: disassembly + text corpora for every script. | `extract_phase2.py` | corpus → `build/disasm/*.asm`, `build/text/{dialogue.jsonl,strings.jsonl,*.strings.txt}`, `build/manifest.json` |
|
| `extract_phase2.py` | Batch: disassembly + text corpora for every script. | `extract_phase2.py` | corpus → `build/disasm/*.asm`, `build/text/{dialogue.jsonl,strings.jsonl,*.strings.txt}`, `build/manifest.json` |
|
||||||
| `extract_message_table.py` | Discover a repeated global-id dispatch chain such as ITMES/SKMES, reconstruct player-facing title/description lines (including furigana surface text and readings), and emit an ID-keyed message table with bytecode provenance. | `extract_message_table.py <MES> [OUTNAME]` | `<MES>.BIN` → `build/data/<OUTNAME>.json` |
|
| `extract_message_table.py` | Discover a repeated global-id dispatch chain such as ITMES/SKMES, reconstruct player-facing title/description lines (including furigana surface text and readings), and emit an ID-keyed message table with bytecode provenance. | `extract_message_table.py <MES> [OUTNAME]` | `<MES>.BIN` → `build/data/<OUTNAME>.json` |
|
||||||
| `extract_init.py` | Parse a `*INIT` data table (auto-detects name / numeric / footer shape). Name tables infer their reserved record span, preserve sparse one-based runtime ids, distinguish lookup bases from first written cells, statically evaluate direct and negative-value writes, and separate parallel `fields` from linked row-major `record_fields`. ITINIT and SKINIT join ITMES/SKMES messages; top-level `field_semantics` maps raw keys to canonical global/column names. Refreshes the generated data index. | `extract_init.py <TABLE> [OUTNAME] [--mode …]` | `<TABLE>.BIN` plus matching `<MES>.BIN` when supported + `build/globals.json` → `build/data/<OUTNAME>.json`, `build/data/README.md` |
|
| `extract_init.py` | Parse a `*INIT` data table (auto-detects name / numeric / footer / mixed shape). Name tables infer their reserved record span, preserve sparse one-based runtime ids, distinguish lookup bases from first written cells, statically evaluate direct and negative-value writes, and separate parallel `fields` from linked row-major `record_fields`. Mixed tables recover selector-dispatched records, condition strings, scalars, preallocated buffer cells, consumer-confirmed strides, and length-prefixed footer arrays; STINIT is the first such table. ITINIT and SKINIT join ITMES/SKMES messages; top-level `field_semantics` maps raw keys to canonical global/column names. Refreshes the generated data index. | `extract_init.py <TABLE> [OUTNAME] [--mode …]` | `<TABLE>.BIN` plus matching `<MES>.BIN` when supported + `build/globals.json` → `build/data/<OUTNAME>.json`, `build/data/README.md` |
|
||||||
| `init_table_profile.py` | Build the static investigation surface for an extracted name/numeric table: message coverage, per-array and per-record-column population/value distributions, representative records with their player-facing descriptions, and direct opcode/script consumers. `--message-query REGEX` shows every matching name/message beside all populated fields for semantic correlation. Findings are evidence only; confirmed meanings go in `vm-map/globals.toml`. | `init_table_profile.py <TABLE> [--build] [--limit N] [--message-query REGEX]` | `build/data/<TABLE>.json` + corpus → stdout; with `--build`, `build/data/<TABLE>-field-profile.{json,md}` |
|
| `init_table_profile.py` | Build the static investigation surface for an extracted name/numeric/mixed table: message coverage, per-scalar/string/array-cell/footer-array population and value distributions, representative records, and direct opcode/script consumers. `--message-query REGEX` shows every matching name/message beside all populated fields for semantic correlation. Findings are evidence only; confirmed meanings go in `vm-map/globals.toml`. | `init_table_profile.py <TABLE> [--build] [--limit N] [--message-query REGEX]` | `build/data/<TABLE>.json` + corpus → stdout; with `--build`, `build/data/<TABLE>-field-profile.{json,md}` |
|
||||||
| `test_extract_init.py`, `test_init_table_profile.py` | Regression checks for sparse one-based INIT extraction, MES dispatch reconstruction/joins, and field/message profiling. | run each directly | — |
|
| `test_extract_init.py`, `test_init_table_profile.py` | Regression checks for sparse one-based and mixed selector-dispatched INIT extraction, MES reconstruction/joins, footer-array accounting, and field/message profiling. | run each directly | — |
|
||||||
| `global_map.py` | Build the partial global-variable name map from static evidence. | `global_map.py` | corpus + `build/data/` → `build/global-var-map.{json,md}` |
|
| `global_map.py` | Build the partial global-variable name map from static evidence. | `global_map.py` | corpus + `build/data/` → `build/global-var-map.{json,md}` |
|
||||||
|
|
||||||
## VM
|
## VM
|
||||||
|
|||||||
@@ -53,8 +53,8 @@
|
|||||||
|
|
||||||
- [x] **2.0 — Project structure.** Established `docs/`, `build/{disasm,text,data,scripts-json}/`, `godot/`; game install stays read-only in place. Also relaxed the loader magic check to the `SYS4` family (`SYS4424` patch scripts now parse — was silently skipping 5 scripts).
|
- [x] **2.0 — Project structure.** Established `docs/`, `build/{disasm,text,data,scripts-json}/`, `godot/`; game install stays read-only in place. Also relaxed the loader magic check to the `SYS4` family (`SYS4424` patch scripts now parse — was silently skipping 5 scripts).
|
||||||
- [x] **2.1 — Text corpora.** `tools/extract_phase2.py` → 481/481 scripts: full disassembly (`build/disasm/*.asm`), per-script strings, `build/text/dialogue.jsonl` (**30,057 show-text lines** — the translation corpus), `build/text/strings.jsonl` (38,449 strings tagged by source opcode), `build/manifest.json`.
|
- [x] **2.1 — Text corpora.** `tools/extract_phase2.py` → 481/481 scripts: full disassembly (`build/disasm/*.asm`), per-script strings, `build/text/dialogue.jsonl` (**30,057 show-text lines** — the translation corpus), `build/text/strings.jsonl` (38,449 strings tagged by source opcode), `build/manifest.json`.
|
||||||
- [x] **2.2 — `*INIT` data tables → JSON.** `tools/extract_init.py` auto-detects table shape (`name`/`numeric`/`footer`) → **SKINIT (129 skills), ITINIT (189 items), EBINIT (277 units)** [name: name+desc+fields], **CGINIT (379 CG entries)** [numeric: index-keyed columns], **MPINIT (1472 map records)** [footer: 50-value arrays from the file footer]. Validated; see `build/data/README.md`. Column addresses are raw engine globals — naming them (attack/cost/…) needs the global-var map (Phase 3-adjacent).
|
- [x] **2.2 — `*INIT` data tables → JSON.** `tools/extract_init.py` auto-detects table shape (`name`/`numeric`/`footer`/`mixed`) → **SKINIT (131 skills), ITINIT (287 items), EBINIT (277 units)** [name: sparse one-based name/description/fields], **CGINIT (379 CG entries)** [numeric: index-keyed columns], **MPINIT (1472 map records)** [footer: length-prefixed arrays], and **STINIT (74 stages)** [mixed: selector-dispatched strings/scalars/buffer cells/footer arrays]. Validated; see `build/data/README.md`. Raw addresses remain bytecode provenance; confirmed semantics come from `vm-map/globals.toml`.
|
||||||
- [ ] **2.3 — `STINIT` (74 stages) needs a bespoke parser.** Heterogeneous per-stage `copy-to-global` param blocks + a variable number of clear-condition strings per stage — fits none of the three auto modes. Its clear-condition text is already in `build/text/STINIT.strings.txt`; only the per-stage numeric params await a custom extractor. Low priority (stages are also encoded in the SC-scene scripts).
|
- [x] **2.3 — Extract `STINIT`'s 74 sparse stage records.** The mixed mode identifies the dominant `scjump_progress_a` dispatch, recovers 29 preallocated buffer layouts (including six consumer-confirmed row strides), and keeps four condition strings, six scalars, fixed-buffer writes, and all 1,396 footer-array copies separated by stage id. The generated profile supplies population/value and direct-consumer evidence; victory/defeat lines, stage BGM, turn limit, and expiry outcome are curated.
|
||||||
- [x] **2.4 — Partial global-var map BUILT + wired into the disassembler.** `tools/global_map.py` → `build/global-var-map.{json,md}` (16,354/49,435 globals labelled: string tables, `*INIT` field arrays, 122 record tables w/ strides, current-entity index pointers). `sys4load` renders the labels inline (`=rec[s30]`, `=current-entity-index?`). See `docs/name-resolution.md`.
|
- [x] **2.4 — Partial global-var map BUILT + wired into the disassembler.** `tools/global_map.py` → `build/global-var-map.{json,md}` (16,354/49,435 globals labelled: string tables, `*INIT` field arrays, 122 record tables w/ strides, current-entity index pointers). `sys4load` renders the labels inline (`=rec[s30]`, `=current-entity-index?`). See `docs/name-resolution.md`.
|
||||||
- [ ] **2.5 — Grow the global-var map (future, incremental).** Static first: fold in `*MES` writers; label 2D record tables by their reader scripts. Then Frida to name *which stat* each field is. Full detail: `docs/name-resolution.md` → "Future step — growing the map". Also deferred: `call-script` id→name resolution (engine-level — SCJUMP.BIN decode or Frida; see `docs/name-resolution.md` #1).
|
- [ ] **2.5 — Grow the global-var map (future, incremental).** Static first: fold in `*MES` writers; label 2D record tables by their reader scripts. Then Frida to name *which stat* each field is. Full detail: `docs/name-resolution.md` → "Future step — growing the map". Also deferred: `call-script` id→name resolution (engine-level — SCJUMP.BIN decode or Frida; see `docs/name-resolution.md` #1).
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""Extract a *INIT data table to JSON. Auto-detects the table's shape.
|
"""Extract a *INIT data table to JSON. Auto-detects the table's shape.
|
||||||
|
|
||||||
*INIT scripts populate parallel global arrays with static game data. Three shapes seen:
|
*INIT scripts populate global arrays and work buffers with static game data. Four shapes seen:
|
||||||
|
|
||||||
name — records keyed by a name string. Each record: set-string(name), static field writes,
|
name — records keyed by a name string. Each record: set-string(name), static field writes,
|
||||||
set-string(desc). Arrays indexed by record id in lockstep (+1/record).
|
set-string(desc). Arrays indexed by record id in lockstep (+1/record).
|
||||||
@@ -10,12 +10,14 @@
|
|||||||
by an incrementing index column. (CGINIT gallery)
|
by an incrementing index column. (CGINIT gallery)
|
||||||
footer — copy-local-array (op 0x64) bulk-loads length-prefixed arrays from the file
|
footer — copy-local-array (op 0x64) bulk-loads length-prefixed arrays from the file
|
||||||
footer into per-record global arrays. The data lives in the footer. (MPINIT maps)
|
footer into per-record global arrays. The data lives in the footer. (MPINIT maps)
|
||||||
|
mixed — a sparse selector dispatch writes strings, scalars, fixed-buffer cells, and
|
||||||
|
footer arrays for one runtime record. (STINIT stages)
|
||||||
|
|
||||||
Records are {id, name?, desc?, fields:{"0x<col_base>": value}} or, for footer tables,
|
Records are {id, name?, desc?, fields:{"0x<col_base>": value}} or, for footer tables,
|
||||||
{id, global_addr, footer_off, values:[...]}. Column addresses are raw engine globals;
|
{id, global_addr, footer_off, values:[...]}. Column addresses are raw engine globals;
|
||||||
naming them (attack, cost, …) needs the engine global-var map — later work.
|
naming them (attack, cost, …) needs the engine global-var map — later work.
|
||||||
|
|
||||||
Usage: py -3.11 -X utf8 tools/extract_init.py <TABLE> [OUTNAME] [--mode name|numeric|footer]
|
Usage: py -3.11 -X utf8 tools/extract_init.py <TABLE> [OUTNAME] [--mode name|numeric|footer|mixed]
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
import json
|
import json
|
||||||
@@ -37,6 +39,7 @@ COPY_LOCAL_ARRAY = 0x64
|
|||||||
T_GLOBAL_INT = 3
|
T_GLOBAL_INT = 3
|
||||||
T_GLOBAL_STRING = 5
|
T_GLOBAL_STRING = 5
|
||||||
T_IMM = 0
|
T_IMM = 0
|
||||||
|
T_LOCAL_INT = 9
|
||||||
|
|
||||||
MESSAGE_TABLES = {
|
MESSAGE_TABLES = {
|
||||||
"ITINIT": "ITMES",
|
"ITINIT": "ITMES",
|
||||||
@@ -85,10 +88,44 @@ def read_footer_array(scr, off):
|
|||||||
return list(dw[off + 1: off + 1 + length])
|
return list(dw[off + 1: off + 1 + length])
|
||||||
|
|
||||||
|
|
||||||
|
def _mixed_guards(scr):
|
||||||
|
"""Find the dominant `eq local, selector-global, record-id; jcc` dispatch."""
|
||||||
|
candidates = []
|
||||||
|
instructions = scr.instructions
|
||||||
|
for index, ins in enumerate(instructions[:-1]):
|
||||||
|
if (sys4load.display_label(ins.opcode) != "eq"
|
||||||
|
or len(ins.args) < 3
|
||||||
|
or ins.args[0][0] != T_LOCAL_INT
|
||||||
|
or ins.args[1][0] != T_GLOBAL_INT
|
||||||
|
or ins.args[2][0] != T_IMM):
|
||||||
|
continue
|
||||||
|
branch = instructions[index + 1]
|
||||||
|
if (sys4load.display_label(branch.opcode) != "jcc"
|
||||||
|
or not branch.args
|
||||||
|
or branch.args[0] != ins.args[0]):
|
||||||
|
continue
|
||||||
|
candidates.append({
|
||||||
|
"index": index,
|
||||||
|
"offset": ins.offset,
|
||||||
|
"selector": ins.args[1][1],
|
||||||
|
"id": ins.args[2][1],
|
||||||
|
})
|
||||||
|
if not candidates:
|
||||||
|
return []
|
||||||
|
selector_counts = {}
|
||||||
|
for guard in candidates:
|
||||||
|
selector = guard["selector"]
|
||||||
|
selector_counts[selector] = selector_counts.get(selector, 0) + 1
|
||||||
|
selector = max(selector_counts, key=lambda value: (selector_counts[value], -value))
|
||||||
|
return [guard for guard in candidates if guard["selector"] == selector]
|
||||||
|
|
||||||
|
|
||||||
def detect_mode(scr):
|
def detect_mode(scr):
|
||||||
ops = [ins.opcode for ins in scr.instructions]
|
ops = [ins.opcode for ins in scr.instructions]
|
||||||
has_str = any(ins.opcode == SET_STRING and ins.args and ins.args[0][0] == T_GLOBAL_STRING
|
has_str = any(ins.opcode == SET_STRING and ins.args and ins.args[0][0] == T_GLOBAL_STRING
|
||||||
for ins in scr.instructions)
|
for ins in scr.instructions)
|
||||||
|
if has_str and len(_mixed_guards(scr)) >= 4:
|
||||||
|
return "mixed"
|
||||||
if has_str:
|
if has_str:
|
||||||
return "name"
|
return "name"
|
||||||
n_footer = ops.count(COPY_LOCAL_ARRAY)
|
n_footer = ops.count(COPY_LOCAL_ARRAY)
|
||||||
@@ -96,6 +133,150 @@ def detect_mode(scr):
|
|||||||
return "footer" if n_footer >= max(4, n_int) else "numeric"
|
return "footer" if n_footer >= max(4, n_int) else "numeric"
|
||||||
|
|
||||||
|
|
||||||
|
def _eval_static_arg(arg, locals_: dict[int, int]):
|
||||||
|
arg_type, value = arg
|
||||||
|
if arg_type == T_IMM:
|
||||||
|
return value
|
||||||
|
if arg_type == T_LOCAL_INT:
|
||||||
|
return locals_.get(value)
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _mixed_array_layouts(scr, first_guard_index: int) -> dict[int, dict]:
|
||||||
|
"""Recover fixed global-buffer lengths initialized before the dispatch."""
|
||||||
|
locals_: dict[int, int] = {}
|
||||||
|
layouts: dict[int, dict] = {}
|
||||||
|
for ins in scr.instructions[:first_guard_index]:
|
||||||
|
label = sys4load.display_label(ins.opcode)
|
||||||
|
if ins.args and ins.args[0][0] == T_LOCAL_INT:
|
||||||
|
destination = ins.args[0][1]
|
||||||
|
operands = [_eval_static_arg(arg, locals_) for arg in ins.args[1:]]
|
||||||
|
value = None
|
||||||
|
if label == "mov" and operands:
|
||||||
|
value = operands[0]
|
||||||
|
elif len(operands) >= 2 and None not in operands[:2]:
|
||||||
|
left, right = operands[:2]
|
||||||
|
if label == "add":
|
||||||
|
value = left + right
|
||||||
|
elif label == "sub":
|
||||||
|
value = left - right
|
||||||
|
elif label == "mul":
|
||||||
|
value = left * right
|
||||||
|
elif label == "div" and right:
|
||||||
|
value = left // right
|
||||||
|
if value is None:
|
||||||
|
locals_.pop(destination, None)
|
||||||
|
else:
|
||||||
|
locals_[destination] = value
|
||||||
|
if (ins.opcode == COPY_TO_GLOBAL
|
||||||
|
and len(ins.args) >= 2
|
||||||
|
and ins.args[0][0] == T_GLOBAL_INT):
|
||||||
|
length = _eval_static_arg(ins.args[1], locals_)
|
||||||
|
if isinstance(length, int) and length > 0:
|
||||||
|
layouts[ins.args[0][1]] = {"length": length}
|
||||||
|
|
||||||
|
known = dict(_known_record_tables())
|
||||||
|
for base, layout in layouts.items():
|
||||||
|
if stride := known.get(base):
|
||||||
|
layout["stride"] = stride
|
||||||
|
if layout["length"] % stride == 0:
|
||||||
|
layout["rows"] = layout["length"] // stride
|
||||||
|
return layouts
|
||||||
|
|
||||||
|
|
||||||
|
def _mixed_buffer_key(destination: int, layouts: dict[int, dict]) -> str | None:
|
||||||
|
matches = [
|
||||||
|
(base, destination - base)
|
||||||
|
for base, layout in layouts.items()
|
||||||
|
if base <= destination < base + layout["length"]
|
||||||
|
]
|
||||||
|
if len(matches) > 1:
|
||||||
|
raise ValueError(f"ambiguous mixed-table destination 0x{destination:x}: {matches}")
|
||||||
|
if not matches:
|
||||||
|
return None
|
||||||
|
base, index = matches[0]
|
||||||
|
return f"0x{base:x}/{index}"
|
||||||
|
|
||||||
|
|
||||||
|
def _store_unique(target: dict, key: str, value, record_id: int) -> None:
|
||||||
|
if key in target and target[key] != value:
|
||||||
|
raise ValueError(f"mixed record {record_id}: conflicting writes to {key}")
|
||||||
|
target[key] = value
|
||||||
|
|
||||||
|
|
||||||
|
def extract_mixed(scr):
|
||||||
|
"""Extract selector-dispatched records that populate a shared runtime buffer."""
|
||||||
|
guards = _mixed_guards(scr)
|
||||||
|
if not guards:
|
||||||
|
return [], {}
|
||||||
|
layouts = _mixed_array_layouts(scr, guards[0]["index"])
|
||||||
|
records = []
|
||||||
|
instructions = scr.instructions
|
||||||
|
for guard_index, guard in enumerate(guards):
|
||||||
|
end = guards[guard_index + 1]["index"] if guard_index + 1 < len(guards) else len(instructions)
|
||||||
|
record = {
|
||||||
|
"id": guard["id"],
|
||||||
|
"guard_offset": f"0x{guard['offset']:x}",
|
||||||
|
"string_fields": {},
|
||||||
|
"fields": {},
|
||||||
|
"array_fields": {},
|
||||||
|
"footer_arrays": {},
|
||||||
|
}
|
||||||
|
for ins in instructions[guard["index"] + 2:end]:
|
||||||
|
if (ins.opcode == SET_STRING
|
||||||
|
and len(ins.args) >= 2
|
||||||
|
and ins.args[0][0] == T_GLOBAL_STRING):
|
||||||
|
text = scr.strings.get(ins.args[1][1], (None,))[0]
|
||||||
|
_store_unique(
|
||||||
|
record["string_fields"], f"0x{ins.args[0][1]:x}", text, record["id"]
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
if (ins.opcode == COPY_LOCAL_ARRAY
|
||||||
|
and len(ins.args) >= 2
|
||||||
|
and ins.args[0][0] == T_GLOBAL_INT
|
||||||
|
and ins.args[1][0] == T_IMM):
|
||||||
|
destination = ins.args[0][1]
|
||||||
|
footer_off = ins.args[1][1]
|
||||||
|
values = read_footer_array(scr, footer_off)
|
||||||
|
if values is None:
|
||||||
|
raise ValueError(
|
||||||
|
f"mixed record {record['id']}: invalid footer array 0x{footer_off:x}"
|
||||||
|
)
|
||||||
|
key = _mixed_buffer_key(destination, layouts) or f"0x{destination:x}"
|
||||||
|
_store_unique(record["footer_arrays"], key, {
|
||||||
|
"footer_off": f"0x{footer_off:x}",
|
||||||
|
"values": values,
|
||||||
|
}, record["id"])
|
||||||
|
continue
|
||||||
|
if (write := _static_global_write(ins)) is not None:
|
||||||
|
destination, value = write
|
||||||
|
key = _mixed_buffer_key(destination, layouts)
|
||||||
|
target = record["array_fields"] if key else record["fields"]
|
||||||
|
_store_unique(target, key or f"0x{destination:x}", value, record["id"])
|
||||||
|
for key in ("string_fields", "fields", "array_fields", "footer_arrays"):
|
||||||
|
if not record[key]:
|
||||||
|
del record[key]
|
||||||
|
records.append(record)
|
||||||
|
|
||||||
|
layouts_json = {
|
||||||
|
f"0x{base:x}": layout for base, layout in sorted(layouts.items())
|
||||||
|
}
|
||||||
|
key_sort = lambda key: tuple(int(part, 0) for part in key.split("/"))
|
||||||
|
return records, {
|
||||||
|
"selector_global": f"0x{guards[0]['selector']:x}",
|
||||||
|
"array_layouts": layouts_json,
|
||||||
|
"string_field_columns": sorted({
|
||||||
|
key for record in records for key in record.get("string_fields", {})
|
||||||
|
}, key=lambda key: int(key, 16)),
|
||||||
|
"array_field_columns": sorted({
|
||||||
|
key for record in records for key in record.get("array_fields", {})
|
||||||
|
}, key=key_sort),
|
||||||
|
"footer_array_columns": sorted({
|
||||||
|
key for record in records for key in record.get("footer_arrays", {})
|
||||||
|
}, key=key_sort),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def _infer_record_span(string_addrs):
|
def _infer_record_span(string_addrs):
|
||||||
"""Infer the reserved width of one parallel string-array column.
|
"""Infer the reserved width of one parallel string-array column.
|
||||||
|
|
||||||
@@ -294,7 +475,13 @@ def field_semantics(records: list[dict]) -> dict[str, str]:
|
|||||||
keys = {
|
keys = {
|
||||||
key
|
key
|
||||||
for record in records
|
for record in records
|
||||||
for key in (*record.get("fields", {}), *record.get("record_fields", {}))
|
for key in (
|
||||||
|
*record.get("string_fields", {}),
|
||||||
|
*record.get("fields", {}),
|
||||||
|
*record.get("array_fields", {}),
|
||||||
|
*record.get("footer_arrays", {}),
|
||||||
|
*record.get("record_fields", {}),
|
||||||
|
)
|
||||||
}
|
}
|
||||||
registry = _global_registry()
|
registry = _global_registry()
|
||||||
semantics = {}
|
semantics = {}
|
||||||
@@ -306,7 +493,9 @@ def field_semantics(records: list[dict]) -> dict[str, str]:
|
|||||||
name = entry.get("name")
|
name = entry.get("name")
|
||||||
if not name:
|
if not name:
|
||||||
continue
|
continue
|
||||||
if len(parts) == 3:
|
if len(parts) == 2:
|
||||||
|
name = f"{name}.index_{parts[1]}"
|
||||||
|
elif len(parts) == 3:
|
||||||
column = parts[2]
|
column = parts[2]
|
||||||
column_name = entry.get("columns", {}).get(column, f"column_{column}")
|
column_name = entry.get("columns", {}).get(column, f"column_{column}")
|
||||||
name = f"{name}.{column_name}"
|
name = f"{name}.{column_name}"
|
||||||
@@ -334,8 +523,8 @@ def write_data_index(data_dir: Path) -> None:
|
|||||||
"bases remain available in every record; confirmed field meanings live in",
|
"bases remain available in every record; confirmed field meanings live in",
|
||||||
"`vm-map/globals.toml` and the generated `docs/global-reference.md`.",
|
"`vm-map/globals.toml` and the generated `docs/global-reference.md`.",
|
||||||
"",
|
"",
|
||||||
"| file | mode | records | messages | array fields | record columns |",
|
"| file | mode | records | messages | scalar/array fields | strings | buffer cells | footer arrays | record columns |",
|
||||||
"|---|---|---:|---:|---:|---:|",
|
"|---|---|---:|---:|---:|---:|---:|---:|---:|",
|
||||||
]
|
]
|
||||||
for filename, data in tables:
|
for filename, data in tables:
|
||||||
columns = len(data.get("field_columns") or [])
|
columns = len(data.get("field_columns") or [])
|
||||||
@@ -345,7 +534,9 @@ def write_data_index(data_dir: Path) -> None:
|
|||||||
)
|
)
|
||||||
lines.append(
|
lines.append(
|
||||||
f"| `{filename}` | {data['mode']} | {data['record_count']} | "
|
f"| `{filename}` | {data['mode']} | {data['record_count']} | "
|
||||||
f"{message_count} | {columns} | {record_columns} |"
|
f"{message_count} | {columns} | {len(data.get('string_field_columns') or [])} | "
|
||||||
|
f"{len(data.get('array_field_columns') or [])} | "
|
||||||
|
f"{len(data.get('footer_array_columns') or [])} | {record_columns} |"
|
||||||
)
|
)
|
||||||
lines += [
|
lines += [
|
||||||
"",
|
"",
|
||||||
@@ -360,16 +551,31 @@ def write_data_index(data_dir: Path) -> None:
|
|||||||
"Top-level `field_semantics` maps raw array/row-column keys to canonical machine-readable",
|
"Top-level `field_semantics` maps raw array/row-column keys to canonical machine-readable",
|
||||||
"names from `vm-map/globals.toml`; raw keys remain intact as bytecode provenance.",
|
"names from `vm-map/globals.toml`; raw keys remain intact as bytecode provenance.",
|
||||||
"",
|
"",
|
||||||
|
"Mixed-mode tables preserve the sparse selector id, branch offset, condition strings,",
|
||||||
|
"scalar fields, cells within preallocated buffers, and length-prefixed footer arrays.",
|
||||||
"Use `tools/init_table_profile.py <TABLE> --build` to generate value/population and",
|
"Use `tools/init_table_profile.py <TABLE> --build` to generate value/population and",
|
||||||
"direct-consumer evidence. `STINIT` still requires a bespoke mixed numeric/string parser.",
|
"direct-consumer evidence.",
|
||||||
"",
|
"",
|
||||||
]
|
]
|
||||||
(data_dir / "README.md").write_text("\n".join(lines), encoding="utf8")
|
(data_dir / "README.md").write_text("\n".join(lines), encoding="utf8")
|
||||||
|
|
||||||
|
|
||||||
def main() -> int:
|
def main() -> int:
|
||||||
argv = [a for a in sys.argv[1:] if not a.startswith("--")]
|
argv = []
|
||||||
mode_arg = next((sys.argv[i + 1] for i, a in enumerate(sys.argv) if a == "--mode"), None)
|
mode_arg = None
|
||||||
|
index = 1
|
||||||
|
while index < len(sys.argv):
|
||||||
|
arg = sys.argv[index]
|
||||||
|
if arg == "--mode":
|
||||||
|
if index + 1 >= len(sys.argv):
|
||||||
|
raise SystemExit("--mode requires a value")
|
||||||
|
mode_arg = sys.argv[index + 1]
|
||||||
|
index += 2
|
||||||
|
continue
|
||||||
|
if arg.startswith("--"):
|
||||||
|
raise SystemExit(f"unknown option: {arg}")
|
||||||
|
argv.append(arg)
|
||||||
|
index += 1
|
||||||
if not argv:
|
if not argv:
|
||||||
raise SystemExit(__doc__)
|
raise SystemExit(__doc__)
|
||||||
name = argv[0].upper().removesuffix(".BIN")
|
name = argv[0].upper().removesuffix(".BIN")
|
||||||
@@ -377,7 +583,12 @@ def main() -> int:
|
|||||||
scr = sys4load.load(resolve(name))
|
scr = sys4load.load(resolve(name))
|
||||||
|
|
||||||
mode = mode_arg or detect_mode(scr)
|
mode = mode_arg or detect_mode(scr)
|
||||||
extractor = {"name": extract_name, "numeric": extract_numeric, "footer": extract_footer}[mode]
|
extractor = {
|
||||||
|
"name": extract_name,
|
||||||
|
"numeric": extract_numeric,
|
||||||
|
"footer": extract_footer,
|
||||||
|
"mixed": extract_mixed,
|
||||||
|
}[mode]
|
||||||
recs, meta = extractor(scr)
|
recs, meta = extractor(scr)
|
||||||
if mode == "name" and name in MESSAGE_TABLES:
|
if mode == "name" and name in MESSAGE_TABLES:
|
||||||
message_name = MESSAGE_TABLES[name]
|
message_name = MESSAGE_TABLES[name]
|
||||||
@@ -403,7 +614,8 @@ def main() -> int:
|
|||||||
print(f" id {r['id']:>4} {r['global_addr']} <- footer {r['footer_off']} "
|
print(f" id {r['id']:>4} {r['global_addr']} <- footer {r['footer_off']} "
|
||||||
f"len {r['length']} head={r['values'][:8]}")
|
f"len {r['length']} head={r['values'][:8]}")
|
||||||
else:
|
else:
|
||||||
f4 = {k: r['fields'][k] for k in list(r['fields'])[:4]}
|
fields = r.get("fields", {})
|
||||||
|
f4 = {k: fields[k] for k in list(fields)[:4]}
|
||||||
print(f" id {r['id']:>4} {r.get('name','')!r:12} desc={r.get('desc','')!r} {f4}")
|
print(f" id {r['id']:>4} {r.get('name','')!r:12} desc={r.get('desc','')!r} {f4}")
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|||||||
@@ -37,8 +37,8 @@ def load_table(name: str) -> dict:
|
|||||||
if not path.exists():
|
if not path.exists():
|
||||||
raise SystemExit(f"missing extracted table: {path}")
|
raise SystemExit(f"missing extracted table: {path}")
|
||||||
data = json.loads(path.read_text(encoding="utf8"))
|
data = json.loads(path.read_text(encoding="utf8"))
|
||||||
if data.get("mode") not in {"name", "numeric"}:
|
if data.get("mode") not in {"name", "numeric", "mixed"}:
|
||||||
raise SystemExit(f"{name}: field profiling requires name/numeric mode")
|
raise SystemExit(f"{name}: field profiling requires name/numeric/mixed mode")
|
||||||
return data
|
return data
|
||||||
|
|
||||||
|
|
||||||
@@ -54,49 +54,87 @@ def profile_columns(data: dict) -> list[dict]:
|
|||||||
values: dict[str, list] = collections.defaultdict(list)
|
values: dict[str, list] = collections.defaultdict(list)
|
||||||
examples: dict[str, list[dict]] = collections.defaultdict(list)
|
examples: dict[str, list[dict]] = collections.defaultdict(list)
|
||||||
identities: dict[str, dict] = {}
|
identities: dict[str, dict] = {}
|
||||||
|
|
||||||
|
def record_name(record: dict) -> str:
|
||||||
|
if record.get("name"):
|
||||||
|
return record["name"]
|
||||||
|
return next(
|
||||||
|
(text for text in record.get("string_fields", {}).values() if text),
|
||||||
|
f"record {record['id']}",
|
||||||
|
)
|
||||||
|
|
||||||
|
def add(key: str, identity: dict, value, record: dict, **extra) -> None:
|
||||||
|
identities[key] = identity
|
||||||
|
values[key].append(value)
|
||||||
|
if len(examples[key]) < 5:
|
||||||
|
example = {
|
||||||
|
"id": record["id"],
|
||||||
|
"name": record_name(record),
|
||||||
|
"value": value,
|
||||||
|
**extra,
|
||||||
|
}
|
||||||
|
if message := record.get("message"):
|
||||||
|
example["message_description"] = message.get("description", "")
|
||||||
|
examples[key].append(example)
|
||||||
|
|
||||||
for record in records:
|
for record in records:
|
||||||
for address, value in record.get("fields", {}).items():
|
for address, value in record.get("fields", {}).items():
|
||||||
base = int(address, 16)
|
base = int(address, 16)
|
||||||
key = f"0x{base:x}"
|
key = f"0x{base:x}"
|
||||||
identities[key] = {
|
add(key, {
|
||||||
"key": key, "kind": "parallel-array", "base": key,
|
"key": key,
|
||||||
|
"kind": "scalar-field" if data.get("mode") == "mixed" else "parallel-array",
|
||||||
|
"base": key,
|
||||||
"stride": None, "column": None,
|
"stride": None, "column": None,
|
||||||
"semantic_name": field_semantics.get(key),
|
"semantic_name": field_semantics.get(key),
|
||||||
}
|
}, value, record)
|
||||||
values[key].append(value)
|
for address, value in record.get("string_fields", {}).items():
|
||||||
if len(examples[key]) < 5:
|
base = int(address, 16)
|
||||||
example = {
|
key = f"0x{base:x}"
|
||||||
"id": record["id"],
|
add(key, {
|
||||||
"name": record.get("name", ""),
|
"key": key, "kind": "string-field", "base": key,
|
||||||
"value": value,
|
"stride": None, "column": None,
|
||||||
}
|
"semantic_name": field_semantics.get(key),
|
||||||
if message := record.get("message"):
|
}, value, record)
|
||||||
example["message_description"] = message.get("description", "")
|
|
||||||
examples[key].append(example)
|
|
||||||
for key, value in record.get("record_fields", {}).items():
|
for key, value in record.get("record_fields", {}).items():
|
||||||
base_text, stride_text, column_text = key.split("/")
|
base_text, stride_text, column_text = key.split("/")
|
||||||
base = int(base_text, 16)
|
base = int(base_text, 16)
|
||||||
stride = int(stride_text)
|
stride = int(stride_text)
|
||||||
column = int(column_text)
|
column = int(column_text)
|
||||||
normalized_key = f"0x{base:x}/{stride}/{column}"
|
normalized_key = f"0x{base:x}/{stride}/{column}"
|
||||||
identities[normalized_key] = {
|
add(normalized_key, {
|
||||||
"key": normalized_key,
|
"key": normalized_key,
|
||||||
"kind": "record-column",
|
"kind": "record-column",
|
||||||
"base": f"0x{base:x}",
|
"base": f"0x{base:x}",
|
||||||
"stride": stride,
|
"stride": stride,
|
||||||
"column": column,
|
"column": column,
|
||||||
"semantic_name": field_semantics.get(normalized_key),
|
"semantic_name": field_semantics.get(normalized_key),
|
||||||
}
|
}, value, record)
|
||||||
values[normalized_key].append(value)
|
for kind, field_name in (
|
||||||
if len(examples[normalized_key]) < 5:
|
("array-cell", "array_fields"),
|
||||||
example = {
|
("footer-array", "footer_arrays"),
|
||||||
"id": record["id"],
|
):
|
||||||
"name": record.get("name", ""),
|
for key, raw_value in record.get(field_name, {}).items():
|
||||||
"value": value,
|
parts = key.split("/")
|
||||||
}
|
base = int(parts[0], 16)
|
||||||
if message := record.get("message"):
|
index = int(parts[1]) if len(parts) == 2 else None
|
||||||
example["message_description"] = message.get("description", "")
|
layout = data.get("array_layouts", {}).get(f"0x{base:x}", {})
|
||||||
examples[normalized_key].append(example)
|
stride = layout.get("stride")
|
||||||
|
column = index % stride if index is not None and stride else None
|
||||||
|
value = raw_value.get("values", []) if kind == "footer-array" else raw_value
|
||||||
|
extra = (
|
||||||
|
{"footer_off": raw_value.get("footer_off")}
|
||||||
|
if kind == "footer-array" else {}
|
||||||
|
)
|
||||||
|
add(key, {
|
||||||
|
"key": key,
|
||||||
|
"kind": kind,
|
||||||
|
"base": f"0x{base:x}",
|
||||||
|
"index": index,
|
||||||
|
"stride": stride,
|
||||||
|
"column": column,
|
||||||
|
"semantic_name": field_semantics.get(key),
|
||||||
|
}, value, record, **extra)
|
||||||
|
|
||||||
rows = []
|
rows = []
|
||||||
for key, vals in values.items():
|
for key, vals in values.items():
|
||||||
@@ -283,6 +321,10 @@ def main() -> int:
|
|||||||
"field_column_count": len(rows),
|
"field_column_count": len(rows),
|
||||||
"parallel_array_count": sum(row["kind"] == "parallel-array" for row in rows),
|
"parallel_array_count": sum(row["kind"] == "parallel-array" for row in rows),
|
||||||
"record_column_count": sum(row["kind"] == "record-column" for row in rows),
|
"record_column_count": sum(row["kind"] == "record-column" for row in rows),
|
||||||
|
"scalar_field_count": sum(row["kind"] == "scalar-field" for row in rows),
|
||||||
|
"string_field_count": sum(row["kind"] == "string-field" for row in rows),
|
||||||
|
"array_cell_count": sum(row["kind"] == "array-cell" for row in rows),
|
||||||
|
"footer_array_count": sum(row["kind"] == "footer-array" for row in rows),
|
||||||
"message_profile": messages,
|
"message_profile": messages,
|
||||||
"columns": sorted(rows, key=lambda row: (
|
"columns": sorted(rows, key=lambda row: (
|
||||||
int(row["base"], 16), row["stride"] or 0, row["column"] or 0
|
int(row["base"], 16), row["stride"] or 0, row["column"] or 0
|
||||||
|
|||||||
@@ -66,6 +66,35 @@ def test_static_negative_write() -> None:
|
|||||||
"INIT subtraction writes preserve negative values")
|
"INIT subtraction writes preserve negative values")
|
||||||
|
|
||||||
|
|
||||||
|
def test_real_mixed_table() -> None:
|
||||||
|
script = sys4load.load(extract_init.resolve("STINIT"))
|
||||||
|
check(extract_init.detect_mode(script) == "mixed",
|
||||||
|
"STINIT auto-detects as a mixed selector table")
|
||||||
|
records, meta = extract_init.extract_mixed(script)
|
||||||
|
check(len(records) == 74, "STINIT extracts all 74 sparse stage records")
|
||||||
|
check(records[0]["id"] == 1 and records[-1]["id"] == 170,
|
||||||
|
"STINIT preserves sparse runtime stage ids")
|
||||||
|
check(meta["selector_global"] == "0x4dfbc",
|
||||||
|
"STINIT records are keyed by scjump_progress_a")
|
||||||
|
check(meta["array_layouts"]["0xe74b5"] == {
|
||||||
|
"length": 350, "stride": 7, "rows": 50,
|
||||||
|
}, "STINIT preamble recovers a consumer-confirmed row buffer")
|
||||||
|
stage1 = records[0]
|
||||||
|
check(list(stage1["string_fields"].values()) == [
|
||||||
|
"オークの撃破", "", "自軍拠点の制圧", "50ターン経過",
|
||||||
|
], "STINIT stage 1 preserves all four condition strings")
|
||||||
|
check(stage1["fields"]["0xe7302"] == 12
|
||||||
|
and stage1["fields"]["0xe730c"] == 50
|
||||||
|
and stage1["fields"]["0xe730d"] == 0,
|
||||||
|
"STINIT stage 1 preserves BGM and turn-limit scalars")
|
||||||
|
check(stage1["array_fields"]["0xe7305/3"] == -2,
|
||||||
|
"STINIT fixed-buffer cells preserve negative values")
|
||||||
|
check(stage1["footer_arrays"]["0xe7889/3"]["values"] == [1, 1, 1],
|
||||||
|
"STINIT length-prefixed footer arrays retain their destination")
|
||||||
|
check(sum(len(record.get("footer_arrays", {})) for record in records) == 1396,
|
||||||
|
"STINIT accounts for every footer-array copy")
|
||||||
|
|
||||||
|
|
||||||
def test_real_message_tables() -> None:
|
def test_real_message_tables() -> None:
|
||||||
scripts = paths.scripts()
|
scripts = paths.scripts()
|
||||||
expected = {
|
expected = {
|
||||||
@@ -143,6 +172,7 @@ def test_field_semantics() -> None:
|
|||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
test_real_name_tables()
|
test_real_name_tables()
|
||||||
test_static_negative_write()
|
test_static_negative_write()
|
||||||
|
test_real_mixed_table()
|
||||||
test_real_message_tables()
|
test_real_message_tables()
|
||||||
test_message_join()
|
test_message_join()
|
||||||
test_field_semantics()
|
test_field_semantics()
|
||||||
|
|||||||
@@ -41,6 +41,35 @@ def main() -> int:
|
|||||||
assert rows["0x30/3/0"]["stride"] == 3
|
assert rows["0x30/3/0"]["stride"] == 3
|
||||||
assert rows["0x30/3/0"]["semantic_name"] == "test_record.zero"
|
assert rows["0x30/3/0"]["semantic_name"] == "test_record.zero"
|
||||||
assert rows["0x30/3/2"]["column"] == 2
|
assert rows["0x30/3/2"]["column"] == 2
|
||||||
|
|
||||||
|
mixed_fixture = {
|
||||||
|
"table": "MIXED",
|
||||||
|
"mode": "mixed",
|
||||||
|
"array_layouts": {"0x100": {"length": 9, "stride": 3, "rows": 3}},
|
||||||
|
"field_semantics": {"0x40": "condition", "0x50": "scalar"},
|
||||||
|
"records": [
|
||||||
|
{
|
||||||
|
"id": 11,
|
||||||
|
"string_fields": {"0x40": "Win"},
|
||||||
|
"fields": {"0x50": 20},
|
||||||
|
"array_fields": {"0x100/4": 7},
|
||||||
|
"footer_arrays": {
|
||||||
|
"0x100/6": {"footer_off": "0x200", "values": [1, 2, 3]},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
mixed = {row["key"]: row for row in profile.profile_columns(mixed_fixture)}
|
||||||
|
assert mixed["0x40"]["kind"] == "string-field"
|
||||||
|
assert mixed["0x40"]["semantic_name"] == "condition"
|
||||||
|
assert mixed["0x50"]["kind"] == "scalar-field"
|
||||||
|
assert mixed["0x100/4"]["kind"] == "array-cell"
|
||||||
|
assert mixed["0x100/4"]["stride"] == 3 and mixed["0x100/4"]["column"] == 1
|
||||||
|
assert mixed["0x100/6"]["kind"] == "footer-array"
|
||||||
|
assert mixed["0x100/6"]["examples"][0]["footer_off"] == "0x200"
|
||||||
|
assert mixed["0x100/6"]["common"][0]["value"] == "[1, 2, 3]"
|
||||||
|
assert mixed["0x40"]["examples"][0]["name"] == "Win"
|
||||||
|
|
||||||
messages = profile.profile_messages(fixture)
|
messages = profile.profile_messages(fixture)
|
||||||
assert messages["population"] == 1
|
assert messages["population"] == 1
|
||||||
assert messages["coverage"] == 1 / 3
|
assert messages["coverage"] == 1 / 3
|
||||||
|
|||||||
@@ -1601,3 +1601,73 @@ usage = "SCJUMP switch input (168 comparison reads) — progression counter/posi
|
|||||||
source = "inference"
|
source = "inference"
|
||||||
confidence = "med"
|
confidence = "med"
|
||||||
depends_on = []
|
depends_on = []
|
||||||
|
|
||||||
|
[[global]]
|
||||||
|
address = "0x27b9"
|
||||||
|
name = "stage_victory_condition_1"
|
||||||
|
category = "string-table"
|
||||||
|
type = "string"
|
||||||
|
value_domain = "first player-facing victory-condition line for the loaded stage"
|
||||||
|
usage = "STINIT writes one value for each of its 74 stage records. AIM renders this line first in the victory-condition section, and FIELD copies it into the current mission-condition display."
|
||||||
|
source = "investigation"
|
||||||
|
confidence = "high"
|
||||||
|
|
||||||
|
[[global]]
|
||||||
|
address = "0x27ba"
|
||||||
|
name = "stage_victory_condition_2"
|
||||||
|
category = "string-table"
|
||||||
|
type = "string"
|
||||||
|
value_domain = "optional second player-facing victory-condition line; empty when unused"
|
||||||
|
usage = "STINIT writes one value for each of its 74 stage records. AIM renders nonempty values after stage_victory_condition_1, and FIELD copies the slot into the current mission-condition display."
|
||||||
|
source = "investigation"
|
||||||
|
confidence = "high"
|
||||||
|
|
||||||
|
[[global]]
|
||||||
|
address = "0x27bb"
|
||||||
|
name = "stage_defeat_condition_1"
|
||||||
|
category = "string-table"
|
||||||
|
type = "string"
|
||||||
|
value_domain = "first player-facing defeat-condition line for the loaded stage"
|
||||||
|
usage = "STINIT writes one value for each of its 74 stage records. AIM renders this line first in the defeat-condition section, and FIELD copies it into the current mission-condition display."
|
||||||
|
source = "investigation"
|
||||||
|
confidence = "high"
|
||||||
|
|
||||||
|
[[global]]
|
||||||
|
address = "0x27bc"
|
||||||
|
name = "stage_defeat_condition_2"
|
||||||
|
category = "string-table"
|
||||||
|
type = "string"
|
||||||
|
value_domain = "optional second player-facing defeat-condition line; empty when unused"
|
||||||
|
usage = "STINIT writes one value for each of its 74 stage records. AIM renders nonempty values after stage_defeat_condition_1, and FIELD copies the slot into the current mission-condition display."
|
||||||
|
source = "investigation"
|
||||||
|
confidence = "high"
|
||||||
|
|
||||||
|
[[global]]
|
||||||
|
address = "0xe7302"
|
||||||
|
name = "stage_bgm_id"
|
||||||
|
category = "data-table"
|
||||||
|
type = "int"
|
||||||
|
value_domain = "{0,11..18}; 0 disables stage BGM"
|
||||||
|
usage = "STINIT's per-stage scalar loaded for all 74 records. FIELD passes the value directly to play-bgm when starting the stage."
|
||||||
|
source = "investigation"
|
||||||
|
confidence = "high"
|
||||||
|
|
||||||
|
[[global]]
|
||||||
|
address = "0xe730c"
|
||||||
|
name = "stage_turn_limit"
|
||||||
|
category = "data-table"
|
||||||
|
type = "int"
|
||||||
|
value_domain = "{0,10,15,30,50,60,70,75,80,100}; 0 means no limit"
|
||||||
|
usage = "STINIT's per-stage turn limit. DRAWCHP presents the value in the stage information, while FIELD compares the current turn against it when checking stage completion."
|
||||||
|
source = "investigation"
|
||||||
|
confidence = "high"
|
||||||
|
|
||||||
|
[[global]]
|
||||||
|
address = "0xe730d"
|
||||||
|
name = "stage_turn_limit_outcome"
|
||||||
|
category = "data-table"
|
||||||
|
type = "int"
|
||||||
|
value_domain = "{0,1}; 0 defeat on expiry, 1 forced-retreat clear on expiry"
|
||||||
|
usage = "STINIT mode paired with stage_turn_limit. Stage 1 stores 0 and describes 50-turn expiry as defeat; stage 2 stores 1 and explicitly describes 15-turn expiry as a forced-retreat clear."
|
||||||
|
source = "investigation"
|
||||||
|
confidence = "high"
|
||||||
|
|||||||
Reference in New Issue
Block a user