diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a1cb856..fed5a8a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -61,6 +61,14 @@ jobs: uv run graphfaker fraud --scale 0.001 --seed 42 --out ./bank --sink pyg --no-report --quiet test -f bank/graph.pt uv run python examples/pyg_baseline.py --scale 0.001 --epochs 5 + - name: The same, for the coordination pack + # Its truth uses different column names (is_coordinated, event_id), so + # the label rules being a convention rather than the fraud pack's + # schema is the thing this checks. + run: | + uv run graphfaker generate coordination --scale 0.0004 --seed 42 --out ./platform --sink pyg --quiet + test -f platform/graph.pt + uv run python examples/coordination_pyg_baseline.py --scale 0.0006 --tradecraft medium --epochs 5 audit: name: dependency audit diff --git a/HISTORY.rst b/HISTORY.rst index 3833129..cb5973f 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -2,6 +2,109 @@ History ======= +1.1.0 (unreleased) +------------------ + +The coordination domain pack: a social platform, and the coordinated behaviour +inside it. And the machinery both packs inject patterns with, lifted out of +them into the schema and the engine. + +* ``Pattern`` and the injection driver moved into the engine, and a domain's + pattern catalogue into the schema. ``graphfaker.schema.PatternCatalog`` and + ``PatternSpec`` declare what a pack injects: every shape with its share of + the budget, its natural span, and, for a decoy, the shape it imitates. + ``Camouflage`` is the set of dials every pack has (signature blend, timing + spread, overlap, decoy ratio, activity camouflage, size scale), which + ``HardnessProfile`` and ``TradecraftProfile`` now subclass and rename to + their own vocabulary. ``graphfaker.engine.injection`` holds the shared + bookkeeping: the pattern record, the recruitment ledger, the budget + allocator, the decoy orders and the driver loop that runs a catalogue. The + fraud and coordination packs keep their own drawing and lose about 200 lines + of duplicated machinery between them, including two byte-identical copies of + the budget allocator. Datasets are unchanged: the same seed gives the same + fingerprint in both packs at every hardness and tradecraft level. + +The social domain has been a ``GraphSchema`` preset since 0.5 and nothing more: +entities and a topology, one latent factor as its only ground truth, and no +tests. The fraud pack, by contrast, has a temporal process, a labelled pattern +catalogue, a measured hardness dial and a scoring harness. ``coordination`` +gives the social side the same treatment, as a domain of its own rather than by +changing ``social``, because the graph is a different one: accounts, topics and +devices rather than people, places and products. + +* ``graphfaker generate coordination`` produces accounts, topics and devices; + a follow graph with heavy-tailed in- and out-degree, ~50% reciprocity and + recoverable communities; and an organic event stream of posts, reshares and + replies with a diurnal rhythm and a lurker majority. +* Eight labelled inauthentic playbooks: ``copypasta``, + ``amplification_ring``, ``reply_brigade``, ``follow_farm``, + ``hashtag_flood``, ``sockpuppet_cluster``, ``astroturf_campaign``, + ``account_handover``. ``follow_farm`` is there because an event-only detector + cannot see it at all, and ``recall_by_playbook`` makes a one-signal detector + visible. +* **Organic decoys, which are the point.** Coordinated behaviour has a strong + legitimate twin: a fandom reacting to a release, a city reacting to an + earthquake and a paid amplification ring are structurally the same thing. A + dataset whose only labelled structures are inauthentic rewards any detector + that fires on synchrony, which is the detector that suspends fan clubs. + ``fandom_burst``, ``breaking_news`` and ``mutual_follow_community`` are + generated by the same machinery, labelled ``is_coordinated=False`` and + counted as false positives when flagged. Each is matched to its twin on the + properties that are not the point, chiefly posting volume. +* ``tradecraft`` (``low`` / ``medium`` / ``high``) is an input to a + measurement, not an assertion. Mean per-playbook best-single-feature AUC + falls 0.984 / 0.852 / 0.789, and the evidence that still works broadens from + identity and timing to structure and content. +* ``hardness_report`` scores sixteen features in five families and reports + which family found each playbook. ``decoy_separability`` scores each organic + structure against the playbook it imitates, and names the feature + responsible, because a decoy separable by one feature is not a decoy. +* ``evaluate`` scores a detector at account, event and campaign granularity, + with ``organic_false_positive_rate`` reported separately: on a real platform + the cost of suspending a fan club is not symmetric with the cost of missing + one ring. +* ``examples/coordination_detectors.py`` scores five rules a platform would try + first, at all three tradecraft levels. The conclusion matches the fraud + pack's Cypher cookbook: single-signal detectors do badly. The best F1 + anywhere is 0.499, at the easiest setting, and the co-occurrence rule goes + from 0.843 precision with no organic false positives at ``low`` to 0.118 + precision and 15% of organic accounts flagged at ``medium``. +* No node attribute names the answer, and a test asserts that no single feature + separates any playbook perfectly — a perfectly separating feature means a + label has leaked into the graph. It caught two such leaks during development: + account handover left ``reshare_count`` at exactly zero, and every sockpuppet + cluster was separable at AUC 1.000 by device sharing until organic device + sharing was given a tail of its own. +* 43 tests, including the pack's own thesis as an assertion: a co-occurrence + detector must flag organic bursts, or the decoys are not doing their job. +* A PyTorch Geometric export for the pack, via the existing ``pyg`` sink: + ``--sink pyg``, ``to_hetero_data`` and ``from_directory`` all work on a + coordination dataset. Account carries ``y`` (coordinated), ``decoy`` + (organic) and stratified splits; ``POSTED``, ``RESHARED`` and ``REPLIED`` + carry ``y`` and ``edge_time``; ``community`` is a latent tensor rather than a + feature. +* The sink's label rules are now a convention rather than the fraud pack's + column names: a truth frame keyed by ``_id`` with a single boolean + column labels those entities. ``accounts.is_fraud`` and + ``accounts.is_coordinated`` both work, and a third domain following the same + shape gets labels without touching the module. Identifiers and foreign keys + are kept out of edge attributes — the coordination pack's interaction edges + carry the topic they are about, which one-hot encoded to 48 columns and would + reach thousands at scale. Matching a label frame on values alone picked + ``campaigns.topic``, which holds real Topic ids next to a boolean, and + labelled every topic; the id column now has to end in ``_id`` as well. +* ``examples/coordination_pyg_baseline.py``: a heterogeneous GraphSAGE against + a logistic regression on account features alone, at all three tradecraft + levels. The graph helps everywhere — average precision more than doubles at + ``medium``, 0.101 to 0.243 — and it **flags three times as many organic + accounts**, 4.0% against 12.0%, because it buys its power by learning + "tightly connected group acting together" and a fan club is exactly that. + Measuring AUC alone says "graphs win", which is true and incomplete; that gap + is what the organic decoys exist to make visible. The baseline reports the + organic share at a fixed operating point, so two models with different score + distributions compare fairly. +* ``docs/domains/coordination.md``; ``docs/pyg.md`` covers both packs. + 1.0.1 (unreleased) ------------------ diff --git a/README.md b/README.md index aaf0a0c..5feac53 100644 --- a/README.md +++ b/README.md @@ -26,7 +26,7 @@ pip install "graphfaker[osm]" # the OpenStreetMap fetcher (osmnx and its pip install "graphfaker[examples]" # adds matplotlib, ladybug and jupyter for the notebooks ``` -The fraud pack, the social domain, schemas and every sink are in the base install. Database clients and the PyG export are extras: `[neo4j]`, `[ladybug]`, `[duckdb]`, `[pyg]`. `graphfaker info` shows which are installed. +The fraud pack, the coordination pack, the social domain, schemas and every sink are in the base install. Database clients and the PyG export are extras: `[neo4j]`, `[ladybug]`, `[duckdb]`, `[pyg]`. `graphfaker info` shows which are installed. Or without a Python environment, as a container ([docs](https://graphfaker.readthedocs.io/en/latest/get-started/docker.html)): @@ -98,6 +98,7 @@ run = fraud.generate(scale=0.01, hardness="medium", seed=42) |---|---|---| | `social` | people, places, organizations, events and products; a few hubs and many quiet nodes, friends who know each other's friends, communities whose members are alike, organizations as big as their headcount | `GraphFaker.generate_graph(source="faker")` or `graphfaker generate social` | | `fraud` | a bank: customers, accounts, merchants, devices, counterparties; a realistic transaction process; eleven labelled laundering typologies with decoys and a measured hardness | `graphfaker fraud` or `graphfaker generate fraud` | +| `coordination` | a social platform: accounts, topics, devices; follows, posts, reshares and replies over time; eight labelled coordination playbooks, **organic bursts that look exactly like them**, and a measured tradecraft level | `graphfaker generate coordination` | | your own | a `GraphSchema`, or a process with injected patterns | [docs/adding-a-domain.md](docs/adding-a-domain.md) | Real-world sources, loaded rather than generated: diff --git a/docs/_readme_pages.py b/docs/_readme_pages.py index e497e52..507690a 100644 --- a/docs/_readme_pages.py +++ b/docs/_readme_pages.py @@ -32,6 +32,7 @@ social fraud +coordination real-world ``` """, diff --git a/docs/adding-a-domain.md b/docs/adding-a-domain.md index 967dcf8..a5c190c 100644 --- a/docs/adding-a-domain.md +++ b/docs/adding-a-domain.md @@ -83,6 +83,46 @@ The contract for `generate.py`: If your domain injects patterns, also provide a way to measure them. The fraud pack's `hardness_report` (single-feature AUCs against the truth) and `evaluate` (precision and recall at entity, event and pattern level) are written for that domain, but the approach transfers: list the naive rules someone would try first, and report how well each one does. +### Patterns you declare, drawing you write + +Two packs inject patterns, and the parts they share are in the engine rather than copied between them. + +Declare the catalogue with [`PatternCatalog`](reference/api.rst): every shape with its share of the budget and its natural span in days, then the decoys, each naming the shape it imitates. + +```python +from graphfaker.schema import PatternCatalog, PatternSpec + +CATALOG = PatternCatalog( + base=1_000, # patterns at scale 1.0 + floor=2, # at least this many of every shape, so small datasets cover the catalogue + patterns=[ + PatternSpec(name="fan_in", share=0.14, span_days=2.0), + PatternSpec(name="cycle", share=0.10, span_days=2.0), + PatternSpec(name="decoy_fan_in", imitates="fan_in", span_days=2.0), + ], +) +``` + +`CATALOG.total(scale)` is how many patterns a dataset gets, `CATALOG.counts(total)` splits them across the shapes, and `CATALOG.decoys`, `CATALOG.twins` and `CATALOG.span_days(name)` are what the injector and the hardness report read. + +Declare the dials with `Camouflage`, or a subclass that adds your own and renames these to whatever your readers call them. The fraud pack calls `signature_blend` `amount_blend`, because in a bank the signature is the amount; the coordination pack calls it `text_blend`. + +Then write a context and the drawing. `InjectionContext` holds the dials, the period, who is already in a pattern and whether overlap is allowed; subclass it and add the methods that make your domain what it is. + +```python +from graphfaker.engine.injection import InjectionContext, Pattern, round_robin_decoys, run_catalog + +class MyContext(InjectionContext): + def pick(self, k): ... # recruitment, your rules + def emit(self, pattern, ...): ... # a row, then pattern.touch(timestamp) + +patterns = run_catalog(ctx, counts, FUNCTIONS, make, round_robin_decoys(CATALOG, n_decoys)) +``` + +`run_catalog` runs the shapes in catalogue order, numbers them, runs the decoys last with overlap off, and returns the records. What it deliberately does not do is draw anything: how many sources a fan-in has and what a copypasta posts is the part that does not generalise, and the part worth writing yourself. + +The order that loop walks in is part of what a seed reproduces, so changing it changes every dataset the domain has produced. Both decoy orders are available (`round_robin_decoys`, `grouped_decoys`) because the two packs chose differently before the code was shared, and changing either would have rewritten published measurements for no gain. + ## Registering Built-in domains are registered in `graphfaker/domains/__init__.py`: diff --git a/docs/domains/coordination.md b/docs/domains/coordination.md new file mode 100644 index 0000000..c33fac1 --- /dev/null +++ b/docs/domains/coordination.md @@ -0,0 +1,199 @@ +# Coordination + +A social platform, and the coordinated behaviour inside it. Accounts, the topics they post about and the devices they post from; follows, posts, reshares and replies over time; and **labelled campaigns** — some of them inauthentic, some of them not. + +```bash +graphfaker generate coordination --scale 0.001 --tradecraft high --seed 42 --out ./platform +``` + +```python +from graphfaker.domains.coordination import generate, hardness_report, evaluate + +run = generate(scale=0.001, tradecraft="high", seed=42) +print(hardness_report(run).summary()) +``` + +## Why the organic decoys are the point + +Laundering has a weak legitimate twin. Coordinated behaviour has a *strong* one. A fandom reacting to an album drop, a city reacting to an earthquake, and a paid amplification ring are structurally the same thing: a burst of near-simultaneous, near-identical activity from a tightly connected group. + +So a coordination dataset whose only labelled structures are inauthentic is worse than useless. Any detector that fires on synchrony scores perfectly on it — and that is precisely the detector that ruins real platforms by suspending fan clubs and protest movements. + +Here the organic bursts are generated by the same machinery as the campaigns, labelled `is_coordinated=False`, and counted as **false positives** when flagged. `decoy_ratio` reaches 1.0 at `tradecraft="high"`: as many organic structures as inauthentic ones. + +Each decoy is paired with the playbook it imitates, matched on the properties that are *not* the point — chiefly posting volume — so the question a detector faces is about text, timing and structure rather than about how much an account posts. + +| organic decoy | imitates | the real difference | +|---|---|---| +| `fandom_burst` | `hashtag_flood` | text varies as organic text does; accounts aged normally; ordinary activity kept | +| `breaking_news` | `copypasta` | drawn across communities rather than within one | +| `mutual_follow_community` | `follow_farm` | the members talk to each other, they do not only follow | + +## What it contains + +| node type | attributes | +|---|---| +| Account | handle, display name, activity, reshare rate, verified, profile complete, language, created at, follower count, community | +| Topic | name, category, prominence, community | +| Device | fingerprint, device type, os | + +| relationship | meaning | +|---|---| +| `FOLLOWS` | the social graph: heavy-tailed in- and out-degree, ~50% reciprocal, recoverable communities | +| `USES` | account to device; a few devices legitimately serve a crowd | +| `POSTED` | account to topic, with a timestamp and a `template_id` | +| `RESHARED`, `REPLIED` | account to account, with a timestamp and a topic | + +**No node attribute names the answer.** There is no `is_bot` column. Everything a detector could legitimately use is in the tables; everything else is in `run.truth`, which `--blind` leaves out. A test asserts that no single feature separates any playbook perfectly, because a perfectly separating feature means a label has leaked. + +## The playbooks + +Eight inauthentic shapes, each with a signature a detector is written to catch: + +| playbook | signature | +|---|---| +| `copypasta` | many accounts, one text, one moment | +| `amplification_ring` | a cluster repeatedly boosting one account | +| `reply_brigade` | coordinated replies flooding one target | +| `follow_farm` | a dense mutual-follow cluster that barely posts | +| `hashtag_flood` | volume on a single topic in a narrow window | +| `sockpuppet_cluster` | several accounts, one device | +| `astroturf_campaign` | a long, low, sustained push across communities | +| `account_handover` | aged dormant accounts activated together | + +`follow_farm` is included specifically because an event-only detector cannot see it at all: its members hardly post. Any detector scoring well on the pack overall and zero on `follow_farm` is a one-signal detector, and `recall_by_playbook` makes that visible. + +## Options + +| option | default | meaning | +|---|---|---| +| `scale` | 0.001 | 1.0 is ~5M accounts / ~100M events | +| `tradecraft` | `medium` | `low` / `medium` / `high`; how well the campaigns hide | +| `communities` | 12 | latent interest groups driving topics, attributes and follows | +| `topics` | ~1% of accounts | distinct topics | +| `follows_per_account` | 15 | mean out-degree | +| `period_start`, `period_days` | 2026-01-01, 90 | the activity window | +| `campaigns` | derived from scale | override campaigns per playbook | + +## What tradecraft does + +`tradecraft` is an input to a measurement, not an assertion. `hardness_report` scores every single feature a naive detector might threshold on by the AUC it achieves separating a playbook's accounts from uninvolved ones. Measured at `scale=0.0006`, seed 7, as the mean of each playbook's best single feature: + +| tradecraft | mean max AUC | evidence that still works | +|---|---|---| +| `low` | 0.984 | content, identity, timing | +| `medium` | 0.852 | content, identity, structure, timing | +| `high` | 0.789 | content, identity, structure, timing | + +At `low` the giveaway is usually `account_age_days`: every campaign account was created days before the campaign, which is the loudest single signal in real platform data and is left in deliberately. At `high` the best feature is more often structural or topical, which is another way of saying the campaign is only findable by combining evidence. + +## Detectors, scored + +Every rule below is one a platform would try first. `examples/coordination_detectors.py` regenerates the table. + +| detector | tradecraft | precision | recall | F1 | organic flagged | +|---|---|---|---|---|---| +| co-occurrence (1h) | low | 0.843 | 0.354 | 0.499 | 0.0% | +| co-occurrence (1h) | medium | 0.118 | 0.046 | 0.066 | 15.2% | +| co-occurrence (1h) | high | 0.022 | 0.012 | 0.015 | 3.2% | +| duplicate text | low | 0.452 | 0.527 | 0.486 | 0.0% | +| duplicate text | medium | 0.186 | 0.264 | 0.218 | 14.3% | +| duplicate text | high | 0.086 | 0.160 | 0.112 | 20.7% | +| fresh + active | low | 0.000 | 0.000 | 0.000 | 0.0% | +| fresh + active | medium | 0.357 | 0.042 | 0.075 | 0.0% | +| fresh + active | high | 0.100 | 0.012 | 0.021 | 0.0% | +| dense reciprocity | low | 0.234 | 0.034 | 0.060 | 0.0% | +| dense reciprocity | medium | 0.234 | 0.046 | 0.077 | 3.8% | +| shared device | medium | 0.102 | 0.192 | 0.134 | 16.2% | + +Three things worth reading off it. + +**Single-signal rules do badly**, the same conclusion the fraud pack reached with Cypher rules. The best F1 anywhere in the table is 0.499, at the easiest setting. + +**They get worse as tradecraft rises, and they start flagging organic structures as they do.** The co-occurrence rule goes from 0.843 precision and no organic false positives at `low` to 0.118 precision and 15% of organic accounts at `medium`. That column is the deployment cost, and it is why it is reported separately: the cost of suspending a fan club is not symmetric with the cost of missing one amplification ring. + +**`fresh + active` scores zero at `low`** — the setting where every campaign account *is* fresh. At `low`, `activity_camouflage` is 0, so campaign accounts carry no ordinary activity and are therefore not "active". A conjunction of two signals can fail precisely where each one alone is strongest, which is the kind of thing a scored harness shows and intuition does not. + +## Training a GNN on it + +```bash +graphfaker generate coordination --scale 0.002 --tradecraft medium --seed 42 --sink pyg --out ./platform +python examples/coordination_pyg_baseline.py --scale 0.002 +``` + +`--sink pyg` writes `graph.pt`: a `HeteroData` with features on every node +type, `community` as a latent tensor, `y` and `decoy` plus stratified splits on +Account, and `y` and `edge_time` on the three event channels. `FOLLOWS` and +`USES` carry no labels, because the truth says nothing about them. + +Measured at `scale=0.002`, seed 42, two-layer heterogeneous GraphSAGE against a +logistic regression on the account features alone: + +| tradecraft | model | AUC | AP | organic accounts flagged | +|---|---|---|---|---| +| low | features only | 0.932 | 0.536 | — (no decoys at `low`) | +| low | features + graph | **0.964** | **0.785** | — | +| medium | features only | 0.680 | 0.101 | 4.0% | +| medium | features + graph | **0.791** | **0.243** | **12.0%** | +| high | features only | 0.590 | 0.050 | 2.6% | +| high | features + graph | **0.696** | **0.082** | **7.7%** | + +Three readings. + +**The graph helps at every level**, and helps most where it matters: average +precision more than doubles at `medium`. A campaign is defined by who acts with +whom, so an account's own attributes say very little — which is why the +features-only row is the control worth having. + +**Tradecraft degrades it as designed.** The graph model falls from 0.964 to +0.696 AUC, and its average precision from 0.785 to 0.082. + +**The graph model flags three times as many organic accounts.** At `medium` it +goes from 4.0% to 12.0%. It buys its detection power by learning "tightly +connected group acting together", and that is exactly what a fan club is. This +is the finding the decoys exist to produce: measure AUC alone and the +conclusion is "graphs win", which is true and incomplete. + +The organic column is computed at a fixed operating point — each model is asked +for as many accounts as there are real campaign members — so two models with +different score distributions are compared fairly. + +## Ground truth + +| table | contents | +|---|---| +| `campaigns` | campaign id, playbook, `is_coordinated`, account list, roles, topic, window, event count | +| `accounts` | account to campaign with its role and `is_coordinated` | +| `events` | event id to campaign, playbook and `is_coordinated` | +| `community` | each latent interest community's parameters | + +`evaluate` scores a detector at three granularities, mirroring the fraud pack's evaluator: + +```python +from graphfaker.domains.coordination import evaluate + +scores = evaluate(run, flagged_accounts=my_accounts, flagged_events=my_events) +print(scores.summary()) +``` + +A campaign counts as found when `cluster_threshold` (default 0.5) of its accounts are flagged. `organic_false_positive_rate` and `recall_by_playbook` are reported alongside precision and recall. + +## Realism + +The organic platform, measured at `scale=0.0006`, seed 7: + +| property | value | why it matters | +|---|---|---| +| silent account share | 27% | real platforms are mostly lurkers; without them "low activity" is uninformative | +| event Gini | 0.65 | a few accounts never stop posting | +| follower Gini | 0.46 | audiences differ by orders of magnitude, so an amplification target is a meaningful thing | +| reciprocal follow share | 50% | a farm pushes this to ~1.0, so the organic rate has to be substantial | +| mean clustering | 0.42 | people who share a follow tend to connect | +| diurnal ratio | 18.7 | sub-minute synchrony is detectable partly because nothing organic is that synchronous | + +## Limitations + +- **Follower Gini is 0.46, and a real platform is higher.** The tail here is bounded by `follows_per_account`; real inequality is driven by accounts with millions of followers, which needs a far larger follow budget than a test-scale dataset carries. +- **No content.** A post carries a `template_id` and nothing else. The pack measures detectors, and a detector does not need the text; generating personas or messaging is out of scope. +- **The playbook catalogue is structural.** These are coordination shapes documented in public platform-manipulation research — who acts with whom, when, and how densely. They are not operational recipes. +- **Validation is weaker than the fraud pack's.** AMLworld provided a typology catalogue to borrow; there is no equivalent public labelled coordination dataset at scale, and platform APIs are closed. The shapes come from published research, and the realism table above is what can be checked directly. diff --git a/docs/index.rst b/docs/index.rst index 67f1fd5..b12bf45 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -104,12 +104,13 @@ GraphFaker How the fraud graph is generated Ways to generate synthetic graphs Adding a domain - Training a GNN on the bank + Training a GNN
Domains Social Fraud and AML + Coordinated behaviour Real-world networks
diff --git a/docs/pyg.md b/docs/pyg.md index dfb36e7..016fea7 100644 --- a/docs/pyg.md +++ b/docs/pyg.md @@ -1,4 +1,4 @@ -# Training a GNN on the bank +# Training a GNN A generated graph exports to a PyTorch Geometric `HeteroData` in one call, with features encoded, the ground truth as labels, and train, validation and test masks already drawn. This page shows the export, then a baseline that answers the question the dataset exists to ask: how much does the graph help a detector, and how fast does that help fade as the fraud gets harder? @@ -38,10 +38,10 @@ Features (`x`) are built per table from the columns a model can use: numeric col Latent factors are treated differently. `region` (and `community` in the social graph) is the hidden variable the generator drew attributes and edges from; putting it in `x` would hand a model the answer to the very correlations it is supposed to learn. It is attached as its own tensor (`data["Account"].region`) so it can serve as a label for community recovery and never as a feature. -Labels come from the truth: +Labels come from the truth, and the rule is a convention rather than one pack's column names: **a truth frame keyed by `_id` with a single boolean column labels those entities.** The fraud pack writes `accounts.is_fraud` and `transactions.is_fraud`; the coordination pack writes `accounts.is_coordinated` and `events.is_coordinated`; a third pack that follows the same shape gets labels without touching this module. - `data["Account"].y` is 1 for an account that takes part in a fraud pattern, 0 otherwise. `decoy` is 1 for an account whose only patterns are legitimate look-alikes (a business paying salaries has the shape of a fan-out); flagging it is a false positive, and the decoys are there so that a model is measured on that. -- Every relationship with a `tx_id` carries `y` (1 for an injected fraud transaction) and `edge_time` in seconds, so edge-level and temporal experiments have what they need. +- A relationship carrying an `*_id` column that a truth frame also carries gets `y` from that frame (1 for an injected fraud transaction, or a coordinated post, reshare or reply), and `edge_time` in seconds wherever there is a timestamp, so edge-level and temporal experiments have what they need. Structural relationships the truth says nothing about — `OWNS`, `USES`, `FOLLOWS` — carry no labels. - `train_mask`, `val_mask` and `test_mask` split the labelled node type 60/20/20, stratified so the rare positive class is present in every part, from the `seed` you pass. The same seed gives the same split on any machine. `--blind` (or `truth=None`) produces the same tensors without `y`, `decoy` or the masks, for a dataset that is handed to someone else to score. @@ -68,6 +68,30 @@ Three things to read off this table. Account attributes on their own say nothing The `high` row is also a reminder to read small numbers carefully: 71 positive accounts leaves 14 in the test split, and the validation AUC during training sat near 0.77, so a single run at that size says "hard", not "0.56". For a paper, repeat over seeds and report the spread; the generator makes that cheap. +## The coordination pack + +The same export, the same call, a different graph: + +```bash +graphfaker generate coordination --scale 0.002 --tradecraft medium --seed 42 --out ./platform --sink pyg +python examples/coordination_pyg_baseline.py --scale 0.002 +``` + +Coordination is a better fit for a GNN than fraud is, because a campaign is *defined* by who acts with whom: an account's own attributes barely say anything, so the features-only control is genuinely weak and the graph has more to add. + +Measured at `scale=0.002`, seed 42: + +| tradecraft | model | AUC | AP | organic accounts flagged | +|---|---|---|---|---| +| low | features only | 0.932 | 0.536 | — | +| low | features + graph | 0.964 | 0.785 | — | +| medium | features only | 0.680 | 0.101 | 4.0% | +| medium | features + graph | 0.791 | 0.243 | 12.0% | +| high | features only | 0.590 | 0.050 | 2.6% | +| high | features + graph | 0.696 | 0.082 | 7.7% | + +The graph helps at every level and average precision more than doubles at `medium`. It also **flags three times as many organic accounts**: the model buys its power by learning "tightly connected group acting together", and a fan club is exactly that. Measuring AUC alone gives "graphs win", which is true and incomplete — and that gap is what the coordination pack's organic decoys exist to make visible. [The coordination domain](domains/coordination.md) has the detail. + ## What to try next - Edge features. `SAGEConv` ignores `edge_attr`; a model that reads amounts, memos and `edge_time` (`GATv2Conv` with `edge_dim`, or a temporal model) has more to work with, and the transaction-level `y` lets it be trained on transactions rather than accounts. diff --git a/examples/coordination_detectors.py b/examples/coordination_detectors.py new file mode 100644 index 0000000..ca51a33 --- /dev/null +++ b/examples/coordination_detectors.py @@ -0,0 +1,115 @@ +"""Score simple coordination detectors against the ground truth. + +Regenerates the table in ``docs/domains/coordination.md``. Every rule here is +one a platform would actually try first, and each is scored at three tradecraft +levels so the reader can see where it stops working. + +The conclusion is the same one the fraud pack reached with Cypher rules: +single-signal detectors do badly, and the signal they do have collapses as the +operation gets more careful. The difference here is the organic decoys, which +put a number on what the detector costs when it is wrong. + + python examples/coordination_detectors.py +""" + +from __future__ import annotations + +import polars as pl + +from graphfaker.domains.coordination import ( + account_features, + evaluate, + generate, + hardness_report, +) +from graphfaker.domains.coordination.process import POSTED + +SCALE = 0.0006 +SEED = 7 + + +def co_occurrence(run, features, *, window: str = "1h", min_accounts: int = 6) -> list[str]: + """Flag accounts that post about one topic in the same hour as many others. + + The standard first pass. It is looking for synchrony, which is exactly what + a fan club and a news event also produce. + """ + posted = run.tables.edges[POSTED] + if posted.height == 0: + return [] + buckets = ( + posted.with_columns(pl.col("timestamp").dt.truncate(window).alias("bucket")) + .group_by(["target", "bucket"]) + .agg(pl.col("source").unique().alias("accounts")) + .filter(pl.col("accounts").list.len() >= min_accounts) + ) + return sorted({str(a) for row in buckets["accounts"].to_list() for a in row}) + + +def duplicate_text(run, features, *, threshold: float = 0.9) -> list[str]: + """Flag accounts whose posts mostly reuse a template another account used.""" + return features.filter(pl.col("template_reuse") >= threshold)["account_id"].to_list() + + +def fresh_and_active(run, features, *, max_age: int = 30) -> list[str]: + """Flag young accounts that post a lot. The oldest heuristic there is.""" + busy = features["event_count"].quantile(0.8) or 0 + return features.filter( + (pl.col("account_age_days") <= max_age) & (pl.col("event_count") > busy) + )["account_id"].to_list() + + +def dense_reciprocal(run, features, *, threshold: float = 0.9) -> list[str]: + """Flag accounts whose follows are almost all mutual: the follow-farm rule.""" + return features.filter( + (pl.col("reciprocity") >= threshold) & (pl.col("following") >= 5) + )["account_id"].to_list() + + +def shared_device(run, features, *, min_sharers: int = 4) -> list[str]: + """Flag accounts sharing a device with several others.""" + return features.filter(pl.col("device_shared_with") >= min_sharers)[ + "account_id" + ].to_list() + + +DETECTORS = { + "co-occurrence (1h)": co_occurrence, + "duplicate text": duplicate_text, + "fresh + active": fresh_and_active, + "dense reciprocity": dense_reciprocal, + "shared device": shared_device, +} + + +def main() -> None: + header = ( + f"{'detector':<22}{'tradecraft':<12}{'precision':>10}{'recall':>8}" + f"{'F1':>7}{'organic FP':>12}" + ) + print(header) + print("-" * len(header)) + for name, detector in DETECTORS.items(): + for level in ("low", "medium", "high"): + run = generate(scale=SCALE, tradecraft=level, seed=SEED) + features = account_features(run) + flagged = detector(run, features) + scores = evaluate(run, flagged_accounts=flagged) + print( + f"{name:<22}{level:<12}{scores.account.precision:>10.3f}" + f"{scores.account.recall:>8.3f}{scores.account.f1:>7.3f}" + f"{scores.organic_false_positive_rate:>11.1%}" + ) + print() + + print("Hardness at each level (mean of per-playbook max single-feature AUC):") + for level in ("low", "medium", "high"): + report = hardness_report(generate(scale=SCALE, tradecraft=level, seed=SEED)) + coordinated = [p for p in report.playbooks if p.is_coordinated] + mean_auc = sum(p.max_auc for p in coordinated) / len(coordinated) + families = sorted({p.best_family for p in coordinated if p.best_family}) + print(f" {level:<8}{mean_auc:.3f} evidence families: {', '.join(families)}") + + +if __name__ == "__main__": + main() diff --git a/examples/coordination_pyg_baseline.py b/examples/coordination_pyg_baseline.py new file mode 100644 index 0000000..6a69241 --- /dev/null +++ b/examples/coordination_pyg_baseline.py @@ -0,0 +1,176 @@ +"""A GNN baseline on the synthetic platform: does structure find the campaigns? + +Generates a platform at a chosen tradecraft level, exports it as a PyG +``HeteroData``, then trains two models on the Account labels and scores them on +held-out accounts: + +* a logistic regression on the account features alone (no graph), and +* a two-layer heterogeneous GraphSAGE that also sees the neighbourhood. + +The gap between them is what the graph is worth. Coordination should show a +larger gap than fraud does, because a campaign is *defined* by who acts with +whom: a lone account's own attributes barely say anything, which is why the +features-only baseline is the interesting control here. + +A third number this reports that the fraud baseline does not: the share of +**organic** accounts flagged at the chosen operating point. Fan clubs and news +reactions are in the graph, labelled legitimate, and a model that finds +campaigns by firing on synchrony will take them with it. That column is the +deployment cost. + + python examples/coordination_pyg_baseline.py --scale 0.002 --tradecraft medium + +Needs ``pip install "graphfaker[pyg]"``. +""" + +from __future__ import annotations + +import argparse +import time +import warnings + +import numpy as np +import polars as pl +import torch +import torch.nn.functional as F +import torch_geometric.transforms as T +from sklearn.linear_model import LogisticRegression +from sklearn.metrics import average_precision_score, roc_auc_score +from torch_geometric.nn import SAGEConv, to_hetero + +from graphfaker.backends.tables import ID +from graphfaker.domains.coordination import generate +from graphfaker.sinks.pyg import to_hetero_data + +warnings.filterwarnings("ignore") + + +class SAGE(torch.nn.Module): + def __init__(self, hidden: int, out: int): + super().__init__() + self.conv1 = SAGEConv((-1, -1), hidden) + self.conv2 = SAGEConv((-1, -1), out) + + def forward(self, x, edge_index): + x = F.relu(self.conv1(x, edge_index)) + return self.conv2(x, edge_index) + + +def organic_flag_rate( + probability: np.ndarray, organic: np.ndarray, positives: int +) -> float: + """Share of organic accounts inside the model's top-``positives`` scores. + + A fixed operating point rather than a threshold, so the comparison between + two models with different score distributions is fair: both are asked for + as many accounts as there are real campaign members. + """ + if not organic.any() or positives <= 0: + return float("nan") + flagged = np.zeros(len(probability), dtype=bool) + flagged[np.argsort(-probability)[:positives]] = True + return float(flagged[organic].mean()) + + +def report( + label: str, + y: np.ndarray, + probability: np.ndarray, + test: np.ndarray, + organic: np.ndarray, +) -> None: + auc = roc_auc_score(y[test], probability[test]) + ap = average_precision_score(y[test], probability[test]) + organic_rate = organic_flag_rate( + probability[test], organic[test], int(y[test].sum()) + ) + organic_text = "n/a" if np.isnan(organic_rate) else f"{organic_rate:6.1%}" + print(f" {label:<42}AUC {auc:.3f} AP {ap:.3f} organic flagged {organic_text}") + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + parser.add_argument("--scale", type=float, default=0.002) + parser.add_argument( + "--tradecraft", default="all", choices=["low", "medium", "high", "all"] + ) + parser.add_argument("--seed", type=int, default=42) + parser.add_argument("--epochs", type=int, default=60) + parser.add_argument("--hidden", type=int, default=64) + args = parser.parse_args() + + levels = ( + ["low", "medium", "high"] if args.tradecraft == "all" else [args.tradecraft] + ) + for level in levels: + torch.manual_seed(args.seed) + run = generate(scale=args.scale, seed=args.seed, tradecraft=level) + data = to_hetero_data(run.tables, run.truth, seed=args.seed) + # Message passing needs both directions; the export keeps the dataset's. + data = T.ToUndirected(merge=False)(data) + account = data["Account"] + y = account.y.numpy() + train = account.train_mask.numpy() + val = account.val_mask.numpy() + test = account.test_mask.numpy() + + # Organic accounts: in a decoy structure and in no campaign. The export + # already separates them, which is what `decoy` is for. + organic = account.decoy.numpy().astype(bool) + + print( + f"\ntradecraft={level}: {y.size:,} accounts, {int(y.sum())} in campaigns " + f"({100 * y.mean():.2f}%), {int(organic.sum())} organic" + ) + + x = account.x.numpy() + classifier = LogisticRegression(max_iter=2000, class_weight="balanced").fit( + x[train], y[train] + ) + report( + "features only (logistic regression)", + y, + classifier.predict_proba(x)[:, 1], + test, + organic, + ) + + model = to_hetero(SAGE(args.hidden, 2), data.metadata(), aggr="sum") + optimizer = torch.optim.Adam(model.parameters(), lr=0.005, weight_decay=5e-4) + weight = torch.tensor( + [1.0, float((y[train] == 0).sum() / max(1, (y[train] == 1).sum()))] + ) + started = time.perf_counter() + best_val, best_state = -1.0, None + for _ in range(args.epochs): + model.train() + optimizer.zero_grad() + out = model(data.x_dict, data.edge_index_dict)["Account"] + loss = F.cross_entropy( + out[account.train_mask], account.y[account.train_mask], weight=weight + ) + loss.backward() + optimizer.step() + model.eval() + with torch.no_grad(): + probability = ( + F.softmax(model(data.x_dict, data.edge_index_dict)["Account"], dim=1)[:, 1] + .numpy() + ) + val_auc = roc_auc_score(y[val], probability[val]) + if val_auc > best_val: + best_val = val_auc + best_state = {k: v.clone() for k, v in model.state_dict().items()} + model.load_state_dict(best_state) + model.eval() + with torch.no_grad(): + probability = ( + F.softmax(model(data.x_dict, data.edge_index_dict)["Account"], dim=1)[:, 1] + .numpy() + ) + report("features plus graph (GraphSAGE)", y, probability, test, organic) + print(f" ({time.perf_counter() - started:.0f}s training)") + + +if __name__ == "__main__": + main() diff --git a/examples/duplication_experiment.py b/examples/duplication_experiment.py index f251267..f02d112 100644 --- a/examples/duplication_experiment.py +++ b/examples/duplication_experiment.py @@ -341,7 +341,7 @@ def main() -> None: for name, adapter in ADAPTERS.items(): try: graph = adapter(corpus) - except Exception as error: # noqa: BLE001 - one bad adapter must not end the run + except Exception as error: print(f" {name}: FAILED ({type(error).__name__}: {error})") skipped.append(f"{name} (error: {type(error).__name__})") continue diff --git a/graphfaker/domains/__init__.py b/graphfaker/domains/__init__.py index b3d9518..8943f2c 100644 --- a/graphfaker/domains/__init__.py +++ b/graphfaker/domains/__init__.py @@ -5,7 +5,7 @@ to add one. """ -from graphfaker.domains import fraud, social +from graphfaker.domains import coordination, fraud, social from graphfaker.domains.registry import Domain, available, get, register register( @@ -17,6 +17,25 @@ schema=social.schema, ) ) +register( + Domain( + name="coordination", + summary=( + "A social platform: accounts, topics, devices; follows, posts, reshares and replies " + "over time; labelled coordination campaigns with organic decoys." + ), + generate=coordination.generate, + options=coordination.CoordinationConfig, + schema=lambda **options: coordination_schema(options), + schema_note=( + "This is the entity half of the coordination domain: its node types, attribute samplers\n" + "and the latent interest-community factor. The follow graph, the activity stream and the\n" + "campaigns come from a process in code (graphfaker/domains/coordination/process.py,\n" + "playbooks.py), not from the schema, so generating from this file gives the entities only.\n" + "Use `graphfaker generate coordination` for the platform." + ), + ) +) register( Domain( name="fraud", @@ -33,4 +52,19 @@ ) ) -__all__ = ["Domain", "available", "fraud", "get", "register", "social"] +def coordination_schema(options: dict): + """The coordination pack's entity schema, for ``graphfaker schema``.""" + from graphfaker.domains.coordination import entities + + return entities.schema(coordination.CoordinationConfig(**options)) + + +__all__ = [ + "Domain", + "available", + "coordination", + "fraud", + "get", + "register", + "social", +] diff --git a/graphfaker/domains/coordination/__init__.py b/graphfaker/domains/coordination/__init__.py new file mode 100644 index 0000000..d65d359 --- /dev/null +++ b/graphfaker/domains/coordination/__init__.py @@ -0,0 +1,32 @@ +"""The coordinated-behaviour domain pack: a social platform's accounts, topics +and devices; an organic activity process; injected, labelled coordination +campaigns with organic decoys and a measurable tradecraft level.""" + +from graphfaker.domains.coordination.config import ( + DECOY_PLAYBOOKS, + PLAYBOOKS, + CoordinationConfig, + TradecraftProfile, +) +from graphfaker.domains.coordination.evaluate import Evaluation, evaluate +from graphfaker.domains.coordination.generate import generate +from graphfaker.domains.coordination.hardness import ( + HardnessReport, + account_features, + hardness_report, + realism_report, +) + +__all__ = [ + "DECOY_PLAYBOOKS", + "PLAYBOOKS", + "CoordinationConfig", + "Evaluation", + "HardnessReport", + "TradecraftProfile", + "account_features", + "evaluate", + "generate", + "hardness_report", + "realism_report", +] diff --git a/graphfaker/domains/coordination/config.py b/graphfaker/domains/coordination/config.py new file mode 100644 index 0000000..5c543a9 --- /dev/null +++ b/graphfaker/domains/coordination/config.py @@ -0,0 +1,257 @@ +"""Configuration for the coordinated-behaviour domain pack. + +A social platform: accounts, the topics they post about, the devices they post +from, and the follows, posts, reshares and replies between them. Inside it, +labelled **campaigns**: groups of accounts acting together, some of them +inauthentic and some of them not. + +Why the decoys matter more here than in the fraud pack +----------------------------------------------------- +Laundering has a weak legitimate twin: a payroll fan-out really does look like +a mule fan-out, but most fan-ins in a bank are not suspicious and a detector +that flags all of them is obviously wrong. Coordinated behaviour has a *strong* +legitimate twin. A fandom reacting to an album drop, a city reacting to an +earthquake, and a paid amplification ring are all, structurally, the same +thing: a burst of near-simultaneous, near-identical activity from a tightly +connected group. + +So a coordination dataset whose only labelled structures are inauthentic is +worse than useless: any detector that fires on synchrony scores perfectly, and +that is exactly the detector that ruins real platforms by suspending fan clubs +and protest movements. Here the organic bursts are generated by the same +machinery as the campaigns, labelled ``is_coordinated=False``, and counted as +false positives when flagged. ``decoy_ratio`` is 1.0 at ``tradecraft="high"``: +as many organic structures as inauthentic ones. + +``tradecraft`` is the hardness dial. Like the fraud pack's, it is an input to a +measurement rather than an assertion: :func:`..hardness.hardness_report` scores +what a single-feature detector can still see at each level. +""" + +from __future__ import annotations + +import datetime as dt +from typing import Literal + +from pydantic import BaseModel, ConfigDict, Field, model_validator + +from graphfaker.schema.patterns import Camouflage, PatternCatalog, PatternSpec + +Tradecraft = Literal["low", "medium", "high"] + +#: Base sizes at scale 1.0. A platform sees far more events per account than a +#: bank sees transactions per account, hence the ratio. +BASE_ACCOUNTS = 5_000_000 +BASE_EVENTS = 100_000_000 +BASE_CAMPAIGNS = 1_000 + +#: Inauthentic playbooks, in the order campaigns are allocated. Each is a +#: coordination *shape* documented in the open literature on influence +#: operations; none of them is a recipe for running one, and none of them +#: generates message content beyond a template id. +#: What the pack injects: the eight inauthentic playbooks with their share of +#: the campaign budget and their natural span in days before +#: ``timing_jitter_hours`` stretches it, then the three organic decoys, each +#: naming the playbook it imitates. The order is the order campaigns are +#: allocated and injected, and a seed only reproduces a dataset because of it. +#: +#: Each playbook is a coordination *shape* documented in the open literature +#: on influence operations; none of them is a recipe for running one, and none +#: generates message content beyond a template id. +CATALOG = PatternCatalog( + base=BASE_CAMPAIGNS, + floor=2, + patterns=[ + PatternSpec(name="copypasta", share=0.16, span_days=0.05), + PatternSpec(name="amplification_ring", share=0.16, span_days=0.2), + PatternSpec(name="reply_brigade", share=0.12, span_days=0.1), + PatternSpec(name="follow_farm", share=0.12, span_days=2.0), + PatternSpec(name="hashtag_flood", share=0.14, span_days=0.3), + PatternSpec(name="sockpuppet_cluster", share=0.12, span_days=5.0), + PatternSpec(name="astroturf_campaign", share=0.10, span_days=30.0), + PatternSpec(name="account_handover", share=0.08, span_days=10.0), + # The organic twins. A fandom reacting to a release and a paid + # amplification ring are structurally the same thing, so a dataset + # whose only labelled structures are inauthentic rewards any detector + # that fires on synchrony, which is the detector that suspends fan + # clubs. Each decoy is matched to the playbook it imitates on the + # properties that are not the point, chiefly posting volume. + PatternSpec(name="fandom_burst", imitates="hashtag_flood", span_days=0.3), + PatternSpec(name="breaking_news", imitates="copypasta", span_days=0.15), + PatternSpec(name="mutual_follow_community", imitates="follow_farm", span_days=20.0), + ], +) + +#: The inauthentic playbooks, in allocation order. +PLAYBOOKS = CATALOG.injected +#: The organic structures, which are labelled ``is_coordinated=False``. +DECOY_PLAYBOOKS = CATALOG.decoys +#: Which inauthentic playbook each decoy is the twin of, for the hardness +#: report: a decoy is only doing its job if it is hard to separate from its +#: twin. +DECOY_TWIN = CATALOG.twins + + +class TradecraftProfile(Camouflage): + """What a tradecraft level does to injected campaigns. + + ``text_blend``: 0 leaves every account in a burst posting the same template + id, which is the signature a duplicate-text detector is written to catch; 1 + draws template ids from the organic distribution of the topic, so text + similarity alone carries nothing. + + ``timing_jitter_hours``: the window a campaign's events are spread over. + Sub-minute synchrony is the classic tell and the easiest thing in the world + to detect; real operations stagger. + + ``activity_camouflage``: fraction of campaign accounts that also carry + organic activity at the population rate, so posting volume and topic + breadth do not single them out. + + ``overlap``: probability a campaign reuses an account already in another, + producing the overlapping clusters that make attribution hard. + + ``decoy_ratio``: organic structures with the same shape, as a fraction of + the inauthentic campaign count. Labelled ``is_coordinated=False``. + + ``account_age_blend``: 0 creates every campaign account just before the + campaign (a bloc of same-day signups is the single loudest signal in real + platform data); 1 uses accounts aged like the population, which is what + buying or farming aged accounts achieves. + + ``size_scale``: multiplier on campaign sizes. + + Degree is the signal hardness cannot blend away, and the fraud pack + learned this the hard way: recruiting campaign members from the most + active accounts was tried as camouflage and measured to do the + opposite, because hubs are outliers already. The same holds here, so + recruitment is uniform over eligible accounts and ``size_scale`` is the + lever that keeps cluster size from being a giveaway. + """ + + #: How aged the campaign's accounts are. This dial is the platform's + #: own: a bloc of same-day signups has no equivalent in a bank. + account_age_blend: float = Field(ge=0.0, le=1.0) + + #: The shared dials are :class:`~graphfaker.schema.patterns.Camouflage`; + #: these names are what a platform reader calls them. On a platform the + #: signature is the text, so ``text_blend`` is the signature blend. + @property + def text_blend(self) -> float: + return self.signature_blend + + @property + def timing_jitter_hours(self) -> float: + return self.timing_spread + + +TRADECRAFT_PROFILES: dict[str, TradecraftProfile] = { + "low": TradecraftProfile( + signature_blend=0.0, + timing_spread=0.05, + activity_camouflage=0.0, + overlap=0.0, + decoy_ratio=0.0, + account_age_blend=0.0, + size_scale=1.0, + ), + "medium": TradecraftProfile( + signature_blend=0.5, + timing_spread=6.0, + activity_camouflage=0.6, + overlap=0.2, + decoy_ratio=0.5, + account_age_blend=0.5, + size_scale=0.75, + ), + "high": TradecraftProfile( + signature_blend=0.9, + timing_spread=72.0, + activity_camouflage=1.0, + overlap=0.4, + decoy_ratio=1.0, + account_age_blend=1.0, + size_scale=0.5, + ), +} + + + +class CoordinationConfig(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + #: ``1.0`` = ~5M accounts / ~100M events. + scale: float = Field(default=0.001, gt=0.0) + tradecraft: Tradecraft = "medium" + #: Activity period. + period_start: dt.date = dt.date(2026, 1, 1) + period_days: int = Field(default=90, ge=7) + locale: str = "en_US" + #: Latent interest communities; topics, attributes and who follows whom are + #: conditioned on them. + communities: int = Field(default=12, ge=1) + #: Distinct topics (hashtags, subjects) accounts post about. + topics: int | None = None + #: Mean follows per account. Real platforms run into the hundreds; the + #: default keeps the follow graph a similar size to the event stream so + #: neither dominates a dataset. + follows_per_account: int = Field(default=15, ge=0) + #: Override campaigns per playbook. ``None`` derives them from scale. + campaigns: dict[str, int] | None = None + + # ---------------------------------------------------------------- derived + + @property + def num_accounts(self) -> int: + return max(200, round(BASE_ACCOUNTS * self.scale)) + + @property + def num_topics(self) -> int: + if self.topics is not None: + return self.topics + # Enough that a campaign pushing one topic is not automatically the + # whole conversation, and at least a few per community. + return max(4 * self.communities, int(self.num_accounts * 0.01)) + + @property + def num_devices(self) -> int: + # Most accounts post from one device; sockpuppet clusters share. + return max(150, int(self.num_accounts * 0.9)) + + @property + def num_follows(self) -> int: + return self.num_accounts * self.follows_per_account + + @property + def num_events(self) -> int: + return max(4_000, round(BASE_EVENTS * self.scale)) + + @property + def num_campaigns(self) -> int: + if self.campaigns is not None: + return sum(self.campaigns.values()) + # At least two of every playbook, so a small dataset still covers the + # catalogue. + return CATALOG.total(self.scale) + + @property + def campaign_counts(self) -> dict[str, int]: + if self.campaigns is not None: + return {name: self.campaigns.get(name, 0) for name in PLAYBOOKS} + return CATALOG.counts(self.num_campaigns) + + @property + def profile(self) -> TradecraftProfile: + return TRADECRAFT_PROFILES[self.tradecraft] + + @property + def period_end(self) -> dt.date: + return self.period_start + dt.timedelta(days=self.period_days) + + @model_validator(mode="after") + def _known_playbooks(self) -> CoordinationConfig: + if self.campaigns: + unknown = set(self.campaigns) - set(PLAYBOOKS) + if unknown: + raise ValueError(f"unknown playbooks: {sorted(unknown)}") + return self diff --git a/graphfaker/domains/coordination/entities.py b/graphfaker/domains/coordination/entities.py new file mode 100644 index 0000000..ef773df --- /dev/null +++ b/graphfaker/domains/coordination/entities.py @@ -0,0 +1,389 @@ +"""Entities of the coordination pack: accounts, topics, devices, and the +structural edges between them (USES, FOLLOWS). + +Attributes are declared as schema node types and drawn by the generic engine, +so the same samplers, latent factors and sharding apply. The follow graph is +built vectorised on top, because a follow graph is the part of a social +platform that has to be right: its in-degree is heavy-tailed over several +orders of magnitude, its communities are recoverable, and a meaningful share of +its edges are reciprocal. Uniform attachment gets none of those. + +**No attribute on any node reveals campaign membership.** There is no +``is_bot`` and no ``suspicious`` column. Everything a detector could legitimately +use is here; everything else lives in ``run.truth``, which ``--blind`` leaves +out entirely. A label leaked into a feature makes the whole dataset worthless +as a benchmark, and it is the easiest mistake to make. +""" + +from __future__ import annotations + +import datetime as dt + +import numpy as np +import polars as pl + +from graphfaker.backends.tables import ID +from graphfaker.domains.coordination.config import CoordinationConfig +from graphfaker.engine.run import node_pool +from graphfaker.engine.sampling import sample_latent, sample_nodes +from graphfaker.engine.seeding import Streams +from graphfaker.schema import ( + BernoulliSampler, + CategorySampler, + FakerSampler, + GaussianSampler, + GraphSchema, + LatentFactor, + LognormalSampler, + NodeType, + UniformTopology, +) + +COMMUNITY = "community" + +TOPIC_CATEGORIES = [ + "politics", + "sport", + "music", + "technology", + "finance", + "health", + "gaming", + "film", + "local_news", + "science", +] +TOPIC_CATEGORY_WEIGHTS = [14, 14, 12, 11, 8, 8, 11, 9, 7, 6] + +#: Share of follows that are followed back. Reciprocity on real platforms sits +#: in this range and is the property a follow-farm exaggerates to ~1.0, so the +#: organic value has to be substantial or the farm is trivially detectable. +RECIPROCITY = 0.30 + +#: Share of follow targets drawn from the follower's own interest community. +SAME_COMMUNITY_FOLLOW_RATE = 0.72 + +#: Probability a follow ignores popularity and picks uniformly. Keeps the tail +#: from running away and stops low-degree accounts being unreachable. +UNIFORM_FOLLOW_RATE = 0.15 + +#: How much an account's accumulated followers count towards attracting the +#: next one, relative to its intrinsic activity. Above 1 because otherwise the +#: lognormal activity draw dominates and the follower distribution never +#: develops a tail: measured follower Gini went from 0.36 to 0.6+ at 3.0, and +#: real platforms are higher still. +FOLLOWER_ATTACHMENT = 3.0 + +#: Share of devices that legitimately serve many accounts, and how many they +#: serve. Shared family tablets, internet cafés, corporate NAT, and plain +#: browser-fingerprint collisions all put unrelated accounts behind one +#: identifier. Without this tail a sockpuppet cluster of seven accounts on one +#: device is separable from every organic device by a single feature, and it +#: measured AUC 1.000 at *every* tradecraft level, which made the dial look +#: broken for that playbook. +BUSY_DEVICE_SHARE = 0.02 +BUSY_DEVICE_ACCOUNTS = (3, 12) + + +def schema(config: CoordinationConfig) -> GraphSchema: + """Node types of the coordination graph. + + Edges are built by the pack rather than by the generic topology models, so + the schema declares none: a follow graph needs directed preferential + attachment with reciprocity, which is not one of the schema's topology + models. + """ + community = LatentFactor( + name=COMMUNITY, + groups=config.communities, + params={ + # How chatty a community is, and how much it reshares rather than + # posts. Both feed the organic process, and both are what a + # campaign's activity has to look ordinary against. + "log_activity": GaussianSampler(mean=0.0, sd=0.35), + "reshare_bias": GaussianSampler(mean=0.5, sd=0.12), + }, + ) + account = NodeType( + name="Account", + count=config.num_accounts, + id_prefix="acc", + attributes={ + "handle": FakerSampler(provider="user_name"), + "display_name": FakerSampler(provider="name"), + # Posting propensity, heavy-tailed like real activity: most + # accounts are near-silent and a few never stop. + "activity": LognormalSampler(mu="@community.log_activity", sigma=0.9, decimals=4), + "reshare_rate": GaussianSampler( + mean="@community.reshare_bias", sd=0.15, low=0.0, high=1.0, decimals=3 + ), + "verified": BernoulliSampler(p=0.004), + "profile_complete": BernoulliSampler(p=0.72), + "language": CategorySampler( + values=["en", "es", "pt", "fr", "de", "id"], weights=[55, 12, 10, 8, 8, 7] + ), + }, + ) + topic = NodeType( + name="Topic", + count=config.num_topics, + id_prefix="top", + attributes={ + "name": FakerSampler(provider="word"), + "category": CategorySampler( + values=TOPIC_CATEGORIES, weights=TOPIC_CATEGORY_WEIGHTS + ), + # A handful of topics take most of the traffic. + "prominence": LognormalSampler(mu=0.0, sigma=1.3, decimals=4), + }, + ) + device = NodeType( + name="Device", + count=config.num_devices, + id_prefix="dev", + attributes={ + "fingerprint": FakerSampler(provider="uuid4"), + "device_type": CategorySampler( + values=["mobile", "desktop", "tablet"], weights=[72, 22, 6] + ), + "os": CategorySampler( + values=["android", "ios", "windows", "macos", "linux"], + weights=[44, 30, 15, 9, 2], + ), + }, + ) + return GraphSchema( + name="coordination", + latent=[community], + nodes=[account, topic, device], + topology=UniformTopology(), + ) + + +def build_nodes( + config: CoordinationConfig, streams: Streams, shard_size: int, workers: int = 1 +) -> tuple[dict[str, pl.DataFrame], dict]: + """Sample every node table and return them with the latent group params.""" + node_schema = schema(config) + latent = sample_latent(node_schema, streams) + children = dict(zip((n.name for n in node_schema.nodes), streams.spawn(len(node_schema.nodes)))) + tables: dict[str, pl.DataFrame] = {} + with node_pool(workers) as executor: + for node in node_schema.generation_order(): + tables[node.name] = sample_nodes( + node, + node_schema, + latent, + tables, + children[node.name], + shard_size, + workers=workers, + executor=executor, + ) + return {node.name: tables[node.name] for node in node_schema.nodes}, latent + + +def add_account_dates( + accounts: pl.DataFrame, config: CoordinationConfig, rng: np.random.Generator +) -> pl.DataFrame: + """``created_at`` strictly before the period. + + Ages are exponential with a long mean: most accounts are a year or two old + and a few date back a decade. Campaign accounts overwrite this when + ``account_age_blend`` is low, which is what makes a bloc of same-day + signups visible; the organic distribution is what that bloc stands out + against. + """ + age_days = rng.exponential(scale=700.0, size=accounts.height).astype(np.int64) + 1 + created = [config.period_start - dt.timedelta(days=int(d)) for d in age_days] + return accounts.with_columns(pl.Series("created_at", created, dtype=pl.Date)) + + +def uses_edges( + accounts: pl.DataFrame, + devices: pl.DataFrame, + config: CoordinationConfig, + rng: np.random.Generator, +) -> pl.DataFrame: + """Account -USES-> Device. + + Fewer devices than accounts, so sharing happens organically: people use a + family tablet, and two accounts run by the same person share a phone. + That innocent collision rate is what a sockpuppet cluster's shared + fingerprint has to be distinguishable from. + """ + n_acc, n_dev = accounts.height, devices.height + acc_ids = accounts[ID].to_numpy() + dev_ids = devices[ID].to_numpy() + + primary = np.empty(n_acc, dtype=np.int64) + # Give the first min(n_acc, n_dev) accounts a device of their own, then + # share the remainder out. Shuffled so device index does not correlate + # with account index (and therefore with community). + order = rng.permutation(n_acc) + direct = min(n_acc, n_dev) + primary[order[:direct]] = np.arange(direct) + if n_acc > direct: + primary[order[direct:]] = rng.integers(0, n_dev, size=n_acc - direct) + + # A few devices legitimately serve a crowd. Reassigning some accounts onto + # them (rather than adding extra edges) keeps one device per account, so + # ``device_shared_with`` still means "how many accounts share my device". + n_busy = max(1, int(n_dev * BUSY_DEVICE_SHARE)) if n_dev else 0 + if n_busy and n_acc > n_busy: + busy = rng.choice(n_dev, size=n_busy, replace=False) + low, high = BUSY_DEVICE_ACCOUNTS + for device in busy: + crowd = int(rng.integers(low, high + 1)) + movers = rng.choice(n_acc, size=min(crowd, n_acc), replace=False) + primary[movers] = device + + days_before = rng.exponential(scale=350.0, size=n_acc).astype(np.int64) + 1 + first_seen = [config.period_start - dt.timedelta(days=int(d)) for d in days_before] + return pl.DataFrame( + { + "source": acc_ids, + "target": dev_ids[primary], + "first_seen": pl.Series(first_seen, dtype=pl.Date), + } + ) + + +def follows_edges( + accounts: pl.DataFrame, + config: CoordinationConfig, + rng: np.random.Generator, +) -> tuple[pl.DataFrame, np.ndarray]: + """Account -FOLLOWS-> Account, plus each account's follower count. + + Three mechanisms, each supplying a property a uniform random follow graph + lacks: + + * **Heavy-tailed out-degree.** Who does the following is drawn by + ``activity`` too, so some accounts follow hundreds and most follow a + handful. Drawing sources uniformly gave every account the same + ``following`` count, which made out-degree useless as a feature and + flattered any detector that keyed on it. + * **Preferential attachment on the target.** Follower counts on a real + platform span orders of magnitude. Targets are drawn in proportion to + ``activity`` and to the followers they already have, which produces that + tail; without it every account has the same audience and the notion of + an amplification target is meaningless. + * **Community homophily.** Most follows stay inside the follower's + interest community, so communities are recoverable and a campaign + clustered in one is not automatically anomalous. + * **Reciprocity.** A third of follows are followed back. A follow farm + pushes reciprocity to ~1.0 inside its cluster, so the organic rate has + to be substantial or the farm is found by one ratio. + + Returns the edge frame and the in-degree array, which the account table + exposes as ``follower_count``. + """ + n = accounts.height + community = accounts[COMMUNITY].to_numpy() + activity = accounts["activity"].to_numpy().astype(np.float64) + activity = np.where(activity > 0, activity, 1e-9) + + by_community: dict[int, np.ndarray] = { + int(c): np.flatnonzero(community == c) for c in np.unique(community) + } + + target_budget = max(0, config.num_follows) + if target_budget == 0 or n < 2: + empty = pl.DataFrame({"source": [], "target": [], "since": []}).with_columns( + pl.col("since").cast(pl.Date) + ) + return empty, np.zeros(n, dtype=np.int64) + + # Followers accumulate, so attachment is to (followers + activity): the + # repeated-entry trick would need a growing list per community, and a + # running weight array is simpler and vectorises per block. + followers = np.zeros(n, dtype=np.float64) + src_parts: list[np.ndarray] = [] + dst_parts: list[np.ndarray] = [] + + # Work in blocks so weights are refreshed as the graph grows without + # recomputing them for every single edge. + block = max(1024, target_budget // 32) + produced = 0 + while produced < target_budget: + take = min(block, target_budget - produced) + sources = rng.choice(n, size=take, p=activity / activity.sum()) + targets = np.empty(take, dtype=np.int64) + + weight = activity + FOLLOWER_ATTACHMENT * followers + local_choice = rng.random(take) < SAME_COMMUNITY_FOLLOW_RATE + uniform_choice = rng.random(take) < UNIFORM_FOLLOW_RATE + + for group, members in by_community.items(): + mask = local_choice & (community[sources] == group) + count = int(mask.sum()) + if not count or len(members) == 0: + continue + w = weight[members] + total = w.sum() + if total <= 0: + targets[mask] = rng.choice(members, size=count) + else: + targets[mask] = rng.choice(members, size=count, p=w / total) + + rest = ~local_choice + count = int(rest.sum()) + if count: + total = weight.sum() + targets[rest] = ( + rng.choice(n, size=count, p=weight / total) + if total > 0 + else rng.integers(0, n, size=count) + ) + # The uniform share overwrites whatever was chosen, popularity and all. + count = int(uniform_choice.sum()) + if count: + targets[uniform_choice] = rng.integers(0, n, size=count) + + keep = sources != targets + sources, targets = sources[keep], targets[keep] + np.add.at(followers, targets, 1.0) + src_parts.append(sources) + dst_parts.append(targets) + produced += take + + src = np.concatenate(src_parts) + dst = np.concatenate(dst_parts) + + # Reciprocate a share, then drop duplicate pairs once. Held in + # temporaries: reassigning src before reading it again works only because + # the original happens to be a prefix of the new array, which is exactly + # the kind of accident that breaks on the next edit. + recip = rng.random(len(src)) < RECIPROCITY + back_src, back_dst = dst[recip], src[recip] + src = np.concatenate([src, back_src]) + dst = np.concatenate([dst, back_dst]) + + acc_ids = accounts[ID].to_numpy() + days_before = rng.exponential(scale=250.0, size=len(src)).astype(np.int64) + 1 + since = np.array( + [config.period_start - dt.timedelta(days=int(d)) for d in days_before], dtype=object + ) + edges = ( + pl.DataFrame( + { + "source": acc_ids[src], + "target": acc_ids[dst], + "since": pl.Series(since.tolist(), dtype=pl.Date), + } + ) + # maintain_order, or "first" is whichever row the hash table + # happened to visit first and the run is not reproducible. + .unique(subset=["source", "target"], keep="first", maintain_order=True) + ) + + counts = ( + edges.group_by("target") + .len() + .join(accounts.select(pl.col(ID).alias("target")), on="target", how="right") + .fill_null(0) + ) + lookup = dict(zip(counts["target"].to_list(), counts["len"].to_list())) + follower_count = np.array([lookup.get(a, 0) for a in acc_ids], dtype=np.int64) + return edges, follower_count diff --git a/graphfaker/domains/coordination/evaluate.py b/graphfaker/domains/coordination/evaluate.py new file mode 100644 index 0000000..be3a7c3 --- /dev/null +++ b/graphfaker/domains/coordination/evaluate.py @@ -0,0 +1,239 @@ +"""Score a detector's output against the ground truth. + +Three granularities, mirroring the fraud pack's evaluator so the two are read +the same way: + +* **account**: did the detector flag the accounts taking part in a campaign? +* **event**: did it flag the posts, reshares and replies a campaign created? +* **campaign**: is a whole campaign considered found? A campaign counts as + detected when at least ``cluster_threshold`` of its accounts are flagged. + +**Organic structures are legitimate: flagging their accounts is a false +positive.** That is what they are there to measure, and it is the difference +between this harness and a synchrony detector's own scorecard. A detector that +fires on every burst gets recall 1.0 and precision near the coordinated share, +and this is where that shows up. + +``organic_false_positive_rate`` is reported separately from overall precision, +because it is the number that decides whether a detector is deployable: on a +real platform the cost of suspending a fan club is not symmetric with the cost +of missing one amplification ring. +""" + +from __future__ import annotations + +from collections.abc import Iterable +from dataclasses import dataclass, field +from pathlib import Path + +import polars as pl + +from graphfaker.engine.run import GraphRun + +#: Share of a campaign's accounts that must be flagged for it to count as found. +DEFAULT_CLUSTER_THRESHOLD = 0.5 + + +@dataclass +class Scores: + tp: int = 0 + fp: int = 0 + fn: int = 0 + tn: int = 0 + + @property + def precision(self) -> float: + return self.tp / (self.tp + self.fp) if (self.tp + self.fp) else 0.0 + + @property + def recall(self) -> float: + return self.tp / (self.tp + self.fn) if (self.tp + self.fn) else 0.0 + + @property + def f1(self) -> float: + p, r = self.precision, self.recall + return 2 * p * r / (p + r) if (p + r) else 0.0 + + def as_dict(self) -> dict[str, float | int]: + return { + "tp": self.tp, + "fp": self.fp, + "fn": self.fn, + "tn": self.tn, + "precision": round(self.precision, 4), + "recall": round(self.recall, 4), + "f1": round(self.f1, 4), + } + + +@dataclass +class Evaluation: + account: Scores = field(default_factory=Scores) + event: Scores = field(default_factory=Scores) + campaign: Scores = field(default_factory=Scores) + #: Share of organic (``is_coordinated=False``) accounts that were flagged. + organic_false_positive_rate: float = 0.0 + #: Per-playbook recall, so a detector that only finds bursts is visible as + #: one: it will score near zero on ``follow_farm``. + recall_by_playbook: dict[str, float] = field(default_factory=dict) + + def as_dict(self) -> dict[str, object]: + return { + "account": self.account.as_dict(), + "event": self.event.as_dict(), + "campaign": self.campaign.as_dict(), + "organic_false_positive_rate": round(self.organic_false_positive_rate, 4), + "recall_by_playbook": { + k: round(v, 4) for k, v in sorted(self.recall_by_playbook.items()) + }, + } + + def summary(self) -> str: + lines = [f"{'level':<12}{'precision':>11}{'recall':>9}{'f1':>8}{'tp':>8}{'fp':>8}{'fn':>8}"] + for name, scores in ( + ("account", self.account), + ("event", self.event), + ("campaign", self.campaign), + ): + lines.append( + f"{name:<12}{scores.precision:>11.3f}{scores.recall:>9.3f}" + f"{scores.f1:>8.3f}{scores.tp:>8}{scores.fp:>8}{scores.fn:>8}" + ) + lines.append("") + lines.append( + f"organic accounts flagged: {self.organic_false_positive_rate:.1%} " + "(these are legitimate; flagging them is the deployment cost)" + ) + if self.recall_by_playbook: + lines.append("") + lines.append("recall by playbook:") + for playbook, value in sorted( + self.recall_by_playbook.items(), key=lambda kv: -kv[1] + ): + lines.append(f" {playbook:<24}{value:>7.3f}") + return "\n".join(lines) + + +def _flagged(values: Iterable[str] | None) -> set[str]: + return set() if values is None else {str(v) for v in values} + + +def evaluate( + run: GraphRun, + flagged_accounts: Iterable[str] | None = None, + flagged_events: Iterable[str] | None = None, + cluster_threshold: float = DEFAULT_CLUSTER_THRESHOLD, +) -> Evaluation: + """Score flagged accounts and events against the run's truth. + + Args: + run: The generated run, with ``truth`` present. A ``--blind`` dataset + has no truth and cannot be scored; score against the run that wrote + it, or a truth-loaded copy. + flagged_accounts: Account ids the detector considers coordinated. + flagged_events: Event ids the detector considers part of a campaign. + cluster_threshold: Share of a campaign's accounts that must be flagged + for the campaign to count as found. + """ + accounts_truth = run.truth.get("accounts") + if accounts_truth is None: + raise ValueError( + "this run has no ground truth; evaluate against the run that generated " + "the dataset, or a copy loaded without --blind" + ) + + flagged_acc = _flagged(flagged_accounts) + flagged_ev = _flagged(flagged_events) + evaluation = Evaluation() + + # ------------------------------------------------------------- accounts + # An account in any coordinated campaign is positive. An account only in + # organic structures is negative, and deliberately so. + coordinated: set[str] = set() + organic: set[str] = set() + for account, is_coordinated in zip( + accounts_truth["account_id"].to_list(), accounts_truth["is_coordinated"].to_list() + ): + (coordinated if is_coordinated else organic).add(str(account)) + organic -= coordinated + + all_accounts = {str(a) for a in run.tables.nodes["Account"][run.tables.nodes["Account"].columns[0]].to_list()} + negatives = all_accounts - coordinated + + evaluation.account.tp = len(flagged_acc & coordinated) + evaluation.account.fp = len(flagged_acc & negatives) + evaluation.account.fn = len(coordinated - flagged_acc) + evaluation.account.tn = len(negatives - flagged_acc) + + if organic: + evaluation.organic_false_positive_rate = len(flagged_acc & organic) / len(organic) + + # --------------------------------------------------------------- events + events_truth = run.truth.get("events") + if events_truth is not None and events_truth.height: + coordinated_events = { + str(e) + for e, flag in zip( + events_truth["event_id"].to_list(), events_truth["is_coordinated"].to_list() + ) + if flag + } + total_events = sum( + frame.height for name, frame in run.tables.edges.items() if "event_id" in frame.columns + ) + evaluation.event.tp = len(flagged_ev & coordinated_events) + evaluation.event.fp = len(flagged_ev - coordinated_events) + evaluation.event.fn = len(coordinated_events - flagged_ev) + evaluation.event.tn = max( + 0, total_events - len(coordinated_events) - evaluation.event.fp + ) + + # ------------------------------------------------------------ campaigns + campaigns_truth = run.truth.get("campaigns") + if campaigns_truth is not None and campaigns_truth.height: + found_by_playbook: dict[str, list[bool]] = {} + for row in campaigns_truth.iter_rows(named=True): + members = [str(a) for a in (row.get("accounts") or [])] + if not members: + continue + share = len(flagged_acc & set(members)) / len(members) + found = share >= cluster_threshold + if row["is_coordinated"]: + found_by_playbook.setdefault(row["playbook"], []).append(found) + if found: + evaluation.campaign.tp += 1 + else: + evaluation.campaign.fn += 1 + elif found: + # An organic structure the detector decided was a campaign. + evaluation.campaign.fp += 1 + else: + evaluation.campaign.tn += 1 + evaluation.recall_by_playbook = { + playbook: sum(results) / len(results) + for playbook, results in found_by_playbook.items() + if results + } + + return evaluation + + +def read_flagged(path: str | Path, column: str) -> list[str]: + """Read a detector's output from CSV or Parquet. + + Accepts either a single-column file or one with a named column, so a + detector can hand over whatever it already writes. + """ + target = Path(path) + frame = ( + pl.read_parquet(target) + if target.suffix in {".parquet", ".pq"} + else pl.read_csv(target) + ) + if column in frame.columns: + return [str(v) for v in frame[column].to_list()] + if frame.width == 1: + return [str(v) for v in frame[frame.columns[0]].to_list()] + raise ValueError( + f"{target} has no column {column!r}; found {frame.columns}" + ) diff --git a/graphfaker/domains/coordination/generate.py b/graphfaker/domains/coordination/generate.py new file mode 100644 index 0000000..26c7adf --- /dev/null +++ b/graphfaker/domains/coordination/generate.py @@ -0,0 +1,444 @@ +"""Assemble a coordination graph: entities, follow graph, injected campaigns, +organic activity, truth, manifest. + +The order matters and is the same as the fraud pack's: campaigns run *before* +the organic process, because they decide which accounts were created late and +which have no cover activity, and the organic process has to respect both as it +goes. Generating organic activity first and filtering afterwards costs a second +copy of the event stream, and leaves accounts posting before they existed if the +filter is ever missed. +""" + +from __future__ import annotations + +import datetime as dt + +import numpy as np +import polars as pl + +from graphfaker.backends.tables import ID, GraphTables +from graphfaker.domains.coordination import entities, playbooks, process +from graphfaker.domains.coordination.config import CoordinationConfig +from graphfaker.engine.run import GraphRun, Manifest +from graphfaker.engine.sampling import DEFAULT_SHARD_SIZE +from graphfaker.engine.seeding import Streams +from graphfaker.logger import logger + +#: Rough events per campaign, used to reserve budget before injection so the +#: total lands near ``num_events``. +EVENTS_PER_CAMPAIGN = 40 + +POST_COLUMNS = ["source", "target", "event_id", "timestamp", "template_id"] +INTERACTION_COLUMNS = ["source", "target", "event_id", "timestamp", "topic"] + + +def generate( + config: CoordinationConfig | None = None, + seed: int | None = None, + shard_size: int = DEFAULT_SHARD_SIZE, + workers: int = 1, + **overrides, +) -> GraphRun: + """Generate the coordination graph. + + Args: + config: A :class:`CoordinationConfig`; keyword overrides build one. + seed: Reproducible for a given seed and shard size. + workers: Processes for entity sampling; does not affect the result. + """ + from graphfaker import __version__ + + config = CoordinationConfig( + **{**(config.model_dump() if config else {}), **overrides} + ) + root = Streams.root(seed) + node_streams, structure_streams, campaign_streams = root.spawn(3) + + # 1. Entities + tables, latent = entities.build_nodes(config, node_streams, shard_size, workers) + rng = structure_streams.rng + tables["Account"] = entities.add_account_dates(tables["Account"], config, rng) + + # 2. Topics belong to interest communities, which is what makes an + # account's topic choice local and a campaign's topic push plausible. + pop = process.Population(tables, config) + topic_community = pop.assign_topic_communities(rng) + tables["Topic"] = tables["Topic"].with_columns( + pl.Series(entities.COMMUNITY, topic_community) + ) + + # 3. The organic follow graph. + follows, _ = entities.follows_edges(tables["Account"], config, rng) + uses = entities.uses_edges(tables["Account"], tables["Device"], config, rng) + logger.info( + "coordination: %d accounts, %d topics, %d follows", + tables["Account"].height, + tables["Topic"].height, + follows.height, + ) + + # 4. Campaigns, before the organic stream. + injection = playbooks.inject( + campaign_streams.rng, pop, config, tables["Device"].height + ) + logger.info( + "coordination: %d campaigns (%d coordinated, %d organic), %d events", + len(injection.campaigns), + sum(1 for c in injection.campaigns if c.is_coordinated), + sum(1 for c in injection.campaigns if not c.is_coordinated), + sum(f.height for f in injection.events.values()), + ) + + # 5. Side effects of campaigns on entities. + tables["Account"] = _apply_fresh_accounts(tables["Account"], pop, injection) + follows = _apply_extra_follows(follows, pop, injection, config) + uses = _apply_shared_devices(uses, pop, tables["Device"], injection, config) + + # The Population was built before created_at moved, so refresh the view the + # organic process reads. + pop = process.Population(tables, config) + pop.topic_community = topic_community + pop.topics_by_community = { + int(g): ( + np.flatnonzero(topic_community == g) + if int((topic_community == g).sum()) + else np.arange(pop.n_topics) + ) + for g in np.unique(pop.account_community) + } + + stripped = ( + set(pop.account_ids.gather(sorted(injection.stripped_accounts)).to_list()) + if injection.stripped_accounts + else set() + ) + fresh = None + if injection.fresh_accounts: + fresh = pl.DataFrame( + { + "account": [ + str(pop.account_ids[i]) for i in sorted(injection.fresh_accounts) + ], + "created": [ + process.to_date(injection.fresh_accounts[i]) + for i in sorted(injection.fresh_accounts) + ], + } + ).with_columns(pl.col("created").cast(pl.Date)) + + dormant = None + if injection.dormant_until: + order = sorted(injection.dormant_until) + dormant = pl.DataFrame( + { + "account": [str(pop.account_ids[i]) for i in order], + "awake": [ + injection.dormant_until[i].astype("datetime64[us]").item() + for i in order + ], + } + ).with_columns(pl.col("awake").cast(pl.Datetime("us"))) + + def finish(part: pl.DataFrame) -> pl.DataFrame: + if stripped: + part = part.filter(~pl.col("source").is_in(stripped)) + if fresh is not None: + part = process.drop_before_creation(part, fresh) + if dormant is not None: + part = process.drop_while_dormant(part, dormant) + return part + + # 6. Organic activity, a block at a time, filtered as it is produced. + audience = process.build_audience(pop, follows) + reserve = len(injection.campaigns) * EVENTS_PER_CAMPAIGN + budget = max(0, config.num_events - reserve) + organic = process.organic_events( + structure_streams.rng, pop, budget, audience, finish=finish + ) + logger.info( + "coordination: %d organic events", + sum(part.height for parts in organic.values() for part in parts), + ) + + # 7. Merge and number in time order. + edges, event_truth = _merge_events(organic, injection) + edges["FOLLOWS"] = follows + edges["USES"] = uses + + # 8. An account cannot act before it existed. Campaign events are placed + # from the campaign's own window rather than filtered like the organic + # stream, so enforce the invariant once, globally, against the merged + # result. This is a clamp and not a filter: the account demonstrably + # existed, so it is the date that is wrong, not the event. + tables["Account"] = _clamp_creation_to_first_event(tables["Account"], edges) + # 9. follower_count is derived from the final follow graph, campaign edges + # included: a farm's members really do have those followers. + tables["Account"] = _add_follower_count(tables["Account"], follows) + + truth = { + "campaigns": _campaigns_frame(injection, pop), + "accounts": _accounts_frame(injection, pop), + "events": event_truth, + "community": pl.DataFrame( + [{"group": g, **v} for g, v in enumerate(latent[entities.COMMUNITY].values)] + ), + } + + node_schema = entities.schema(config) + order = ("FOLLOWS", "USES", process.POSTED, process.RESHARED, process.REPLIED) + ordered_edges = {k: edges[k] for k in order if k in edges} + graph = GraphTables(nodes=tables, edges=ordered_edges) + manifest = Manifest( + schema_name="coordination", + schema_digest=node_schema.digest(), + seed=seed, + shard_size=shard_size, + graphfaker_version=__version__, + created_at=dt.datetime.now(dt.timezone.utc).isoformat(timespec="seconds"), + node_counts={k: f.height for k, f in graph.nodes.items()}, + edge_counts={k: f.height for k, f in graph.edges.items()}, + extra={"coordination": config.model_dump(mode="json")}, + ) + return GraphRun(schema=node_schema, tables=graph, manifest=manifest, truth=truth) + + +# ------------------------------------------------------------------ helpers + + +def _apply_fresh_accounts(accounts, pop, injection) -> pl.DataFrame: + if not injection.fresh_accounts: + return accounts + created = accounts["created_at"].to_list() + for idx, day in injection.fresh_accounts.items(): + created[idx] = process.to_date(day) + return accounts.with_columns(pl.Series("created_at", created, dtype=pl.Date)) + + +def _apply_extra_follows(follows, pop, injection, config) -> pl.DataFrame: + """Add the follow edges farms and communities created, without duplicates.""" + if not injection.extra_follows: + return follows + ids = pop.account_ids + rows = pl.DataFrame( + { + "source": ids.gather([a for a, _ in injection.extra_follows]), + "target": ids.gather([b for _, b in injection.extra_follows]), + "since": pl.Series( + [config.period_start - dt.timedelta(days=3)] + * len(injection.extra_follows), + dtype=pl.Date, + ), + } + ) + return pl.concat([follows, rows.select(follows.columns)]).unique( + subset=["source", "target"], keep="first", maintain_order=True + ) + + +def _apply_shared_devices(uses, pop, devices, injection, config) -> pl.DataFrame: + if not injection.shared_devices: + return uses + account_ids = pop.account_ids + device_ids = devices[ID] + existing = set(zip(uses["source"].to_list(), uses["target"].to_list())) + rows = [] + for account_idx, device_idx in injection.shared_devices: + pair = (str(account_ids[account_idx]), str(device_ids[device_idx])) + if pair in existing: + continue + existing.add(pair) + rows.append( + { + "source": pair[0], + "target": pair[1], + "first_seen": config.period_start - dt.timedelta(days=1), + } + ) + if not rows: + return uses + extra = pl.DataFrame(rows).with_columns(pl.col("first_seen").cast(pl.Date)) + return pl.concat([uses, extra.select(uses.columns)]) + + +def _clamp_creation_to_first_event( + accounts: pl.DataFrame, edges: dict[str, pl.DataFrame] +) -> pl.DataFrame: + """Move ``created_at`` back where an account acts before it. + + Only ever moves a date earlier, so the fresh-signup signal that low + tradecraft relies on survives wherever it is consistent. + """ + firsts = [ + frame.select( + pl.col("source").alias(ID), pl.col("timestamp").dt.date().alias("first_event") + ) + for name, frame in edges.items() + if "event_id" in frame.columns and frame.height + ] + if not firsts: + return accounts + earliest = pl.concat(firsts).group_by(ID).agg(pl.col("first_event").min()) + return ( + accounts.join(earliest, on=ID, how="left") + .with_columns( + pl.min_horizontal("created_at", pl.col("first_event").fill_null(pl.col("created_at"))) + .alias("created_at") + ) + .drop("first_event") + ) + + +def _add_follower_count(accounts: pl.DataFrame, follows: pl.DataFrame) -> pl.DataFrame: + if follows.height == 0: + return accounts.with_columns(pl.lit(0, dtype=pl.Int64).alias("follower_count")) + counts = follows.group_by("target").len().rename({"target": ID, "len": "follower_count"}) + return accounts.join(counts, on=ID, how="left").with_columns( + pl.col("follower_count").fill_null(0).cast(pl.Int64) + ) + + +def _merge_events( + organic: dict[str, list[pl.DataFrame]], + injection: playbooks.Injection, +) -> tuple[dict[str, pl.DataFrame], pl.DataFrame]: + """Organic plus injected events per channel, numbered in time order across + channels, with the truth's event labels. + + ``event_id`` carries the rank in time across every channel, so rows stay in + generation order and a global ordering is still recoverable without sorting + the whole stream. This is the same convention the fraud pack uses for + ``tx_id``. + """ + per_channel: dict[str, list[pl.DataFrame]] = {} + for channel in process.CHANNELS: + columns = ( + playbooks.POST_COLUMNS if channel == process.POSTED else playbooks.INTERACTION_COLUMNS + ) + parts = [part.select(columns) for part in organic.pop(channel, []) if part.height] + injected = injection.events.get(channel) + if injected is not None and injected.height: + parts.append(injected.select(columns)) + if parts: + per_channel[channel] = parts + organic.clear() + + empty_truth = pl.DataFrame( + { + "event_id": pl.Series([], dtype=pl.String), + "campaign_id": pl.Series([], dtype=pl.String), + "playbook": pl.Series([], dtype=pl.String), + "is_coordinated": pl.Series([], dtype=pl.Boolean), + } + ) + if not per_channel: + return {}, empty_truth + + total = sum(part.height for parts in per_channel.values() for part in parts) + stamps = np.empty(total, dtype=np.int64) + at = 0 + for parts in per_channel.values(): + for part in parts: + stamps[at : at + part.height] = ( + part["timestamp"] + .cast(pl.Datetime("us")) + .to_numpy() + .astype("datetime64[us]") + .view(np.int64) + ) + at += part.height + order = np.argsort(stamps, kind="stable") + del stamps + ranks = np.empty(total, dtype=np.int64) + ranks[order] = np.arange(total) + del order + + by_campaign = {c.campaign_id: c for c in injection.campaigns} + edges: dict[str, pl.DataFrame] = {} + truth_parts = [] + offset = 0 + for channel in list(per_channel): + parts = per_channel.pop(channel) + finished = [] + columns = POST_COLUMNS if channel == process.POSTED else INTERACTION_COLUMNS + for i in range(len(parts)): + part, parts[i] = parts[i], None # release the original as we go + ids = pl.Series("event_id", ranks[offset : offset + part.height]) + offset += part.height + part = part.with_columns(("ev_" + ids.cast(pl.String)).alias("event_id")) + truth_parts.append( + part.filter(pl.col("campaign_id").is_not_null()).select( + ["event_id", "campaign_id"] + ) + ) + finished.append(part.select(columns)) + edges[channel] = pl.concat(finished, rechunk=False) if len(finished) > 1 else finished[0] + del ranks + + labelled = pl.concat(truth_parts) if truth_parts else empty_truth.select(["event_id", "campaign_id"]) + if labelled.height == 0: + return edges, empty_truth + event_truth = labelled.with_columns( + pl.col("campaign_id") + .map_elements(lambda c: by_campaign[c].playbook, return_dtype=pl.String) + .alias("playbook"), + pl.col("campaign_id") + .map_elements(lambda c: by_campaign[c].is_coordinated, return_dtype=pl.Boolean) + .alias("is_coordinated"), + ) + return edges, event_truth + + +def _campaigns_frame(injection: playbooks.Injection, pop: process.Population) -> pl.DataFrame: + rows = [ + { + "campaign_id": c.campaign_id, + "playbook": c.playbook, + "is_coordinated": c.is_coordinated, + "n_accounts": len(c.roles), + "n_events": c.n_events, + "topic": str(pop.topic_ids[c.topic]) if c.topic is not None else None, + "start": c.start.astype("datetime64[us]").item() if c.start is not None else None, + "end": c.end.astype("datetime64[us]").item() if c.end is not None else None, + "accounts": [str(pop.account_ids[i]) for i in c.roles], + "roles": list(c.roles.values()), + } + for c in injection.campaigns + ] + return ( + pl.DataFrame(rows) + if rows + else pl.DataFrame( + { + "campaign_id": pl.Series([], dtype=pl.String), + "playbook": pl.Series([], dtype=pl.String), + "is_coordinated": pl.Series([], dtype=pl.Boolean), + } + ) + ) + + +def _accounts_frame(injection: playbooks.Injection, pop: process.Population) -> pl.DataFrame: + rows = [ + { + "account_id": str(pop.account_ids[idx]), + "campaign_id": c.campaign_id, + "playbook": c.playbook, + "role": role, + "is_coordinated": c.is_coordinated, + } + for c in injection.campaigns + for idx, role in c.roles.items() + ] + return ( + pl.DataFrame(rows) + if rows + else pl.DataFrame( + { + "account_id": pl.Series([], dtype=pl.String), + "campaign_id": pl.Series([], dtype=pl.String), + "playbook": pl.Series([], dtype=pl.String), + "role": pl.Series([], dtype=pl.String), + "is_coordinated": pl.Series([], dtype=pl.Boolean), + } + ) + ) diff --git a/graphfaker/domains/coordination/hardness.py b/graphfaker/domains/coordination/hardness.py new file mode 100644 index 0000000..dcbc191 --- /dev/null +++ b/graphfaker/domains/coordination/hardness.py @@ -0,0 +1,432 @@ +"""Measure, rather than assert, how hard the injected coordination is to find. + +For every playbook, each single feature a naive detector might threshold on is +scored by the AUC it achieves separating that playbook's accounts from accounts +in no campaign at all. ``max_auc`` is the number to quote: at +``tradecraft="high"`` no single feature should carry much, which means the +campaign is only findable by looking at structure and timing together. + +Two things here that the fraud pack's report does not need: + +* **Decoys are scored too.** An organic structure is doing its job only if it + is *hard to separate from its twin*. ``decoy_separability`` scores each + decoy's accounts against the accounts of the inauthentic playbook it + imitates. A low number is the good outcome; a high one means the decoy is + not actually a decoy and the benchmark is easier than it looks. +* **A follow-only playbook has no events.** A follow farm barely posts, so any + report built on activity features alone will say it is undetectable. The + feature set therefore spans the follow graph, the event stream, the device + graph and account age, and the report says which family carried the signal. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + +import numpy as np +import polars as pl + +from graphfaker.backends.tables import ID +from graphfaker.domains.coordination.config import DECOY_TWIN +from graphfaker.domains.coordination.process import POSTED, REPLIED, RESHARED +from graphfaker.engine.run import GraphRun + +#: Features grouped by the family they come from, so a report can say *which* +#: kind of evidence found a playbook rather than only how much. +FEATURE_FAMILIES = { + "activity": ["event_count", "post_count", "reshare_count", "reply_count", "events_per_day"], + "structure": ["following", "follower_count", "reciprocity", "clustering_proxy"], + "timing": ["burst_share", "active_hours"], + "content": ["topic_concentration", "template_reuse"], + "identity": ["account_age_days", "device_shared_with"], +} +ACCOUNT_FEATURES = [name for names in FEATURE_FAMILIES.values() for name in names] + + +def auc(positive: np.ndarray, negative: np.ndarray) -> float: + """Rank-based AUC (Mann-Whitney), returned as distance from chance. + + Symmetrised to ``max(a, 1 - a)``: a feature that is strongly *low* for a + playbook is just as useful to a detector as one that is strongly high, and + reporting 0.05 as "weak" would be wrong. + """ + n_pos, n_neg = len(positive), len(negative) + if n_pos == 0 or n_neg == 0: + return float("nan") + joined = np.concatenate([positive, negative]) + order = joined.argsort(kind="stable") + ranks = np.empty(len(joined), dtype=np.float64) + ranks[order] = np.arange(1, len(joined) + 1) + # Average ranks within ties, or a constant feature scores 1.0. + _, inverse, counts = np.unique(joined, return_inverse=True, return_counts=True) + sums = np.zeros(len(counts)) + np.add.at(sums, inverse, ranks) + ranks = (sums / counts)[inverse] + value = (ranks[:n_pos].sum() - n_pos * (n_pos + 1) / 2) / (n_pos * n_neg) + return float(max(value, 1.0 - value)) + + +def account_features(run: GraphRun) -> pl.DataFrame: + """One row per account, with everything a detector could legitimately use. + + Nothing derived from ``run.truth`` appears here. That is the whole contract: + these are the features, the truth is the answer, and the two must not touch. + """ + accounts = run.tables.nodes["Account"] + ids = accounts[ID] + n = accounts.height + index = {str(a): i for i, a in enumerate(ids.to_list())} + + counts = {name: np.zeros(n, dtype=np.float64) for name in ("post", "reshare", "reply")} + channel_names = {POSTED: "post", RESHARED: "reshare", REPLIED: "reply"} + topic_pairs: list[tuple[int, str]] = [] + template_pairs: list[tuple[int, int]] = [] + hour_pairs: list[tuple[int, int]] = [] + + for channel, key in channel_names.items(): + frame = run.tables.edges.get(channel) + if frame is None or frame.height == 0: + continue + sources = np.array([index.get(s, -1) for s in frame["source"].to_list()]) + valid = sources >= 0 + np.add.at(counts[key], sources[valid], 1.0) + + topic_column = "target" if channel == POSTED else "topic" + if topic_column in frame.columns: + topics = frame[topic_column].to_list() + topic_pairs.extend( + (int(s), str(t)) for s, t, ok in zip(sources, topics, valid) if ok + ) + if channel == POSTED and "template_id" in frame.columns: + templates = frame["template_id"].to_list() + template_pairs.extend( + (int(s), int(t)) for s, t, ok in zip(sources, templates, valid) if ok + ) + stamps = frame["timestamp"].cast(pl.Datetime("us")).to_numpy().astype("datetime64[h]") + hours = stamps.view(np.int64) + hour_pairs.extend((int(s), int(h)) for s, h, ok in zip(sources, hours, valid) if ok) + + event_count = counts["post"] + counts["reshare"] + counts["reply"] + + # Follow graph: out-degree, in-degree, reciprocity, and a cheap clustering + # proxy (how many of an account's followees also follow each other). + following = np.zeros(n, dtype=np.float64) + reciprocal = np.zeros(n, dtype=np.float64) + follows = run.tables.edges.get("FOLLOWS") + pairs: set[tuple[int, int]] = set() + out_neighbours: dict[int, set[int]] = {} + if follows is not None and follows.height: + src = np.array([index.get(s, -1) for s in follows["source"].to_list()]) + dst = np.array([index.get(t, -1) for t in follows["target"].to_list()]) + ok = (src >= 0) & (dst >= 0) + src, dst = src[ok], dst[ok] + np.add.at(following, src, 1.0) + pairs = set(zip(src.tolist(), dst.tolist())) + for a, b in pairs: + out_neighbours.setdefault(a, set()).add(b) + for a, b in pairs: + if (b, a) in pairs: + reciprocal[a] += 1.0 + + clustering_proxy = np.zeros(n, dtype=np.float64) + for account, neighbours in out_neighbours.items(): + if len(neighbours) < 2: + continue + linked = sum( + 1 + for other in neighbours + if out_neighbours.get(other) and out_neighbours[other] & neighbours + ) + clustering_proxy[account] = linked / len(neighbours) + + follower_count = ( + accounts["follower_count"].to_numpy().astype(np.float64) + if "follower_count" in accounts.columns + else np.zeros(n) + ) + + # Content concentration and template reuse. + topic_concentration = np.zeros(n, dtype=np.float64) + if topic_pairs: + by_account: dict[int, dict[str, int]] = {} + for account, topic in topic_pairs: + by_account.setdefault(account, {}).setdefault(topic, 0) + by_account[account][topic] += 1 + for account, topics in by_account.items(): + total = sum(topics.values()) + topic_concentration[account] = max(topics.values()) / total if total else 0.0 + + template_reuse = np.zeros(n, dtype=np.float64) + if template_pairs: + shared: dict[int, set[int]] = {} + for account, template in template_pairs: + shared.setdefault(template, set()).add(account) + per_account: dict[int, list[int]] = {} + for account, template in template_pairs: + per_account.setdefault(account, []).append(template) + for account, templates in per_account.items(): + if not templates: + continue + # Share of this account's posts whose template another account also + # used: the duplicate-text signal, measured across accounts rather + # than within one. + template_reuse[account] = sum( + 1 for t in templates if len(shared.get(t, ())) > 1 + ) / len(templates) + + # Timing: the largest share of an account's events falling in one hour, and + # how many distinct hours it was active in. + burst_share = np.zeros(n, dtype=np.float64) + active_hours = np.zeros(n, dtype=np.float64) + if hour_pairs: + by_account_hours: dict[int, dict[int, int]] = {} + for account, hour in hour_pairs: + by_account_hours.setdefault(account, {}).setdefault(hour, 0) + by_account_hours[account][hour] += 1 + for account, hours in by_account_hours.items(): + total = sum(hours.values()) + burst_share[account] = max(hours.values()) / total if total else 0.0 + active_hours[account] = len(hours) + + period_days = max(1, int(run.manifest.extra.get("coordination", {}).get("period_days", 90))) + created = accounts["created_at"].to_numpy().astype("datetime64[D]") + period_start = np.datetime64( + run.manifest.extra.get("coordination", {}).get("period_start", "2026-01-01") + ).astype("datetime64[D]") + account_age_days = (period_start - created).astype(np.int64).astype(np.float64) + + device_shared_with = np.zeros(n, dtype=np.float64) + uses = run.tables.edges.get("USES") + if uses is not None and uses.height: + per_device = uses.group_by("target").len().rename({"len": "sharers"}) + joined = uses.join(per_device, on="target", how="left") + src = np.array([index.get(s, -1) for s in joined["source"].to_list()]) + sharers = joined["sharers"].to_numpy().astype(np.float64) + ok = src >= 0 + np.maximum.at(device_shared_with, src[ok], sharers[ok]) + + return pl.DataFrame( + { + "account_id": ids, + "event_count": event_count, + "post_count": counts["post"], + "reshare_count": counts["reshare"], + "reply_count": counts["reply"], + "events_per_day": event_count / period_days, + "following": following, + "follower_count": follower_count, + "reciprocity": np.divide( + reciprocal, following, out=np.zeros(n), where=following > 0 + ), + "clustering_proxy": clustering_proxy, + "burst_share": burst_share, + "active_hours": active_hours, + "topic_concentration": topic_concentration, + "template_reuse": template_reuse, + "account_age_days": account_age_days, + "device_shared_with": device_shared_with, + } + ) + + +@dataclass +class PlaybookHardness: + playbook: str + is_coordinated: bool + n_accounts: int + feature_auc: dict[str, float] = field(default_factory=dict) + + @property + def max_auc(self) -> float: + values = [v for v in self.feature_auc.values() if not np.isnan(v)] + return max(values) if values else float("nan") + + @property + def best_feature(self) -> str | None: + ranked = [(v, k) for k, v in self.feature_auc.items() if not np.isnan(v)] + return max(ranked)[1] if ranked else None + + @property + def best_family(self) -> str | None: + best = self.best_feature + if best is None: + return None + for family, names in FEATURE_FAMILIES.items(): + if best in names: + return family + return None + + +@dataclass +class HardnessReport: + tradecraft: str + playbooks: list[PlaybookHardness] + #: decoy playbook -> (AUC, feature) separating it from the playbook it + #: imitates, by the best single feature. Low is good: it means no one + #: feature tells the organic structure from the inauthentic one, and a + #: detector has to combine evidence. It should not be near 0.5 either — + #: then the task is impossible rather than hard. The feature name matters + #: as much as the number: separability on ``account_age_days`` is a real + #: signal a platform would use, while separability on ``template_reuse`` + #: means the decoy's text is not organic enough. + decoy_separability: dict[str, tuple[float, str]] = field(default_factory=dict) + + def as_dict(self) -> dict[str, Any]: + return { + "tradecraft": self.tradecraft, + "playbooks": [ + { + "playbook": p.playbook, + "is_coordinated": p.is_coordinated, + "n_accounts": p.n_accounts, + "max_auc": None if np.isnan(p.max_auc) else round(p.max_auc, 4), + "best_feature": p.best_feature, + "best_family": p.best_family, + "feature_auc": { + k: None if np.isnan(v) else round(v, 4) + for k, v in p.feature_auc.items() + }, + } + for p in self.playbooks + ], + "decoy_separability": { + k: { + "auc": None if np.isnan(v) else round(v, 4), + "feature": feature, + "twin": DECOY_TWIN.get(k), + } + for k, (v, feature) in self.decoy_separability.items() + }, + } + + def summary(self) -> str: + lines = [ + f"tradecraft: {self.tradecraft}", + f"{'playbook':<24}{'kind':<14}{'n':>6}{'max AUC':>10} best feature (family)", + ] + for p in sorted(self.playbooks, key=lambda p: -(0 if np.isnan(p.max_auc) else p.max_auc)): + kind = "coordinated" if p.is_coordinated else "organic" + auc_text = "n/a" if np.isnan(p.max_auc) else f"{p.max_auc:.3f}" + lines.append( + f"{p.playbook:<24}{kind:<14}{p.n_accounts:>6}{auc_text:>10} " + f"{p.best_feature or '-'} ({p.best_family or '-'})" + ) + if self.decoy_separability: + lines.append("") + lines.append("decoy separability from its twin (lower is better):") + for decoy, (value, feature) in sorted(self.decoy_separability.items()): + twin = DECOY_TWIN.get(decoy, "?") + text = "n/a" if np.isnan(value) else f"{value:.3f}" + lines.append(f" {decoy:<24} vs {twin:<20}{text:>7} by {feature}") + return "\n".join(lines) + + +def hardness_report(run: GraphRun, features: pl.DataFrame | None = None) -> HardnessReport: + """Score every playbook's separability from uninvolved accounts.""" + features = account_features(run) if features is None else features + truth = run.truth.get("accounts") + tradecraft = str(run.manifest.extra.get("coordination", {}).get("tradecraft", "?")) + if truth is None or truth.height == 0: + return HardnessReport(tradecraft=tradecraft, playbooks=[]) + + position = {a: i for i, a in enumerate(features["account_id"].to_list())} + members: dict[str, set[int]] = {} + coordinated: dict[str, bool] = {} + for account, playbook, is_coordinated in zip( + truth["account_id"].to_list(), + truth["playbook"].to_list(), + truth["is_coordinated"].to_list(), + ): + index = position.get(account) + if index is None: + continue + members.setdefault(playbook, set()).add(index) + coordinated[playbook] = bool(is_coordinated) + + involved = {index for group in members.values() for index in group} + clean = np.array(sorted(set(range(features.height)) - involved), dtype=np.int64) + + columns = {name: features[name].to_numpy().astype(np.float64) for name in ACCOUNT_FEATURES} + reports = [] + for playbook, group in sorted(members.items()): + rows = np.array(sorted(group), dtype=np.int64) + reports.append( + PlaybookHardness( + playbook=playbook, + is_coordinated=coordinated.get(playbook, True), + n_accounts=len(rows), + feature_auc={ + name: auc(values[rows], values[clean]) for name, values in columns.items() + }, + ) + ) + + separability = {} + for decoy, twin in DECOY_TWIN.items(): + if decoy in members and twin in members: + decoy_rows = np.array(sorted(members[decoy]), dtype=np.int64) + twin_rows = np.array(sorted(members[twin]), dtype=np.int64) + scored = [ + (auc(values[decoy_rows], values[twin_rows]), name) + for name, values in columns.items() + ] + scored = [(v, n) for v, n in scored if not np.isnan(v)] + separability[decoy] = max(scored) if scored else (float("nan"), "-") + return HardnessReport( + tradecraft=tradecraft, playbooks=reports, decoy_separability=separability + ) + + +def realism_report(run: GraphRun) -> dict[str, Any]: + """The properties of the organic platform a reader would check first.""" + features = account_features(run) + follows = run.tables.edges.get("FOLLOWS") + followers = features["follower_count"].to_numpy().astype(np.float64) + events = features["event_count"].to_numpy().astype(np.float64) + + def gini(values: np.ndarray) -> float: + if len(values) == 0 or values.sum() == 0: + return 0.0 + ordered = np.sort(values) + n = len(ordered) + weighted = ((np.arange(1, n + 1)) * ordered).sum() + return float((2 * weighted) / (n * ordered.sum()) - (n + 1) / n) + + reciprocal_share = float("nan") + if follows is not None and follows.height: + pairs = set(zip(follows["source"].to_list(), follows["target"].to_list())) + reciprocal_share = sum(1 for a, b in pairs if (b, a) in pairs) / len(pairs) + + hour_counts = np.zeros(24) + for channel in (POSTED, RESHARED, REPLIED): + frame = run.tables.edges.get(channel) + if frame is None or frame.height == 0: + continue + hours = ( + frame["timestamp"].cast(pl.Datetime("us")).dt.hour().to_numpy().astype(np.int64) + ) + np.add.at(hour_counts, hours, 1) + + return { + "accounts": int(features.height), + "events": int(events.sum()), + "follows": 0 if follows is None else int(follows.height), + "follower_gini": round(gini(followers), 4), + "follower_max": int(followers.max()) if len(followers) else 0, + "follower_mean": round(float(followers.mean()), 2) if len(followers) else 0.0, + "event_gini": round(gini(events), 4), + "silent_account_share": round(float((events == 0).mean()), 4) if len(events) else 0.0, + "reciprocal_follow_share": ( + None if np.isnan(reciprocal_share) else round(reciprocal_share, 4) + ), + "mean_reciprocity": round(float(features["reciprocity"].mean() or 0.0), 4), + "mean_clustering_proxy": round(float(features["clustering_proxy"].mean() or 0.0), 4), + # Peak-to-trough of the diurnal rhythm: flat means the hour profile + # never made it into the data. + "diurnal_ratio": ( + round(float(hour_counts.max() / max(hour_counts.min(), 1)), 2) + if hour_counts.sum() + else None + ), + } diff --git a/graphfaker/domains/coordination/playbooks.py b/graphfaker/domains/coordination/playbooks.py new file mode 100644 index 0000000..f349b75 --- /dev/null +++ b/graphfaker/domains/coordination/playbooks.py @@ -0,0 +1,696 @@ +"""Injected, labelled campaigns. + +Each playbook has a *signature*: the thing a detector was written to catch. +Identical text across accounts. A burst inside a minute. A cluster whose +follows are all mutual. A bloc of accounts created on the same day. A set of +accounts posting from one device. Tradecraft decides how much of that signature +survives: ``text_blend`` swaps identical templates for draws from the topic's +organic distribution, ``timing_jitter_hours`` stretches a burst from seconds to +days, ``activity_camouflage`` keeps ordinary activity on campaign accounts, +``account_age_blend`` uses aged accounts instead of fresh ones, ``overlap`` +lets campaigns share members, and ``decoy_ratio`` adds organic structures with +the same shape. + +Every campaign records its accounts with roles and every event it creates; the +truth tables are built from those records. + +Scope +----- +These are coordination *shapes* — who acts with whom, when, and how densely — +and every one of them is described in the public literature on platform +manipulation. Nothing here generates message content: a post carries a +``template_id``, an integer, and nothing else. The pack exists to measure +detectors, and what it withholds (text, personas, targeting) is what a detector +does not need. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +import numpy as np +import polars as pl + +from graphfaker.domains.coordination.config import ( + CATALOG, + CoordinationConfig, + TradecraftProfile, +) +from graphfaker.domains.coordination.process import ( + POSTED, + REPLIED, + RESHARED, + TEMPLATES_PER_TOPIC, + Population, +) +from graphfaker.engine.injection import ( + InjectionContext, + grouped_decoys, + run_catalog, +) +from graphfaker.engine.injection import Pattern as BasePattern + + +@dataclass +class Campaign(BasePattern): + """One injected campaign, inauthentic or organic. + + The record is :class:`graphfaker.engine.injection.Pattern`; these names + are what a platform reader calls its fields. ``campaign_id`` is the + pattern id, ``playbook`` the shape, ``is_coordinated`` whether it is the + thing being looked for. + """ + + @property + def campaign_id(self) -> str: + return self.pattern_id + + @property + def playbook(self) -> str: + return self.name + + @property + def is_coordinated(self) -> bool: + return self.labelled + + @property + def n_events(self) -> int: + return self.events + + @n_events.setter + def n_events(self, value: int) -> None: + self.events = value + + @property + def topic(self) -> int | None: + """The topic a campaign pushes, where it pushes one.""" + return self.extra.get("topic") + + @topic.setter + def topic(self, value: int | None) -> None: + self.extra["topic"] = value + + +@dataclass +class Injection: + """Everything the playbooks produced, to be merged into the run.""" + + campaigns: list[Campaign] + events: dict[str, pl.DataFrame] # channel -> frame with campaign_id + #: (account idx, device idx) pairs to add as USES edges. + shared_devices: list[tuple[int, int]] + #: (follower idx, followee idx) pairs to add as FOLLOWS edges. + extra_follows: list[tuple[int, int]] + #: account idx -> new created_at, for campaigns that open fresh accounts. + fresh_accounts: dict[int, np.datetime64] + #: account idx whose organic activity is removed (no camouflage). + stripped_accounts: set[int] + #: account idx -> the moment it woke up. Organic activity *before* this is + #: removed and activity after it is kept, which is what dormancy means. + dormant_until: dict[int, np.datetime64] + + +class PlaybookContext(InjectionContext): + """The platform's own drawing on top of the shared pattern bookkeeping. + + What is inherited: the dials, the period, who is already in a campaign, + and whether overlap is allowed. What is here: recruiting inside an + interest community, template ids, and the event rows, because those are + the platform. + """ + + def __init__( + self, + rng: np.random.Generator, + pop: Population, + config: CoordinationConfig, + n_devices: int, + ): + super().__init__(rng, config.profile, CATALOG, pop.period_start, pop.period_days) + self.pop = pop + self.config = config + self.profile: TradecraftProfile = config.profile + self.n_devices = n_devices + + # Recruitment is uniform over accounts. Weighting it towards active + # accounts was tried in the fraud pack as camouflage and measured to do + # the opposite: hubs are outliers already, so a cluster built from hubs + # is found by degree alone. Ordinary accounts plus a few extra edges + # hide better, and ``size_scale`` is the lever for cluster size. + self.eligible = np.arange(pop.n_accounts) + self.rows: dict[str, list[tuple]] = {POSTED: [], RESHARED: [], REPLIED: []} + self.shared_devices: list[tuple[int, int]] = [] + self.extra_follows: list[tuple[int, int]] = [] + self.fresh_accounts: dict[int, np.datetime64] = {} + self.stripped_accounts: set[int] = set() + self.dormant_until: dict[int, np.datetime64] = {} + + # ------------------------------------------------------------- selection + + def size(self, base: int, spread: int = 0) -> int: + """A campaign size, scaled by tradecraft and never below two.""" + drawn = base + (self.rng.integers(0, spread + 1) if spread else 0) + return self.scaled(drawn) + + def pick(self, count: int, community: int | None = None) -> np.ndarray: + """Recruit ``count`` accounts. + + Draws from one interest community when asked, which is what makes a + campaign sit inside the social graph rather than across it. Accounts + already in a campaign are avoided unless ``overlap`` says otherwise. + """ + pool = self.eligible + if community is not None: + members = np.flatnonzero(self.pop.account_community == community) + if len(members) >= count: + pool = members + if not self.allow_overlap or self.rng.random() > self.profile.overlap: + free = pool[~np.isin(pool, list(self.used))] if self.used else pool + if len(free) >= count: + pool = free + if len(pool) == 0: + return np.empty(0, dtype=np.int64) + chosen = self.rng.choice(pool, size=min(count, len(pool)), replace=False) + self.used.update(int(c) for c in chosen) + return np.asarray(chosen, dtype=np.int64) + + def a_community(self) -> int: + return int(self.rng.choice(np.unique(self.pop.account_community))) + + def a_topic(self, community: int | None = None) -> int: + """A topic to push, preferring the campaign's own community.""" + if community is not None: + candidates = self.pop.topics_by_community.get(community) + if candidates is not None and len(candidates): + return int(self.rng.choice(candidates)) + return self.rng.integers(0, self.pop.n_topics).item() + + # ------------------------------------------------------------------ time + + def window(self, playbook: str) -> tuple[np.datetime64, np.timedelta64]: + """Start and width of a campaign's activity window. + + Width is the playbook's natural span stretched by + ``timing_jitter_hours``. At ``low`` a copypasta burst is inside three + minutes; at ``high`` it is spread over days and there is no burst left + to find. + """ + span_days = self.catalog.span_days(playbook) + width_s = max(60.0, span_days * 86_400 * (self.profile.timing_jitter_hours / 1.0)) + total = float((self.period_end - self.period_start) / np.timedelta64(1, "s")) + width_s = min(width_s, max(60.0, total * 0.9)) + start_offset = self.rng.uniform(0.0, max(1.0, total - width_s)) + return ( + self.period_start + np.timedelta64(int(start_offset), "s"), + np.timedelta64(int(width_s), "s"), + ) + + def stamps(self, start: np.datetime64, width: np.timedelta64, count: int) -> np.ndarray: + if count <= 0: + return np.empty(0, dtype="datetime64[s]") + seconds = int(width / np.timedelta64(1, "s")) + offsets = self.rng.integers(0, max(1, seconds), size=count) + return start + offsets.astype("timedelta64[s]") + + # ------------------------------------------------------------------ text + + def templates(self, topic: int, count: int) -> np.ndarray: + """Template ids for a campaign's posts. + + At ``text_blend=0`` every post carries the same id, which is the + duplicate-text signature. At 1 the ids are drawn as an organic post + about that topic would draw them, so text similarity carries nothing + and only structure and timing remain. + """ + signature = self.rng.integers(0, TEMPLATES_PER_TOPIC).item() + organic = self.rng.integers(0, TEMPLATES_PER_TOPIC, size=count) + blended = self.rng.random(count) < self.profile.text_blend + return np.where(blended, organic, signature) + + # ----------------------------------------------------------- bookkeeping + + def post(self, campaign: Campaign, authors: np.ndarray, topic: int, stamps: np.ndarray) -> None: + templates = self.templates(topic, len(authors)) + for author, stamp, template in zip(authors, stamps, templates): + self.rows[POSTED].append( + (int(author), int(topic), stamp, int(template), campaign.campaign_id) + ) + campaign.n_events += len(authors) + + def interact( + self, + campaign: Campaign, + channel: str, + sources: np.ndarray, + targets: np.ndarray, + topic: int, + stamps: np.ndarray, + ) -> None: + for source, target, stamp in zip(sources, targets, stamps): + if int(source) == int(target): + continue + self.rows[channel].append( + (int(source), int(target), stamp, int(topic), campaign.campaign_id) + ) + campaign.n_events += 1 + + def age(self, campaign: Campaign, accounts: np.ndarray, start: np.datetime64) -> None: + """Decide when a campaign's accounts were created. + + At ``account_age_blend=0`` they are all created a few days before the + campaign: a bloc of same-day signups is the loudest single signal in + real platform data, and at ``low`` it is left in deliberately. At 1 + they keep the population's ages, which is what buying aged accounts + buys. + """ + for account in accounts: + if self.rng.random() < self.profile.account_age_blend: + continue + lead_days = self.rng.integers(1, 8).item() + created = start - np.timedelta64(lead_days * 86_400, "s") + if created < self.period_start: + created = self.period_start + day = created.astype("datetime64[D]") + # An account can belong to two campaigns when ``overlap`` allows + # it. Keep the earliest date, or the second campaign would move + # creation after the first campaign's events. + existing = self.fresh_accounts.get(int(account)) + self.fresh_accounts[int(account)] = min(day, existing) if existing else day + + def camouflage(self, campaign: Campaign, accounts: np.ndarray) -> None: + """Strip organic activity from the accounts tradecraft says are bare. + + At ``activity_camouflage=1`` nothing is stripped and every campaign + account also behaves normally. At 0 they do nothing but the campaign, + which makes a single-purpose account obvious by its topic breadth. + """ + for account in accounts: + if self.rng.random() >= self.profile.activity_camouflage: + self.stripped_accounts.add(int(account)) + + +# --------------------------------------------------------------- inauthentic + + +def copypasta(ctx: PlaybookContext, campaign: Campaign) -> None: + """Many accounts post the same text about one topic, near-simultaneously.""" + community = ctx.a_community() + members = ctx.pick(ctx.size(14, 10), community=community) + if len(members) < 2: + return + topic = ctx.a_topic(community) + start, width = ctx.window("copypasta") + stamps = ctx.stamps(start, width, len(members)) + for member in members: + campaign.roles[int(member)] = "poster" + campaign.topic = topic + ctx.post(campaign, members, topic, stamps) + campaign.start, campaign.end = stamps.min(), stamps.max() + ctx.age(campaign, members, start) + ctx.camouflage(campaign, members) + + +def amplification_ring(ctx: PlaybookContext, campaign: Campaign) -> None: + """A cluster repeatedly reshares one account's output.""" + community = ctx.a_community() + members = ctx.pick(ctx.size(12, 8), community=community) + if len(members) < 3: + return + target, amplifiers = int(members[0]), members[1:] + campaign.roles[target] = "target" + for member in amplifiers: + campaign.roles[int(member)] = "amplifier" + topic = ctx.a_topic(community) + campaign.topic = topic + start, width = ctx.window("amplification_ring") + # Each amplifier boosts several times: volume on one target is the shape. + per = max(2, round(4 * ctx.profile.size_scale)) + sources = np.repeat(amplifiers, per) + stamps = ctx.stamps(start, width, len(sources)) + ctx.interact(campaign, RESHARED, sources, np.full(len(sources), target), topic, stamps) + campaign.start, campaign.end = stamps.min(), stamps.max() + ctx.age(campaign, amplifiers, start) + ctx.camouflage(campaign, amplifiers) + + +def reply_brigade(ctx: PlaybookContext, campaign: Campaign) -> None: + """Coordinated accounts flood one account's replies in a short window.""" + members = ctx.pick(ctx.size(18, 12)) + if len(members) < 3: + return + target, brigaders = int(members[0]), members[1:] + campaign.roles[target] = "target" + for member in brigaders: + campaign.roles[int(member)] = "brigader" + topic = ctx.a_topic() + campaign.topic = topic + start, width = ctx.window("reply_brigade") + per = max(1, round(3 * ctx.profile.size_scale)) + sources = np.repeat(brigaders, per) + stamps = ctx.stamps(start, width, len(sources)) + ctx.interact(campaign, REPLIED, sources, np.full(len(sources), target), topic, stamps) + campaign.start, campaign.end = stamps.min(), stamps.max() + ctx.age(campaign, brigaders, start) + ctx.camouflage(campaign, brigaders) + + +def follow_farm(ctx: PlaybookContext, campaign: Campaign) -> None: + """A dense mutual-follow cluster, inflating each member's audience. + + The signature is reciprocity and density, not activity: the cluster barely + posts. That makes it the one playbook an event-only detector cannot see at + all, which is the point of including it. + """ + members = ctx.pick(ctx.size(16, 10)) + if len(members) < 3: + return + for member in members: + campaign.roles[int(member)] = "member" + # Near-complete mutual follow. Density falls with size_scale so a high + # tradecraft farm is not a clique. + density = 0.95 * ctx.profile.size_scale + 0.35 + for i, a in enumerate(members): + for b in members[i + 1 :]: + if ctx.rng.random() < min(1.0, density): + ctx.extra_follows.append((int(a), int(b))) + ctx.extra_follows.append((int(b), int(a))) + start, width = ctx.window("follow_farm") + # A little posting, so the cluster is not inert. + posters = members[: max(1, len(members) // 3)] + topic = ctx.a_topic() + campaign.topic = topic + stamps = ctx.stamps(start, width, len(posters)) + ctx.post(campaign, posters, topic, stamps) + campaign.start, campaign.end = start, start + width + ctx.age(campaign, members, start) + ctx.camouflage(campaign, members) + + +def hashtag_flood(ctx: PlaybookContext, campaign: Campaign) -> None: + """A group pushes one topic hard enough to trend. + + Unlike copypasta the text varies; it is volume on a single topic in a + narrow window that is the signal, which is exactly what a breaking-news + decoy also looks like. + """ + community = ctx.a_community() + members = ctx.pick(ctx.size(20, 14), community=community) + if len(members) < 3: + return + for member in members: + campaign.roles[int(member)] = "poster" + topic = ctx.a_topic(community) + campaign.topic = topic + start, width = ctx.window("hashtag_flood") + per = max(2, round(5 * ctx.profile.size_scale)) + authors = np.repeat(members, per) + stamps = ctx.stamps(start, width, len(authors)) + ctx.post(campaign, authors, topic, stamps) + campaign.start, campaign.end = stamps.min(), stamps.max() + ctx.age(campaign, members, start) + ctx.camouflage(campaign, members) + + +def sockpuppet_cluster(ctx: PlaybookContext, campaign: Campaign) -> None: + """Several accounts run by one operator, sharing a device. + + The device fingerprint is the signature, and it is the hardest one to + blend: the accounts have to post from somewhere. Organic device sharing + (households, one person with two accounts) is what it hides in. + """ + members = ctx.pick(ctx.size(6, 4)) + if len(members) < 2 or ctx.n_devices == 0: + return + device = ctx.rng.integers(0, ctx.n_devices).item() + for member in members: + campaign.roles[int(member)] = "puppet" + ctx.shared_devices.append((int(member), device)) + start, width = ctx.window("sockpuppet_cluster") + # Puppets post across a few topics, staggered: a single-topic cluster is + # easier to find by topic concentration alone. + topics = [ctx.a_topic() for _ in range(max(1, len(members) // 2))] + campaign.topic = topics[0] + per = max(2, round(6 * ctx.profile.size_scale)) + stamps_all = [] + for topic in topics: + authors = ctx.rng.choice(members, size=per, replace=True) + stamps = ctx.stamps(start, width, per) + ctx.post(campaign, authors, topic, stamps) + stamps_all.append(stamps) + joined = np.concatenate(stamps_all) + campaign.start, campaign.end = joined.min(), joined.max() + ctx.age(campaign, members, start) + ctx.camouflage(campaign, members) + + +def astroturf_campaign(ctx: PlaybookContext, campaign: Campaign) -> None: + """A sustained push on one topic over weeks, posting and resharing. + + Long and low: no burst, no duplicate text, members spread across + communities. It is the playbook designed to be missed, and at high + tradecraft it should be. + """ + members = ctx.pick(ctx.size(24, 16)) + if len(members) < 4: + return + topic = ctx.a_topic() + campaign.topic = topic + start, width = ctx.window("astroturf_campaign") + split = max(1, len(members) // 2) + posters, amplifiers = members[:split], members[split:] + for member in posters: + campaign.roles[int(member)] = "poster" + for member in amplifiers: + campaign.roles[int(member)] = "amplifier" + + per = max(2, round(4 * ctx.profile.size_scale)) + authors = np.repeat(posters, per) + post_stamps = ctx.stamps(start, width, len(authors)) + ctx.post(campaign, authors, topic, post_stamps) + + if len(amplifiers): + sources = np.repeat(amplifiers, per) + targets = ctx.rng.choice(posters, size=len(sources), replace=True) + share_stamps = ctx.stamps(start, width, len(sources)) + ctx.interact(campaign, RESHARED, sources, targets, topic, share_stamps) + joined = np.concatenate([post_stamps, share_stamps]) + else: + joined = post_stamps + campaign.start, campaign.end = joined.min(), joined.max() + ctx.age(campaign, members, start) + ctx.camouflage(campaign, members) + + +def account_handover(ctx: PlaybookContext, campaign: Campaign) -> None: + """Aged, dormant accounts activated together. + + Ages are left alone whatever ``account_age_blend`` says: the accounts being + old is the whole premise. The signal is the simultaneous end of dormancy, + so their organic activity before the activation date is stripped. + """ + members = ctx.pick(ctx.size(10, 8)) + if len(members) < 2: + return + for member in members: + campaign.roles[int(member)] = "revived" + topic = ctx.a_topic() + campaign.topic = topic + start, width = ctx.window("account_handover") + per = max(2, round(4 * ctx.profile.size_scale)) + authors = np.repeat(members, per) + stamps = ctx.stamps(start, width, len(authors)) + ctx.post(campaign, authors, topic, stamps) + campaign.start, campaign.end = stamps.min(), stamps.max() + # Dormant until activation, then ordinary. Removing *all* of an account's + # organic activity instead left its counts at exactly zero, which separated + # the playbook at AUC 0.97 on reshare count at every tradecraft level: an + # artifact of the implementation, not a property of account handover. Real + # revived accounts have a quiet history and a normal present. + for member in members: + ctx.dormant_until[int(member)] = campaign.start + + +# ---------------------------------------------------------------- organic + + +def fandom_burst(ctx: PlaybookContext, campaign: Campaign) -> None: + """A real community posting hard about one topic. Twin of hashtag_flood. + + Same shape: one community, one topic, one window, sustained volume. The + differences a detector has to find are that the text varies as organic text + does, the accounts are aged normally, and they keep their ordinary activity. + Flagging this is a false positive. + """ + community = ctx.a_community() + members = ctx.pick(ctx.size(16, 12), community=community) + if len(members) < 2: + return + for member in members: + campaign.roles[int(member)] = "fan" + topic = ctx.a_topic(community) + campaign.topic = topic + start, width = ctx.window("fandom_burst") + # Matched to hashtag_flood, its twin: a decoy separable by how much it + # posts is not testing anything a detector should have to get right. + per = max(2, round(5 * ctx.profile.size_scale)) + authors = np.repeat(members, per) + stamps = ctx.stamps(start, width, len(authors)) + # Organic text spread whatever the tradecraft level: this is not tradecraft, + # it is how people actually write. + for author, stamp in zip(authors, stamps): + ctx.rows[POSTED].append( + ( + int(author), + int(topic), + stamp, + ctx.rng.integers(0, TEMPLATES_PER_TOPIC).item(), + campaign.campaign_id, + ) + ) + campaign.n_events += len(authors) + campaign.start, campaign.end = stamps.min(), stamps.max() + # A fandom follows each other, which is why it clusters. + for i, a in enumerate(members): + for b in members[i + 1 :]: + if ctx.rng.random() < 0.3: + ctx.extra_follows.append((int(a), int(b))) + + +def breaking_news(ctx: PlaybookContext, campaign: Campaign) -> None: + """Everyone posts about one event inside an hour. Twin of copypasta. + + One post each, in a tight window, about a single topic: structurally the + same as a copypasta ring. Drawn across communities rather than within one, + which is the only structural difference, and a detector keying on a + synchronous single-topic burst will not see it. + """ + members = ctx.pick(ctx.size(26, 18)) + if len(members) < 3: + return + for member in members: + campaign.roles[int(member)] = "witness" + topic = ctx.a_topic() + campaign.topic = topic + start, width = ctx.window("breaking_news") + # One post each, matching copypasta. + per = 1 + authors = np.repeat(members, per) + stamps = ctx.stamps(start, width, len(authors)) + for author, stamp in zip(authors, stamps): + ctx.rows[POSTED].append( + ( + int(author), + int(topic), + stamp, + ctx.rng.integers(0, TEMPLATES_PER_TOPIC).item(), + campaign.campaign_id, + ) + ) + campaign.n_events += len(authors) + campaign.start, campaign.end = stamps.min(), stamps.max() + + +def mutual_follow_community(ctx: PlaybookContext, campaign: Campaign) -> None: + """A tight-knit real community. Twin of follow_farm. + + Dense and reciprocal, because that is what a close community is. A + reciprocity-and-density rule cannot separate this from a farm; the + difference is that these accounts talk to each other rather than only + following. + """ + community = ctx.a_community() + members = ctx.pick(ctx.size(14, 10), community=community) + if len(members) < 3: + return + for member in members: + campaign.roles[int(member)] = "member" + density = 0.7 * ctx.profile.size_scale + 0.3 + for i, a in enumerate(members): + for b in members[i + 1 :]: + if ctx.rng.random() < min(1.0, density): + ctx.extra_follows.append((int(a), int(b))) + ctx.extra_follows.append((int(b), int(a))) + start, width = ctx.window("mutual_follow_community") + topic = ctx.a_topic(community) + campaign.topic = topic + # They interact, which a farm does not. + per = max(1, round(3 * ctx.profile.size_scale)) + sources = np.repeat(members, per) + targets = ctx.rng.choice(members, size=len(sources), replace=True) + stamps = ctx.stamps(start, width, len(sources)) + ctx.interact(campaign, REPLIED, sources, targets, topic, stamps) + campaign.start, campaign.end = start, start + width + + +PATTERN_FUNCTIONS = { + "copypasta": copypasta, + "amplification_ring": amplification_ring, + "reply_brigade": reply_brigade, + "follow_farm": follow_farm, + "hashtag_flood": hashtag_flood, + "sockpuppet_cluster": sockpuppet_cluster, + "astroturf_campaign": astroturf_campaign, + "account_handover": account_handover, + "fandom_burst": fandom_burst, + "breaking_news": breaking_news, + "mutual_follow_community": mutual_follow_community, +} + +POST_COLUMNS = ["source", "target", "timestamp", "template_id", "campaign_id"] +INTERACTION_COLUMNS = ["source", "target", "timestamp", "topic", "campaign_id"] + + +def inject( + rng: np.random.Generator, + pop: Population, + config: CoordinationConfig, + n_devices: int, +) -> Injection: + """Run every campaign the config asks for, then the organic decoys.""" + ctx = PlaybookContext(rng, pop, config, n_devices) + + def make(playbook: str, index: int, coordinated: bool) -> Campaign: + return Campaign(pattern_id=f"{playbook}_{index}", name=playbook, labelled=coordinated) + + # Two passes, because the decoy budget is a fraction of the campaigns that + # actually recruited somebody rather than of the ones that were asked for. + campaigns = run_catalog(ctx, config.campaign_counts, PATTERN_FUNCTIONS, make, drop_empty=True) + decoys = grouped_decoys(CATALOG, round(len(campaigns) * config.profile.decoy_ratio)) + campaigns += run_catalog(ctx, {}, PATTERN_FUNCTIONS, make, decoys, drop_empty=True) + + events: dict[str, pl.DataFrame] = {} + for channel, rows in ctx.rows.items(): + if not rows: + continue + if channel == POSTED: + sources, targets, stamps, templates, ids = zip(*rows) + events[channel] = pl.DataFrame( + { + "source": pop.account_ids.gather(list(sources)), + "target": pop.topic_ids.gather(list(targets)), + "timestamp": np.array(stamps, dtype="datetime64[us]"), + "template_id": np.array(templates, dtype=np.int64), + "campaign_id": list(ids), + } + ) + else: + sources, targets, stamps, topics, ids = zip(*rows) + events[channel] = pl.DataFrame( + { + "source": pop.account_ids.gather(list(sources)), + "target": pop.account_ids.gather(list(targets)), + "timestamp": np.array(stamps, dtype="datetime64[us]"), + "topic": pop.topic_ids.gather(list(topics)), + "campaign_id": list(ids), + } + ) + + return Injection( + campaigns=campaigns, + events=events, + shared_devices=ctx.shared_devices, + extra_follows=ctx.extra_follows, + fresh_accounts=ctx.fresh_accounts, + stripped_accounts=ctx.stripped_accounts, + dormant_until=ctx.dormant_until, + ) diff --git a/graphfaker/domains/coordination/process.py b/graphfaker/domains/coordination/process.py new file mode 100644 index 0000000..fb1bede --- /dev/null +++ b/graphfaker/domains/coordination/process.py @@ -0,0 +1,347 @@ +"""The organic activity process. + +Everything here is vectorised with numpy, so the event stream is a matter of +memory rather than of Python loops. + +Realism comes from four things a uniform random activity model leaves out, and +each of them is a signal a coordination detector would otherwise get for free: + +* **Heavy-tailed volume.** Account activity is log-normal, so most accounts + post a handful of times over the period and a few thousand post constantly. + If volume were uniform, any campaign account would be a volume outlier and + the detection problem would be trivial. +* **Topic affinity.** An account posts mostly about topics in its own interest + community, weighted by topic prominence. A campaign pushing one topic is + only anomalous against a population that *also* concentrates, and the + organic concentration is what makes topic focus a weak signal rather than a + decisive one. +* **Reshares follow the follow graph.** An account reshares and replies to + accounts it follows, in proportion to their audience. This gives the + interaction graph its own hubs and clustering rather than an + Erdős–Rényi soup, and it means an amplification ring has to beat real + amplification. +* **Time of day and day of week.** Human posting has a diurnal rhythm. + Sub-minute synchrony is detectable partly because *nothing* organic is that + synchronous, so the organic rhythm has to be present for that contrast to be + real. + +Channels are the three event kinds: ``POSTED`` (account to topic), ``RESHARED`` +and ``REPLIED`` (account to account). +""" + +from __future__ import annotations + +import datetime as dt + +import numpy as np +import polars as pl + +from graphfaker.backends.tables import ID +from graphfaker.domains.coordination.config import CoordinationConfig +from graphfaker.domains.coordination.entities import COMMUNITY + +POSTED = "POSTED" +RESHARED = "RESHARED" +REPLIED = "REPLIED" +CHANNELS = (POSTED, RESHARED, REPLIED) + +#: Share of events of each kind. Reading and resharing is cheaper than +#: authoring, so reshares outnumber original posts on every real platform. +CHANNEL_MIX = {POSTED: 0.34, RESHARED: 0.48, REPLIED: 0.18} + +#: Relative posting rate by hour, local. Two humps: a commute-and-lunch rise +#: and an evening peak, with a trough overnight. +HOUR_PROFILE = np.array( + [ + 0.25, 0.15, 0.10, 0.08, 0.08, 0.12, 0.30, 0.65, + 0.95, 1.05, 1.05, 1.00, 1.10, 1.05, 1.00, 1.00, + 1.05, 1.20, 1.40, 1.55, 1.60, 1.40, 0.95, 0.55, + ], + dtype=np.float64, +) +#: Monday to Sunday. Weekends are busier for everything except work topics, +#: which this model does not distinguish. +DAY_PROFILE = np.array([1.0, 1.0, 1.0, 1.0, 1.05, 1.15, 1.10], dtype=np.float64) + +#: Distinct ways an organic post about a topic can be phrased. A campaign at +#: ``text_blend=0`` collapses to one of them, which is the duplicate-text +#: signal; the organic spread is what that collapse is measured against. +#: +#: Large on purpose. At 400 nearly every organic template was also used by +#: another account, so ``template_reuse`` sat near 1.0 for everyone and a +#: duplicate-text rule flagged 90% of *organic* accounts — the canonical +#: copypasta signal reduced to noise by a generator artifact. The number of +#: ways to phrase a real post is effectively unbounded, so collisions between +#: unrelated accounts should be rare. +TEMPLATES_PER_TOPIC = 20_000 + +#: Rows built per block. Keeps peak memory proportional to the block rather +#: than to the whole stream. +BLOCK_ROWS = 2_000_000 + +#: Share of accounts that read without posting. Every measured platform is +#: mostly lurkers, and leaving them out had two consequences: only 1.5% of +#: accounts were silent (against 20-40% reported for real networks), and "low +#: activity" stopped being informative, which flatters any detector using +#: volume as a feature. +LURKER_SHARE = 0.30 + + +class Population: + """Numpy views of the node tables, plus the derived weights the process + needs. Built once; every channel reads from it.""" + + def __init__(self, tables: dict[str, pl.DataFrame], config: CoordinationConfig): + accounts, topics = tables["Account"], tables["Topic"] + self.config = config + self.n_accounts = accounts.height + self.n_topics = topics.height + + self.account_ids = accounts[ID] + self.topic_ids = topics[ID] + self.account_community = accounts[COMMUNITY].to_numpy() + self.topic_community = np.asarray( + topics[COMMUNITY].to_numpy() if COMMUNITY in topics.columns else np.zeros(self.n_topics) + ) + self.activity = accounts["activity"].to_numpy().astype(np.float64) + self.reshare_rate = accounts["reshare_rate"].to_numpy().astype(np.float64) + self.topic_prominence = topics["prominence"].to_numpy().astype(np.float64) + self.created_at = accounts["created_at"].to_numpy().astype("datetime64[D]") + + # Who authors an event: activity, normalised. An account created + # part-way through the period is weighted down proportionally so it is + # not busier than its lifetime allows. + self.period_start = np.datetime64(config.period_start, "s") + self.period_days = config.period_days + self.period_end = self.period_start + np.timedelta64(config.period_days * 86_400, "s") + + self.author_weight = self.activity.copy() + self.author_weight[self.author_weight < 0] = 0.0 + if self.author_weight.sum() <= 0: + self.author_weight = np.ones(self.n_accounts) + # Lurkers are chosen deterministically from the account's position so + # the set does not depend on how many events are drawn. + lurkers = (np.arange(self.n_accounts) % 100) < int(LURKER_SHARE * 100) + self.author_weight = np.where(lurkers, 0.0, self.author_weight) + if self.author_weight.sum() <= 0: # tiny graphs: everyone posts + self.author_weight = np.ones(self.n_accounts) + self.lurkers = lurkers + + # Topic choice per community: prominence, restricted to the topics of + # that community where it has any. + self.topics_by_community: dict[int, np.ndarray] = {} + for group in np.unique(self.account_community): + members = np.flatnonzero(self.topic_community == group) + self.topics_by_community[int(group)] = ( + members if len(members) else np.arange(self.n_topics) + ) + + def assign_topic_communities(self, rng: np.random.Generator) -> np.ndarray: + """Topics belong to an interest community. + + The schema has no latent reference for Topic, so the assignment is made + here and written back onto the table; it is what makes an account's + topic choice community-local. + """ + groups = np.unique(self.account_community) + self.topic_community = rng.choice(groups, size=self.n_topics) + self.topics_by_community = { + int(g): ( + np.flatnonzero(self.topic_community == g) + if int((self.topic_community == g).sum()) + else np.arange(self.n_topics) + ) + for g in groups + } + return self.topic_community + + +def _timestamps(rng: np.random.Generator, pop: Population, size: int) -> np.ndarray: + """Draw event times over the period, following the hour and day profiles.""" + if size == 0: + return np.empty(0, dtype="datetime64[s]") + day = rng.integers(0, pop.period_days, size=size) + weekday = (np.datetime64(pop.config.period_start).astype("datetime64[D]").astype(int) + day) % 7 + keep_day = rng.random(size) < (DAY_PROFILE[weekday] / DAY_PROFILE.max()) + # Rejection on the day profile is cheap and keeps the marginal exact; + # rejected draws simply fall on a different day. + day = np.where(keep_day, day, rng.integers(0, pop.period_days, size=size)) + + hour = rng.choice(24, size=size, p=HOUR_PROFILE / HOUR_PROFILE.sum()) + minute = rng.integers(0, 60, size=size) + second = rng.integers(0, 60, size=size) + offsets = day.astype(np.int64) * 86_400 + hour.astype(np.int64) * 3_600 + minute * 60 + second + return pop.period_start + offsets.astype("timedelta64[s]") + + +def _authors(rng: np.random.Generator, pop: Population, size: int) -> np.ndarray: + weight = pop.author_weight + return rng.choice(pop.n_accounts, size=size, p=weight / weight.sum()) + + +def _topics_for(rng: np.random.Generator, pop: Population, authors: np.ndarray) -> np.ndarray: + """A topic per event, drawn community-locally and weighted by prominence.""" + out = np.empty(len(authors), dtype=np.int64) + author_group = pop.account_community[authors] + for group, candidates in pop.topics_by_community.items(): + mask = author_group == group + count = int(mask.sum()) + if not count: + continue + w = pop.topic_prominence[candidates] + total = w.sum() + out[mask] = ( + rng.choice(candidates, size=count, p=w / total) + if total > 0 + else rng.choice(candidates, size=count) + ) + return out + + +def build_audience(pop: Population, follows: pl.DataFrame) -> dict[int, np.ndarray]: + """Who each account can reshare or reply to: the accounts it follows. + + Indexed by account position. An account following nobody reshares from the + population at large, which is what a logged-out-style feed does. + """ + if follows.height == 0: + return {} + index = {str(a): i for i, a in enumerate(pop.account_ids.to_list())} + src = np.array([index.get(s, -1) for s in follows["source"].to_list()], dtype=np.int64) + dst = np.array([index.get(t, -1) for t in follows["target"].to_list()], dtype=np.int64) + keep = (src >= 0) & (dst >= 0) + src, dst = src[keep], dst[keep] + order = np.argsort(src, kind="stable") + src, dst = src[order], dst[order] + bounds = np.searchsorted(src, np.arange(pop.n_accounts + 1)) + return { + i: dst[bounds[i] : bounds[i + 1]] + for i in range(pop.n_accounts) + if bounds[i + 1] > bounds[i] + } + + +def _interaction_targets( + rng: np.random.Generator, + pop: Population, + authors: np.ndarray, + audience: dict[int, np.ndarray], +) -> np.ndarray: + """Pick who each event is directed at. + + Mostly someone the author follows, which is what puts the hubs of the + follow graph at the centre of the interaction graph too. Falling back to + the population keeps accounts that follow nobody from being inert. + """ + size = len(authors) + out = rng.choice( + pop.n_accounts, size=size, p=pop.author_weight / pop.author_weight.sum() + ) + for position, author in enumerate(authors): + pool = audience.get(int(author)) + if pool is not None and len(pool): + out[position] = pool[rng.integers(0, len(pool))] + return out + + +def organic_events( + rng: np.random.Generator, + pop: Population, + budget: int, + audience: dict[int, np.ndarray], + finish=None, +) -> dict[str, list[pl.DataFrame]]: + """The organic event stream, as a list of parts per channel. + + Parts rather than one frame per channel: the caller concatenates without + holding a second copy, which is what keeps peak memory near the block size. + + ``finish`` is applied to each part as it is produced, so events that a + campaign has retrospectively made impossible (an account created after the + event, or one whose organic activity is stripped) never take up memory. + """ + parts: dict[str, list[pl.DataFrame]] = {channel: [] for channel in CHANNELS} + if budget <= 0 or pop.n_accounts == 0: + return parts + + for channel in CHANNELS: + remaining = int(budget * CHANNEL_MIX[channel]) + while remaining > 0: + take = min(BLOCK_ROWS, remaining) + remaining -= take + + authors = _authors(rng, pop, take) + stamps = _timestamps(rng, pop, take) + + if channel == POSTED: + topics = _topics_for(rng, pop, authors) + frame = pl.DataFrame( + { + "source": pop.account_ids.gather(authors), + "target": pop.topic_ids.gather(topics), + "timestamp": stamps.astype("datetime64[us]"), + "template_id": rng.integers(0, TEMPLATES_PER_TOPIC, size=take), + "campaign_id": pl.Series([None] * take, dtype=pl.String), + } + ) + else: + targets = _interaction_targets(rng, pop, authors, audience) + topics = _topics_for(rng, pop, authors) + frame = pl.DataFrame( + { + "source": pop.account_ids.gather(authors), + "target": pop.account_ids.gather(targets), + "timestamp": stamps.astype("datetime64[us]"), + "topic": pop.topic_ids.gather(topics), + "campaign_id": pl.Series([None] * take, dtype=pl.String), + } + ) + + if finish is not None: + frame = finish(frame) + if frame.height: + parts[channel].append(frame) + return parts + + +def drop_before_creation(frame: pl.DataFrame, created: pl.DataFrame) -> pl.DataFrame: + """Remove events authored by an account before it existed. + + A campaign that opens fresh accounts moves their ``created_at`` forward, + and the organic activity drawn for them beforehand has to go: an account + posting a month before signup is the kind of tell that makes a dataset + unusable, and it would be the strongest feature in the whole graph. + """ + if frame.height == 0 or created.height == 0: + return frame + return ( + frame.join(created.rename({"account": "source"}), on="source", how="left") + .filter(pl.col("created").is_null() | (pl.col("timestamp").dt.date() >= pl.col("created"))) + .drop("created") + ) + + +def drop_while_dormant(frame: pl.DataFrame, dormant: pl.DataFrame) -> pl.DataFrame: + """Remove an account's organic activity from before it woke up. + + Distinct from :func:`drop_before_creation`: the account existed and simply + was not being used. Keeping the activity after the wake-up moment is what + makes a revived account look ordinary once it is running, instead of + carrying an all-zero activity profile that gives the playbook away. + """ + if frame.height == 0 or dormant.height == 0: + return frame + return ( + frame.join(dormant.rename({"account": "source"}), on="source", how="left") + .filter(pl.col("awake").is_null() | (pl.col("timestamp") >= pl.col("awake"))) + .drop("awake") + ) + + +def period_bounds(config: CoordinationConfig) -> tuple[np.datetime64, np.datetime64]: + start = np.datetime64(config.period_start, "s") + return start, start + np.timedelta64(config.period_days * 86_400, "s") + + +def to_date(value: np.datetime64) -> dt.date: + return value.astype("datetime64[D]").astype(dt.date) diff --git a/graphfaker/domains/fraud/config.py b/graphfaker/domains/fraud/config.py index 50a5161..89fae5f 100644 --- a/graphfaker/domains/fraud/config.py +++ b/graphfaker/domains/fraud/config.py @@ -17,6 +17,8 @@ from pydantic import BaseModel, ConfigDict, Field, model_validator +from graphfaker.schema.patterns import Camouflage, PatternCatalog, PatternSpec + Hardness = Literal["low", "medium", "high"] #: Base sizes at scale 1.0. @@ -24,38 +26,45 @@ BASE_TRANSACTIONS = 90_000_000 BASE_PATTERNS = 1_000 -#: Every typology the pack can inject, in the order patterns are allocated. -TYPOLOGIES = ( - "fan_in", - "fan_out", - "gather_scatter", - "scatter_gather", - "cycle", - "stack", - "bipartite", - "structuring", - "mule_network", - "bust_out", - "synthetic_identity", +#: What the pack injects: every typology with its share of the budget and its +#: natural span in days before ``timing_spread_days`` stretches it, then the +#: three decoys, each naming the typology it imitates. The order is the order +#: patterns are allocated and injected, and it is part of what a seed +#: reproduces. +#: +#: The catalogue is the AMLworld set (fan-in, fan-out, gather-scatter, +#: scatter-gather, cycle, stack, bipartite) plus the behaviours banks file +#: suspicious activity reports on. +CATALOG = PatternCatalog( + base=BASE_PATTERNS, + floor=2, + patterns=[ + PatternSpec(name="fan_in", share=0.14, span_days=2.0), + PatternSpec(name="fan_out", share=0.12, span_days=1.0), + PatternSpec(name="gather_scatter", share=0.10, span_days=3.0), + PatternSpec(name="scatter_gather", share=0.10, span_days=3.0), + PatternSpec(name="cycle", share=0.10, span_days=2.0), + PatternSpec(name="stack", share=0.08, span_days=2.0), + PatternSpec(name="bipartite", share=0.06, span_days=3.0), + PatternSpec(name="structuring", share=0.10, span_days=10.0), + PatternSpec(name="mule_network", share=0.08, span_days=1.0), + PatternSpec(name="bust_out", share=0.06, span_days=60.0), + PatternSpec(name="synthetic_identity", share=0.06, span_days=20.0), + # Legitimate structures with the same shape, labelled as such. Money + # does move in circles between honest people, and groups of friends + # do all pay one person. + PatternSpec(name="decoy_fan_in", imitates="fan_in", span_days=2.0), + PatternSpec(name="decoy_fan_out", imitates="fan_out", span_days=1.0), + PatternSpec(name="decoy_cycle", imitates="cycle", span_days=2.0), + ], ) -#: Relative frequency of each typology when counts are derived from scale. -TYPOLOGY_MIX = { - "fan_in": 0.14, - "fan_out": 0.12, - "gather_scatter": 0.10, - "scatter_gather": 0.10, - "cycle": 0.10, - "stack": 0.08, - "bipartite": 0.06, - "structuring": 0.10, - "mule_network": 0.08, - "bust_out": 0.06, - "synthetic_identity": 0.06, -} +#: The injectable typologies, in allocation order. Kept as a name because it +#: reads better than ``CATALOG.injected`` at the call sites that loop over it. +TYPOLOGIES = CATALOG.injected -class HardnessProfile(BaseModel): +class HardnessProfile(Camouflage): """What a hardness level does to injected patterns. ``amount_blend``: 0 keeps a typology's signature amounts (round, near a @@ -85,37 +94,43 @@ class HardnessProfile(BaseModel): under the radar, and how ``high`` keeps degree from being a giveaway. """ - model_config = ConfigDict(extra="forbid", frozen=True) + #: The dials are :class:`~graphfaker.schema.patterns.Camouflage`; these + #: names are what a fraud reader calls them. ``amount_blend`` is the + #: signature blend, because in a bank the signature is the amount. + @property + def amount_blend(self) -> float: + return self.signature_blend - amount_blend: float = Field(ge=0.0, le=1.0) - timing_spread_days: float = Field(gt=0.0) - ring_overlap: float = Field(ge=0.0, le=1.0) - decoy_ratio: float = Field(ge=0.0) - activity_camouflage: float = Field(ge=0.0, le=1.0) - size_scale: float = Field(default=1.0, gt=0.0, le=1.0) + @property + def timing_spread_days(self) -> float: + return self.timing_spread + + @property + def ring_overlap(self) -> float: + return self.overlap HARDNESS_PROFILES: dict[str, HardnessProfile] = { "low": HardnessProfile( - amount_blend=0.0, - timing_spread_days=0.1, - ring_overlap=0.0, + signature_blend=0.0, + timing_spread=0.1, + overlap=0.0, decoy_ratio=0.0, activity_camouflage=0.0, size_scale=1.0, ), "medium": HardnessProfile( - amount_blend=0.5, - timing_spread_days=3.0, - ring_overlap=0.2, + signature_blend=0.5, + timing_spread=3.0, + overlap=0.2, decoy_ratio=0.5, activity_camouflage=0.6, size_scale=0.75, ), "high": HardnessProfile( - amount_blend=0.9, - timing_spread_days=14.0, - ring_overlap=0.4, + signature_blend=0.9, + timing_spread=14.0, + overlap=0.4, decoy_ratio=1.0, activity_camouflage=1.0, size_scale=0.5, @@ -123,22 +138,6 @@ class HardnessProfile(BaseModel): } -def _largest_remainder(total: int, mix: dict[str, float], floor: int) -> dict[str, int]: - """Allocate ``total`` across ``mix`` proportionally, every key at least - ``floor``, remainders to the largest fractional parts.""" - counts = dict.fromkeys(mix, floor) - remaining = total - floor * len(mix) - if remaining <= 0: - return counts - exact = {name: remaining * share for name, share in mix.items()} - for name, value in exact.items(): - counts[name] += int(value) - leftover = remaining - sum(int(v) for v in exact.values()) - for name in sorted(exact, key=lambda n: exact[n] - int(exact[n]), reverse=True)[:leftover]: - counts[name] += 1 - return counts - - class FraudConfig(BaseModel): model_config = ConfigDict(extra="forbid", frozen=True) @@ -153,7 +152,7 @@ class FraudConfig(BaseModel): #: Reporting threshold structuring stays under (USD CTR threshold). reporting_threshold: float = 10_000.0 #: Override the number of patterns per typology. ``None`` derives them - #: from scale with :data:`TYPOLOGY_MIX`. + #: from scale with the shares in :data:`CATALOG`. patterns: dict[str, int] | None = None #: Latent regions; attributes and partner choice are conditioned on them. regions: int = Field(default=8, ge=1) @@ -195,13 +194,13 @@ def num_patterns(self) -> int: return sum(self.patterns.values()) # At least two of every typology, so small datasets still cover the # catalog; gen-fraud-graph's floor of 10 would leave most at zero. - return max(2 * len(TYPOLOGIES), int(BASE_PATTERNS * self.scale)) + return CATALOG.total(self.scale) @property def pattern_counts(self) -> dict[str, int]: if self.patterns is not None: return {name: self.patterns.get(name, 0) for name in TYPOLOGIES} - return _largest_remainder(self.num_patterns, TYPOLOGY_MIX, floor=2) + return CATALOG.counts(self.num_patterns) @property def profile(self) -> HardnessProfile: diff --git a/graphfaker/domains/fraud/typologies.py b/graphfaker/domains/fraud/typologies.py index b3f0585..a75a857 100644 --- a/graphfaker/domains/fraud/typologies.py +++ b/graphfaker/domains/fraud/typologies.py @@ -21,13 +21,14 @@ from __future__ import annotations -from dataclasses import dataclass, field +from dataclasses import dataclass +from itertools import count from typing import Any import numpy as np import polars as pl -from graphfaker.domains.fraud.config import TYPOLOGIES, FraudConfig, HardnessProfile +from graphfaker.domains.fraud.config import CATALOG, FraudConfig, HardnessProfile from graphfaker.domains.fraud.process import ( BUSINESS_HOURS, HOUR_PROFILE, @@ -40,34 +41,38 @@ merchant_amounts, transfer_amounts, ) +from graphfaker.engine.injection import ( + InjectionContext, + round_robin_decoys, + run_catalog, + stripped_members, +) +from graphfaker.engine.injection import Pattern as BasePattern -#: Natural span, in days at ``timing_spread_days = 1``, of each typology. -NATURAL_SPAN = { - "fan_in": 2.0, - "fan_out": 1.0, - "gather_scatter": 3.0, - "scatter_gather": 3.0, - "cycle": 2.0, - "stack": 2.0, - "bipartite": 3.0, - "structuring": 10.0, - "mule_network": 1.0, - "bust_out": 60.0, - "synthetic_identity": 20.0, -} -#: Shapes that have a legitimate twin. -DECOY_TYPOLOGIES = ("fan_in", "fan_out", "cycle") +#: Shapes that have a legitimate twin, named by the decoy that imitates each. +DECOY_TYPOLOGIES = tuple(CATALOG.twins[name] for name in CATALOG.decoys) @dataclass -class Pattern: - pattern_id: str - typology: str - is_fraud: bool - roles: dict[int, str] = field(default_factory=dict) # account idx -> role - start: np.datetime64 | None = None - end: np.datetime64 | None = None - n_transactions: int = 0 +class Pattern(BasePattern): + """An injected laundering structure. + + The record is :class:`graphfaker.engine.injection.Pattern`; these three + names are what a fraud reader calls its fields, and what the truth tables + and every metric in the pack already use. + """ + + @property + def typology(self) -> str: + return self.name + + @property + def is_fraud(self) -> bool: + return self.labelled + + @property + def n_transactions(self) -> int: + return self.events @dataclass @@ -86,7 +91,15 @@ class Injection: stripped_accounts: set[int] -class TypologyContext: +class TypologyContext(InjectionContext): + """The bank's own drawing on top of the shared pattern bookkeeping. + + What is inherited: the dials, the period, who is already in a pattern, + and whether overlap is allowed. What is here: recruitment from eligible + accounts, amounts drawn against the legitimate distribution, and the + transaction rows themselves, because those are the bank. + """ + def __init__( self, rng: np.random.Generator, @@ -96,18 +109,13 @@ def __init__( n_customers: int, n_devices: int, ): - self.rng = rng + super().__init__(rng, config.profile, CATALOG, pop.period_start, pop.period_days) self.pop = pop self.config = config self.profile: HardnessProfile = config.profile self.merchants = merchants self.n_customers = n_customers self.n_devices = n_devices - self.used: set[int] = set() - #: While True, ``pick`` may reuse pattern accounts (ring overlap). - #: Decoys turn it off: a legitimate payroll must not share members - #: with a mule ring, or its label would be ambiguous. - self.allow_overlap = True usable = np.isin(pop.account_type, ["checking", "business", "savings"]) & (pop.account_status != "closed") self.eligible = np.flatnonzero(usable) # Recruitment is uniform over eligible accounts. Weighting it towards @@ -127,7 +135,6 @@ def __init__( self.shared_devices: list[tuple[int, int]] = [] self.customer_overrides: dict[int, dict[str, Any]] = {} self.fresh_accounts: dict[int, np.datetime64] = {} - self.period_end = pop.period_start + np.timedelta64(pop.period_days * 86_400, "s") # ---------------------------------------------------------------- picks @@ -172,14 +179,13 @@ def pick(self, k: int, pool: np.ndarray | None = None) -> np.ndarray: chosen.append(candidate) if len(pool) <= k and len(chosen) == len(set(pool.tolist())): break # tiny populations: accept what there is - self.used.update(chosen) + self.claim(chosen) return np.array(chosen, dtype=np.int64) def size(self, lo: int, hi: int, floor: int = 3) -> int: """A pattern size in ``[lo, hi]`` scaled down by hardness, never below ``floor``.""" - scale = self.profile.size_scale - lo, hi = max(floor, round(lo * scale)), max(floor + 1, round(hi * scale)) + lo, hi = self.scaled(lo, floor), self.scaled(hi, floor + 1) return int(self.rng.integers(lo, hi + 1)) def bystanders(self, k: int) -> np.ndarray: @@ -196,7 +202,7 @@ def bystanders(self, k: int) -> np.ndarray: # --------------------------------------------------------------- timing def span_seconds(self, typology: str) -> int: - days = NATURAL_SPAN[typology] * self.profile.timing_spread_days + days = self.catalog.span_days(typology) * self.profile.timing_spread_days days = min(days, self.pop.period_days - 1) return max(3_600, int(days * 86_400)) @@ -248,9 +254,7 @@ def tx(self, channel: str, src: int, dst: Any, amount: float, ts: np.datetime64, else: target = self.pop.account_ids[dst] self.rows[channel].append((source, target, round(float(amount), 2), ts, memo, pattern.pattern_id)) - pattern.n_transactions += 1 - pattern.start = ts if pattern.start is None or ts < pattern.start else pattern.start - pattern.end = ts if pattern.end is None or ts > pattern.end else pattern.end + pattern.touch(ts) def frames(self) -> dict[str, pl.DataFrame]: out = {} @@ -527,7 +531,7 @@ def decoy_cycle(ctx: TypologyContext, pattern: Pattern) -> None: ctx.tx(TRANSFERS, src, dst, ctx.legit_amount(TRANSFERS, src) * 8, ts, pattern, "invoice") -TYPOLOGY_FUNCTIONS = { +PATTERN_FUNCTIONS = { "fan_in": fan_in, "fan_out": fan_out, "gather_scatter": gather_scatter, @@ -540,7 +544,9 @@ def decoy_cycle(ctx: TypologyContext, pattern: Pattern) -> None: "bust_out": bust_out, "synthetic_identity": synthetic_identity, } -DECOY_FUNCTIONS = {"fan_in": decoy_fan_in, "fan_out": decoy_fan_out, "cycle": decoy_cycle} +PATTERN_FUNCTIONS.update( + {"decoy_fan_in": decoy_fan_in, "decoy_fan_out": decoy_fan_out, "decoy_cycle": decoy_cycle} +) def inject( @@ -552,27 +558,26 @@ def inject( n_devices: int, ) -> Injection: ctx = TypologyContext(rng, pop, config, merchants, n_customers, n_devices) - patterns: list[Pattern] = [] - counter = 0 - for typology in TYPOLOGIES: - for _ in range(config.pattern_counts.get(typology, 0)): - pattern = Pattern(pattern_id=f"pat_{counter}", typology=typology, is_fraud=True) - TYPOLOGY_FUNCTIONS[typology](ctx, pattern) - patterns.append(pattern) - counter += 1 - - n_decoys = round(config.profile.decoy_ratio * config.num_patterns) - ctx.allow_overlap = False - for i in range(n_decoys): - typology = DECOY_TYPOLOGIES[i % len(DECOY_TYPOLOGIES)] - pattern = Pattern(pattern_id=f"pat_{counter}", typology=typology, is_fraud=False) - DECOY_FUNCTIONS[typology](ctx, pattern) - patterns.append(pattern) - counter += 1 - - fraud_accounts = {acc for p in patterns if p.is_fraud for acc in p.roles} - keep = config.profile.activity_camouflage - stripped = {acc for acc in fraud_accounts if rng.random() >= keep} + counter = count(0) + + def make(name: str, _index: int, labelled: bool) -> Pattern: + # Patterns are numbered in injection order rather than per shape, and + # a decoy reports the typology it imitates: the truth says what the + # structure looks like, and ``is_fraud`` says whether it is one. + return Pattern( + pattern_id=f"pat_{next(counter)}", + name=CATALOG.twins.get(name, name), + labelled=labelled, + ) + + patterns = run_catalog( + ctx, + config.pattern_counts, + PATTERN_FUNCTIONS, + make, + round_robin_decoys(CATALOG, round(config.profile.decoy_ratio * config.num_patterns)), + ) + stripped = stripped_members(rng, patterns, config.profile.activity_camouflage) return Injection( patterns=patterns, diff --git a/graphfaker/engine/__init__.py b/graphfaker/engine/__init__.py index 4bb7d10..18fab3d 100644 --- a/graphfaker/engine/__init__.py +++ b/graphfaker/engine/__init__.py @@ -16,6 +16,9 @@ "fingerprint": "graphfaker.engine.run", "generate": "graphfaker.engine.run", "Streams": "graphfaker.engine.seeding", + "InjectionContext": "graphfaker.engine.injection", + "Pattern": "graphfaker.engine.injection", + "run_catalog": "graphfaker.engine.injection", } __all__ = sorted(_EXPORTS) @@ -31,6 +34,9 @@ def __getattr__(name: str): if TYPE_CHECKING: # pragma: no cover + from graphfaker.engine.injection import InjectionContext as InjectionContext + from graphfaker.engine.injection import Pattern as Pattern + from graphfaker.engine.injection import run_catalog as run_catalog from graphfaker.engine.run import GraphRun as GraphRun from graphfaker.engine.run import Manifest as Manifest from graphfaker.engine.run import fingerprint as fingerprint diff --git a/graphfaker/engine/injection.py b/graphfaker/engine/injection.py new file mode 100644 index 0000000..6ec23f2 --- /dev/null +++ b/graphfaker/engine/injection.py @@ -0,0 +1,192 @@ +"""Running a pattern catalogue against a generated population. + +The fraud pack hides laundering in a bank; the coordination pack hides +amplification in a social platform. Underneath they do the same six things: +recruit members without reusing them unless the dials say to, place the +pattern somewhere in the period, record every row it creates against its id, +keep the roles for the truth tables, strip ordinary activity from the +accounts that are meant to look bare, and then do all of it again for the +decoys with overlap turned off. + +That is what lives here. What does not is the drawing itself: how many +sources a fan-in has, what a copypasta posts, what an amount looks like. +Those differ per domain and per shape, and the useful thing a shared layer +can do is leave them alone while making sure the bookkeeping around them is +identical, because the bookkeeping is what the truth tables and every metric +are computed from. + +A pack subclasses :class:`InjectionContext`, keeps its own drawing methods on +the subclass, and calls :func:`run_catalog`. +""" + +from __future__ import annotations + +from collections.abc import Callable, Sequence +from dataclasses import dataclass, field +from typing import Any + +import numpy as np + +from graphfaker.schema.patterns import Camouflage, PatternCatalog + + +@dataclass +class Pattern: + """One injected structure and everything the truth needs to describe it. + + ``name`` is the shape (``fan_in``, ``copypasta``); ``labelled`` is whether + it is the thing being looked for, so a decoy is a pattern with + ``labelled=False`` rather than a different kind of object. Packs subclass + this to give the two fields the names their readers use. + """ + + pattern_id: str + name: str + labelled: bool + roles: dict[int, str] = field(default_factory=dict) # member index -> role + start: np.datetime64 | None = None + end: np.datetime64 | None = None + events: int = 0 + #: Anything the shape needs to record that the truth reports, such as the + #: topic a campaign pushes. + extra: dict[str, Any] = field(default_factory=dict) + + def touch(self, stamp: np.datetime64) -> None: + """Record one event at ``stamp``, widening the pattern's window.""" + self.events += 1 + if self.start is None or stamp < self.start: + self.start = stamp + if self.end is None or stamp > self.end: + self.end = stamp + + +class InjectionContext: + """State shared by every shape in one run: who is taken, which dials are + set, and where the period starts and ends. + + Subclasses add the domain's own drawing. They should call :meth:`claim` + when they recruit, so that overlap and decoy separation hold across + shapes, and :meth:`Pattern.touch` for every row they emit, so that a + pattern's window and event count are right whatever drew them. + """ + + def __init__( + self, + rng: np.random.Generator, + profile: Camouflage, + catalog: PatternCatalog, + period_start: np.datetime64, + period_days: int, + ): + self.rng = rng + self.profile = profile + self.catalog = catalog + self.period_start = period_start + self.period_days = period_days + self.period_end = period_start + np.timedelta64(period_days * 86_400, "s") + #: Members already in a pattern. Recruitment consults it so that + #: patterns overlap only as often as the dial says. + self.used: set[int] = set() + #: While True, recruitment may reuse members. Decoys turn it off: an + #: innocent structure sharing members with a real one would have an + #: ambiguous label, and being unambiguous is the whole point of it. + self.allow_overlap = True + + def claim(self, members: np.ndarray | list[int]) -> None: + self.used.update(int(m) for m in members) + + def scaled(self, size: int, floor: int = 2) -> int: + """A pattern size after the size dial, never below ``floor``. + + Size is the one property camouflage cannot blend away: a collector + with fifteen senders is an outlier wherever an ordinary account has + three partners a quarter, so hiding costs members. + """ + return max(floor, round(size * self.profile.size_scale)) + + +def run_catalog( + ctx: InjectionContext, + counts: dict[str, int], + functions: dict[str, Callable[[Any, Any], None]], + make: Callable[[str, int, bool], Any], + decoys: Sequence[str] = (), + drop_empty: bool = False, +) -> list[Any]: + """Inject every pattern the budget asks for, then the decoys. + + ``counts`` is how many of each injected shape, ``functions`` draws one, + and ``make(name, index, labelled)`` builds the pattern record, so a pack + keeps its own numbering and its own subclass. ``decoys`` is the exact + sequence of decoy shapes to inject, which :func:`round_robin_decoys` and + :func:`grouped_decoys` build. ``drop_empty`` discards a pattern that + recruited nobody, which happens on populations too small for a shape. + + Order is part of the contract, not an implementation detail: shapes in + catalogue order, each shape's patterns in index order, decoys last and + with overlap off. A dataset is reproducible only if this loop is, so a + pack that changes the order changes its data. + """ + patterns: list[Any] = [] + for name in ctx.catalog.injected: + for index in range(counts.get(name, 0)): + pattern = make(name, index, True) + functions[name](ctx, pattern) + if pattern.roles or not drop_empty: + patterns.append(pattern) + + if len(decoys): + # Decoys do not share members with anything: an innocent structure + # that overlapped a real one would have an ambiguous label, and being + # unambiguous is the only reason it is in the data. + ctx.allow_overlap = False + seen: dict[str, int] = {} + for name in decoys: + index = seen.get(name, 0) + seen[name] = index + 1 + pattern = make(name, index, False) + functions[name](ctx, pattern) + if pattern.roles or not drop_empty: + patterns.append(pattern) + return patterns + + +def round_robin_decoys(catalog: PatternCatalog, total: int) -> list[str]: + """``total`` decoys, cycling through the decoy shapes one at a time. + + Use this when the budget is a count rather than a count per shape: every + shape gets its first decoy before any gets its second, so a small budget + still answers each of the naive rules the decoys exist to challenge. + """ + shapes = catalog.decoys + if not shapes or total <= 0: + return [] + return [shapes[i % len(shapes)] for i in range(total)] + + +def grouped_decoys(catalog: PatternCatalog, total: int) -> list[str]: + """``total`` decoys split evenly, all of one shape before the next. + + The same budget as :func:`round_robin_decoys` spent in a different order, + which matters because the order is part of what a seed reproduces. + """ + shapes = catalog.decoys + if not shapes or total <= 0: + return [] + per_shape = max(1, total // len(shapes)) + return [name for name in shapes for _ in range(per_shape)] + + +def stripped_members( + rng: np.random.Generator, patterns: list[Any], camouflage: float +) -> set[int]: + """Members whose ordinary activity is removed. + + An account that only ever does the pattern is the loudest signal in the + data, so ``activity_camouflage`` decides what share keep behaving + normally. Only labelled patterns are considered: a decoy that lost its + ordinary activity would stop being innocent-looking, which would defeat + the purpose of having it. + """ + members = {member for p in patterns if p.labelled for member in p.roles} + return {member for member in members if rng.random() >= camouflage} diff --git a/graphfaker/schema/__init__.py b/graphfaker/schema/__init__.py index b6938ac..6592d0d 100644 --- a/graphfaker/schema/__init__.py +++ b/graphfaker/schema/__init__.py @@ -1,5 +1,6 @@ """Declarative graph schemas: node types, samplers, latent factors, edge -families and topology models. See ``docs/design/synthetic-at-scale.md``.""" +families, topology models, and the catalogue of patterns a domain injects. +See ``docs/design/synthetic-at-scale.md``.""" from graphfaker.schema.graph import ( DegreeDerived, @@ -10,6 +11,11 @@ RealismTargets, Relationship, ) +from graphfaker.schema.patterns import ( + Camouflage, + PatternCatalog, + PatternSpec, +) from graphfaker.schema.samplers import ( BernoulliSampler, CategorySampler, @@ -36,6 +42,7 @@ __all__ = [ "BernoulliSampler", + "Camouflage", "CategorySampler", "ConstantSampler", "DegreeDerived", @@ -51,6 +58,8 @@ "MixtureSampler", "NodeType", "NumericAffinity", + "PatternCatalog", + "PatternSpec", "PoissonSampler", "RealismTargets", "ReferenceSampler", diff --git a/graphfaker/schema/patterns.py b/graphfaker/schema/patterns.py new file mode 100644 index 0000000..dfc54b8 --- /dev/null +++ b/graphfaker/schema/patterns.py @@ -0,0 +1,148 @@ +"""Injected patterns, declared. + +A domain pack that hides something in its data needs the same four things, +whether the thing is money laundering or a paid amplification ring: a +catalogue of shapes to inject, a budget saying how many of each, a set of +dials deciding how well they hide, and decoys that look like the real thing +and are not. This module is where a pack declares them. + +What it does not hold is how a shape is drawn. ``fan_in`` and ``copypasta`` +are code, because the thing that makes an injected pattern worth having is +exactly the part that does not generalise: which accounts, in what order, +with what amounts or text. The catalogue says what exists and how much of it; +:mod:`graphfaker.engine.injection` runs it. +""" + +from __future__ import annotations + +from pydantic import BaseModel, ConfigDict, Field + + +class Camouflage(BaseModel): + """How well injected patterns hide, as numbers rather than adjectives. + + Every dial trades detectability for something. ``signature_blend`` + replaces the giveaway a detector was written to catch (a round amount, an + identical message) with a draw from the legitimate distribution, so the + feature stops carrying signal. ``timing_spread`` stretches a pattern's + steps from a burst to a background. ``overlap`` lets patterns share + members, which is realistic and makes attribution harder. + ``activity_camouflage`` keeps ordinary behaviour on the accounts a + pattern recruits, so none of them is single-purpose. ``size_scale`` + shrinks patterns, because degree is the one signal blending cannot hide. + ``decoy_ratio`` adds innocent structures of the same shape, which is what + makes precision measurable at all. + + A pack subclasses this to add dials of its own and to give these ones the + names its readers use. + """ + + model_config = ConfigDict(frozen=True, extra="forbid") + + signature_blend: float = Field(ge=0.0, le=1.0) + timing_spread: float = Field(gt=0.0) + overlap: float = Field(ge=0.0, le=1.0) + decoy_ratio: float = Field(ge=0.0) + activity_camouflage: float = Field(ge=0.0, le=1.0) + size_scale: float = Field(default=1.0, gt=0.0, le=1.0) + + +class PatternSpec(BaseModel): + """One shape a domain can inject. + + ``span_days`` is the natural width of the pattern before + :attr:`Camouflage.timing_spread` stretches it: a mule hand-off is hours, a + bust-out is months. ``share`` is its slice of the pattern budget. + A spec with ``imitates`` set is a decoy: the same shape as the spec it + names, labelled innocent, and counted as a false positive when a detector + flags it. + """ + + model_config = ConfigDict(frozen=True, extra="forbid") + + name: str + span_days: float = Field(default=1.0, gt=0.0) + share: float = Field(default=0.0, ge=0.0) + imitates: str | None = None + + @property + def is_decoy(self) -> bool: + return self.imitates is not None + + +class PatternCatalog(BaseModel): + """Every shape a pack injects, and how the budget is split between them. + + ``base`` is the number of patterns at ``scale=1.0``; ``floor`` is the + minimum per shape, so a small dataset still covers the catalogue instead + of containing three copies of the most common thing. + """ + + model_config = ConfigDict(frozen=True, extra="forbid") + + patterns: list[PatternSpec] = Field(min_length=1) + base: int = Field(default=1_000, gt=0) + floor: int = Field(default=2, ge=0) + + def spec(self, name: str) -> PatternSpec: + for spec in self.patterns: + if spec.name == name: + return spec + raise KeyError(f"unknown pattern {name!r}; the catalogue has {', '.join(self.names)}") + + @property + def names(self) -> tuple[str, ...]: + """Every shape, in the order patterns are allocated and injected.""" + return tuple(spec.name for spec in self.patterns) + + @property + def injected(self) -> tuple[str, ...]: + """The shapes that are the thing to find.""" + return tuple(spec.name for spec in self.patterns if not spec.is_decoy) + + @property + def decoys(self) -> tuple[str, ...]: + """The shapes that imitate another and are labelled innocent.""" + return tuple(spec.name for spec in self.patterns if spec.is_decoy) + + @property + def twins(self) -> dict[str, str]: + """Decoy name to the shape it imitates.""" + return {spec.name: spec.imitates for spec in self.patterns if spec.imitates} + + def span_days(self, name: str) -> float: + return self.spec(name).span_days + + def total(self, scale: float) -> int: + """How many patterns a dataset of this scale gets. + + At least ``floor`` of every injected shape, so the catalogue is + covered even at the sizes people try first. + """ + return max(self.floor * len(self.injected), int(self.base * scale)) + + def counts(self, total: int) -> dict[str, int]: + """Allocate ``total`` patterns across the injected shapes. + + Proportional to ``share``, every shape at least ``floor``, the + remainder going to the largest fractional parts, so the counts add up + to ``total`` exactly rather than drifting with rounding. + """ + mix = {spec.name: spec.share for spec in self.patterns if not spec.is_decoy} + weight = sum(mix.values()) + if weight <= 0: # no shares declared: split evenly + mix = dict.fromkeys(mix, 1.0 / len(mix)) + elif abs(weight - 1.0) > 1e-9: + mix = {name: share / weight for name, share in mix.items()} + + counts = dict.fromkeys(mix, self.floor) + remaining = total - self.floor * len(mix) + if remaining <= 0: + return counts + exact = {name: remaining * share for name, share in mix.items()} + for name, value in exact.items(): + counts[name] += int(value) + leftover = remaining - sum(int(v) for v in exact.values()) + for name in sorted(exact, key=lambda n: exact[n] - int(exact[n]), reverse=True)[:leftover]: + counts[name] += 1 + return counts diff --git a/graphfaker/sinks/pyg.py b/graphfaker/sinks/pyg.py index 8a56f23..ac4379b 100644 --- a/graphfaker/sinks/pyg.py +++ b/graphfaker/sinks/pyg.py @@ -21,14 +21,20 @@ out of ``x`` and attached as their own tensor (``data["Person"].community``) for use as a label, not a feature. -Labels, when ``truth`` is given: - -* the node type the truth's ``accounts`` frame refers to gets ``y`` (1 for a - node in a fraud pattern, 0 otherwise), ``decoy`` (1 for a node that is in - a legitimate look-alike pattern only), and stratified ``train_mask``, - ``val_mask`` and ``test_mask``; -* every relationship with a ``tx_id`` gets ``y`` (1 for an injected fraud - transaction) and ``edge_time`` in seconds, for temporal splits. +Labels, when ``truth`` is given. Rather than naming one domain's columns, +these follow the convention every domain pack already writes: a truth frame +keyed by ``_id`` with a single boolean column labels those entities. +The fraud pack writes ``accounts.is_fraud`` and ``transactions.is_fraud``; the +coordination pack writes ``accounts.is_coordinated`` and +``events.is_coordinated``; a third pack gets labels for free. + +* the node type a truth frame's ``*_id`` column refers to gets ``y`` (1 where + the boolean is true), ``decoy`` (1 for an entity that appears *only* in + rows where it is false — a legitimate look-alike), and stratified + ``train_mask``, ``val_mask`` and ``test_mask``; +* a relationship carrying an ``*_id`` column that a truth frame also carries + gets ``y`` from that frame, and ``edge_time`` in seconds wherever there is a + timestamp, for temporal splits. Feature names are kept in ``data[type].feature_names`` so a column can be found again after training, and the standardisation statistics in @@ -171,19 +177,86 @@ def encode_features( return np.stack(columns, axis=1), names, stats +def _single_boolean(frame: pl.DataFrame) -> str | None: + """The frame's one boolean column, or None if there is not exactly one. + + A label frame carries a single flag. Requiring exactly one avoids guessing + between two, which is the point at which a convention should fail loudly + rather than pick. + """ + booleans = [c for c, t in frame.schema.items() if t == pl.Boolean] + return booleans[0] if len(booleans) == 1 else None + + def _latent_names(truth: dict[str, pl.DataFrame] | None) -> set[str]: + """Truth frames that describe a latent factor rather than label anything. + + A latent frame is keyed by ``group`` and carries that group's generation + parameters. Naming the label frames to exclude instead (``patterns``, + ``accounts``, ``transactions``) only worked for the fraud pack; a frame with + an ``*_id`` column is labelling entities, whatever it is called. + """ if not truth: return set() - return {name for name, frame in truth.items() if name not in ("patterns", "accounts", "transactions") and "group" in frame.columns} + return { + name + for name, frame in truth.items() + if "group" in frame.columns + and not any(c.endswith("_id") for c in frame.columns) + } + + +def _node_label_frame( + tables: GraphTables, truth: dict[str, pl.DataFrame] +) -> tuple[str, str, pl.DataFrame, str] | None: + """``(node_type, id_column, frame, label_column)`` for the frame that + labels nodes, or None. + + The id column has to *end in* ``_id`` as well as match a node table's ids. + Matching on values alone picked the coordination pack's + ``campaigns.topic`` column — which holds real Topic ids next to a boolean — + and cheerfully labelled every topic. + """ + for frame in truth.values(): + label = _single_boolean(frame) + if label is None: + continue + for column, dtype in frame.schema.items(): + if not column.endswith("_id") or dtype not in (pl.String, pl.Utf8): + continue + sample = set(frame[column].drop_nulls().to_list()[:200]) + if not sample: + continue + for node_type, node_frame in tables.nodes.items(): + if sample <= set(node_frame[ID].to_list()): + return node_type, column, frame, label + return None -def _labelled_node_type(tables: GraphTables, members: pl.DataFrame) -> str | None: - """Which node table the truth's ``accounts`` frame refers to.""" - wanted = set(members["account_id"].to_list()[:100]) - for name, frame in tables.nodes.items(): - if wanted and wanted <= set(frame[ID].to_list()): - return name - return None +def _edge_label_ids( + tables: GraphTables, truth: dict[str, pl.DataFrame] +) -> dict[str, tuple[str, pl.Series]]: + """``relationship -> (id_column, positive ids)`` for labelled edges.""" + found: dict[str, tuple[str, pl.Series]] = {} + for rel, frame in tables.edges.items(): + candidates = [ + c for c in frame.columns if c.endswith("_id") and c not in (SOURCE, TARGET) + ] + for column in candidates: + for truth_frame in truth.values(): + if column not in truth_frame.columns: + continue + label = _single_boolean(truth_frame) + if label is None: + continue + found[rel] = ( + column, + truth_frame.filter(pl.col(label))[column].unique(), + ) + break + if rel in found: + break + return found def _split_masks(y: np.ndarray, split: tuple[float, float, float], seed: int) -> dict[str, np.ndarray]: @@ -243,7 +316,7 @@ def arrays( nodes[name] = node edges: dict[tuple[str, str, str], EdgeArrays] = {} - tx_truth = truth.get("transactions") if truth else None + edge_labels = _edge_label_ids(tables, truth) if truth else {} for rel, frame in tables.edges.items(): if rel not in endpoints or frame.height == 0: continue @@ -255,30 +328,60 @@ def arrays( np.fromiter((dst_pos[v] for v in frame[TARGET].to_list()), dtype=np.int64, count=frame.height), ] ) - skip = {SOURCE, TARGET, "tx_id", *exclude.get(rel, [])} + # Identifiers are not features: a standardised ``tx_id`` or + # ``template_id`` is a meaningless axis, and the information it stands + # for is in the structure. + skip = { + SOURCE, + TARGET, + *(c for c in frame.columns if c.endswith("_id")), + *exclude.get(rel, []), + } attrs = frame.drop([c for c in (SOURCE, TARGET) if c in frame.columns]) + # A foreign key on an edge is structure too. The coordination pack's + # interaction edges carry the topic they are about, which one-hot + # encoded to 48 columns here and would reach thousands at scale. + skip |= { + c + for c, t in attrs.schema.items() + if t in (pl.String, pl.Utf8) and _is_foreign_key(attrs[c], all_ids) + } x, names, _ = encode_features(attrs, exclude=skip, max_categories=max_categories, standardize=standardize) edge = EdgeArrays(src_type=src, dst_type=dst, edge_index=edge_index, edge_attr=x if names != ["constant"] else None, feature_names=names if names != ["constant"] else []) if "timestamp" in frame.columns: edge.edge_time = frame["timestamp"].cast(pl.Datetime("us")).dt.epoch("s").to_numpy().astype(np.int64) - if tx_truth is not None and "tx_id" in frame.columns: - fraud_ids = tx_truth.filter(pl.col("is_fraud"))["tx_id"].unique() - edge.y = frame["tx_id"].is_in(fraud_ids.implode()).to_numpy().astype(np.int64) + if rel in edge_labels: + column, positive = edge_labels[rel] + edge.y = frame[column].is_in(positive.implode()).to_numpy().astype(np.int64) edges[(src, rel, dst)] = edge - members = truth.get("accounts") if truth else None - if members is not None and members.height: - label_type = _labelled_node_type(tables, members) - if label_type is None: - logger.warning("pyg: the truth's accounts match no node table; no node labels written") - else: - node = nodes[label_type] - fraud_ids = set(members.filter(pl.col("is_fraud"))["account_id"].to_list()) - decoy_ids = set(members.filter(~pl.col("is_fraud"))["account_id"].to_list()) - fraud_ids - ids = node.ids.tolist() - node.y = np.fromiter((1 if v in fraud_ids else 0 for v in ids), dtype=np.int64, count=len(ids)) - node.decoy = np.fromiter((1 if v in decoy_ids else 0 for v in ids), dtype=np.int64, count=len(ids)) - node.masks = _split_masks(node.y, split, seed) + labelled = _node_label_frame(tables, truth) if truth else None + if truth and labelled is None: + logger.warning( + "pyg: no truth frame has an *_id column matching a node table; " + "no node labels written" + ) + elif labelled is not None: + label_type, id_column, members, label_column = labelled + node = nodes[label_type] + positive_ids = set(members.filter(pl.col(label_column))[id_column].to_list()) + # A decoy is an entity that appears only in rows where the flag is + # false: a legitimate structure that looks like the real thing. + decoy_ids = ( + set(members.filter(~pl.col(label_column))[id_column].to_list()) - positive_ids + ) + ids = node.ids.tolist() + node.y = np.fromiter((1 if v in positive_ids else 0 for v in ids), dtype=np.int64, count=len(ids)) + node.decoy = np.fromiter((1 if v in decoy_ids else 0 for v in ids), dtype=np.int64, count=len(ids)) + node.masks = _split_masks(node.y, split, seed) + logger.debug( + "pyg: labelled %s from %s.%s (%d positive, %d decoy)", + label_type, + id_column, + label_column, + int(node.y.sum()), + int(node.decoy.sum()), + ) return GraphArrays(nodes=nodes, edges=edges) diff --git a/pyproject.toml b/pyproject.toml index 30934dc..6dec1d6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -64,7 +64,12 @@ dependencies = [ "requests>=2.32.3", "tqdm>=4.66.0", "typer", - "urllib3>=2.8.0", # 2.8.0 fixes PYSEC-2026-4175/4176/4177 + # >=2.8.0 for PYSEC-2026-4175/4176/4177 (CVE-2026-97688, CVE-2026-97689 and + # the redirect one): the first two are in the chunked streaming path, which + # is how the flights fetcher downloads the BTS zip (requests.get(..., + # stream=True)). One can hang on a crafted deflate response, the other + # buffers an unbounded chunk-size line. + "urllib3>=2.8.0", "wikipedia>=1.4.0", ] diff --git a/tests/test_coordination.py b/tests/test_coordination.py new file mode 100644 index 0000000..78cf92b --- /dev/null +++ b/tests/test_coordination.py @@ -0,0 +1,492 @@ +# tests/test_coordination.py +"""The coordinated-behaviour domain pack.""" + +from __future__ import annotations + +import numpy as np +import polars as pl +import pytest + +from graphfaker.backends.tables import ID +from graphfaker.domains import available, get +from graphfaker.domains.coordination import ( + DECOY_PLAYBOOKS, + PLAYBOOKS, + CoordinationConfig, + account_features, + evaluate, + generate, + hardness_report, + realism_report, +) +from graphfaker.domains.coordination.config import DECOY_TWIN +from graphfaker.domains.coordination.process import CHANNELS, POSTED +from graphfaker.engine.run import fingerprint + +SCALE = 0.0006 + + +@pytest.fixture(scope="module") +def run(): + return generate(scale=SCALE, tradecraft="medium", seed=7) + + +@pytest.fixture(scope="module") +def features(run): + return account_features(run) + + +# --------------------------------------------------------------------------- # +# contract +# --------------------------------------------------------------------------- # + + +def test_registered_as_a_domain(): + assert "coordination" in available() + domain = get("coordination") + assert domain.options is CoordinationConfig + assert domain.schema is not None + # The schema covers entities only, so it has to say so. + assert domain.schema_note and "entities only" in domain.schema_note + + +def test_reproducible_for_a_seed(): + a = generate(scale=SCALE, tradecraft="medium", seed=11) + b = generate(scale=SCALE, tradecraft="medium", seed=11) + assert fingerprint(a) == fingerprint(b) + + +def test_different_seeds_differ(): + a = generate(scale=SCALE, tradecraft="medium", seed=11) + b = generate(scale=SCALE, tradecraft="medium", seed=12) + assert fingerprint(a) != fingerprint(b) + + +def test_manifest_records_the_config(run): + extra = run.manifest.extra["coordination"] + assert extra["tradecraft"] == "medium" + assert extra["scale"] == SCALE + assert run.manifest.schema_name == "coordination" + + +def test_tables_and_truth_present(run): + for node_type in ("Account", "Topic", "Device"): + assert run.tables.nodes[node_type].height > 0 + for relationship in ("FOLLOWS", "USES", *CHANNELS): + assert relationship in run.tables.edges + for table in ("campaigns", "accounts", "events", "community"): + assert table in run.truth + + +def test_unknown_playbook_is_rejected(): + with pytest.raises(ValueError, match="unknown playbooks"): + CoordinationConfig(campaigns={"not_a_playbook": 3}) + + +def test_writes_and_reads_back(run, tmp_path): + root = run.write(tmp_path / "dataset") + assert (root / "manifest.json").exists() + assert (root / "schema.yaml").exists() + assert (root / "truth" / "campaigns.parquet").exists() + + +# --------------------------------------------------------------------------- # +# no label leakage +# --------------------------------------------------------------------------- # + + +def test_no_node_attribute_names_the_answer(run): + """A column called ``is_bot`` would make the dataset worthless.""" + banned = {"is_bot", "is_coordinated", "campaign_id", "playbook", "suspicious", "label"} + for node_type, frame in run.tables.nodes.items(): + assert not (banned & set(frame.columns)), node_type + + +def test_no_single_feature_separates_campaigns_perfectly(run, features): + """Every feature is legitimately available, so none may be a giveaway. + + A perfectly separating feature means a label leaked into the graph. This is + the test that would have caught ``reshare_count`` being exactly zero for + every handover account. + """ + report = hardness_report(run, features) + for playbook in report.playbooks: + if not playbook.is_coordinated: + continue + assert playbook.max_auc < 0.999, ( + f"{playbook.playbook} is perfectly separable by " + f"{playbook.best_feature}; a label has leaked into a feature" + ) + + +def test_features_carry_no_truth_columns(features): + assert "is_coordinated" not in features.columns + assert "campaign_id" not in features.columns + + +# --------------------------------------------------------------------------- # +# truth integrity +# --------------------------------------------------------------------------- # + + +def test_every_campaign_account_exists(run): + accounts = set(run.tables.nodes["Account"][ID].to_list()) + assert set(run.truth["accounts"]["account_id"].to_list()) <= accounts + + +def test_every_labelled_event_exists(run): + present = set() + for frame in run.tables.edges.values(): + if "event_id" in frame.columns: + present |= set(frame["event_id"].to_list()) + labelled = set(run.truth["events"]["event_id"].to_list()) + assert labelled <= present + assert labelled, "no events were labelled at all" + + +def test_campaign_and_account_truth_agree(run): + campaigns = run.truth["campaigns"] + per_campaign = dict(zip(campaigns["campaign_id"].to_list(), campaigns["n_accounts"].to_list())) + counted = ( + run.truth["accounts"].group_by("campaign_id").len().rename({"len": "n"}) + ) + for campaign_id, n in zip(counted["campaign_id"].to_list(), counted["n"].to_list()): + assert per_campaign[campaign_id] == n + + +def test_roles_are_named(run): + roles = set(run.truth["accounts"]["role"].to_list()) + assert roles + assert all(isinstance(role, str) and role for role in roles) + + +# --------------------------------------------------------------------------- # +# decoys +# --------------------------------------------------------------------------- # + + +def test_organic_decoys_are_present_and_labelled(run): + campaigns = run.truth["campaigns"] + organic = campaigns.filter(~pl.col("is_coordinated")) + assert organic.height > 0 + assert set(organic["playbook"].to_list()) <= set(DECOY_PLAYBOOKS) + coordinated = campaigns.filter(pl.col("is_coordinated")) + assert set(coordinated["playbook"].to_list()) <= set(PLAYBOOKS) + + +def test_low_tradecraft_has_no_decoys(): + """``decoy_ratio`` is 0 at ``low``: the easy setting is easy on purpose.""" + low = generate(scale=SCALE, tradecraft="low", seed=7) + assert low.truth["campaigns"].filter(~pl.col("is_coordinated")).height == 0 + + +def test_decoys_never_share_accounts_with_campaigns(run): + """An organic structure sharing members would have an ambiguous label.""" + truth = run.truth["accounts"] + coordinated = set( + truth.filter(pl.col("is_coordinated"))["account_id"].to_list() + ) + organic = set(truth.filter(~pl.col("is_coordinated"))["account_id"].to_list()) + assert not (coordinated & organic) + + +def test_each_decoy_has_a_twin_in_the_catalogue(): + for decoy, twin in DECOY_TWIN.items(): + assert decoy in DECOY_PLAYBOOKS + assert twin in PLAYBOOKS + + +def test_decoys_are_not_trivially_separable_from_their_twin(run): + """If one feature tells a fandom from a copypasta ring, it is not a decoy.""" + report = hardness_report(run) + assert report.decoy_separability + for decoy, (value, feature) in report.decoy_separability.items(): + assert value < 0.95, f"{decoy} separable from its twin by {feature} alone" + + +# --------------------------------------------------------------------------- # +# the tradecraft dial +# --------------------------------------------------------------------------- # + + +def test_tradecraft_makes_campaigns_harder(): + """The dial has to move the measurement, not just the label. + + Compared on the mean of per-playbook max AUC, because individual playbooks + are noisy at test scale while the aggregate is not. + """ + + def mean_max_auc(level: str) -> float: + report = hardness_report(generate(scale=SCALE, tradecraft=level, seed=3)) + values = [ + p.max_auc + for p in report.playbooks + if p.is_coordinated and not np.isnan(p.max_auc) + ] + return float(np.mean(values)) + + low, high = mean_max_auc("low"), mean_max_auc("high") + assert low > high + 0.1, f"tradecraft did not bite: low={low:.3f} high={high:.3f}" + + +def test_low_tradecraft_leaves_an_obvious_signal(): + report = hardness_report(generate(scale=SCALE, tradecraft="low", seed=3)) + best = max( + p.max_auc for p in report.playbooks if p.is_coordinated and not np.isnan(p.max_auc) + ) + assert best > 0.9 + + +def test_high_tradecraft_needs_structure_or_timing(): + """At ``high`` the giveaway should not be a bare identity attribute.""" + report = hardness_report(generate(scale=SCALE, tradecraft="high", seed=3)) + families = [ + p.best_family for p in report.playbooks if p.is_coordinated and p.best_family + ] + assert families + # Identity alone (fresh accounts, shared device) must not dominate. + assert families.count("identity") <= len(families) // 2 + + +def test_hardness_report_is_serialisable(run): + import json + + payload = hardness_report(run).as_dict() + assert json.loads(json.dumps(payload))["tradecraft"] == "medium" + + +# --------------------------------------------------------------------------- # +# temporal and structural consistency +# --------------------------------------------------------------------------- # + + +def test_no_account_acts_before_it_exists(run): + created = dict( + zip( + run.tables.nodes["Account"][ID].to_list(), + run.tables.nodes["Account"]["created_at"].to_list(), + ) + ) + for channel in CHANNELS: + frame = run.tables.edges[channel] + if frame.height == 0: + continue + offenders = [ + source + for source, stamp in zip( + frame["source"].to_list(), frame["timestamp"].dt.date().to_list() + ) + if stamp < created[source] + ] + assert not offenders, f"{channel}: {len(offenders)} events predate the account" + + +def test_events_fall_inside_the_period(run): + config = CoordinationConfig(**run.manifest.extra["coordination"]) + for channel in CHANNELS: + frame = run.tables.edges[channel] + if frame.height == 0: + continue + dates = frame["timestamp"].dt.date() + assert dates.min() >= config.period_start + assert dates.max() <= config.period_end + + +def test_event_ids_are_unique(run): + seen = [] + for frame in run.tables.edges.values(): + if "event_id" in frame.columns: + seen.extend(frame["event_id"].to_list()) + assert len(seen) == len(set(seen)) + + +def test_posts_point_at_topics_and_interactions_at_accounts(run): + topics = set(run.tables.nodes["Topic"][ID].to_list()) + accounts = set(run.tables.nodes["Account"][ID].to_list()) + posted = run.tables.edges[POSTED] + if posted.height: + assert set(posted["target"].to_list()) <= topics + assert set(posted["source"].to_list()) <= accounts + for channel in ("RESHARED", "REPLIED"): + frame = run.tables.edges[channel] + if frame.height: + assert set(frame["target"].to_list()) <= accounts + + +def test_no_self_interactions(run): + for channel in ("RESHARED", "REPLIED"): + frame = run.tables.edges[channel] + if frame.height: + assert not (frame["source"] == frame["target"]).any() + + +def test_follows_have_no_duplicates_or_self_loops(run): + follows = run.tables.edges["FOLLOWS"] + assert follows.height == follows.unique(subset=["source", "target"]).height + assert not (follows["source"] == follows["target"]).any() + + +# --------------------------------------------------------------------------- # +# realism +# --------------------------------------------------------------------------- # + + +def test_most_accounts_are_quiet(run): + """Real platforms are mostly lurkers; without them volume is uninformative.""" + report = realism_report(run) + assert 0.15 < report["silent_account_share"] < 0.6 + + +def test_audience_and_activity_are_heavy_tailed(run, features): + report = realism_report(run) + assert report["follower_gini"] > 0.3 + assert report["event_gini"] > 0.4 + following = features["following"].to_numpy() + # Out-degree too: drawing sources uniformly gave everyone the same count. + assert following.max() > 5 * max(1.0, float(np.median(following))) + + +def test_activity_follows_a_daily_rhythm(run): + """Sub-minute synchrony is only detectable because nothing organic is.""" + assert realism_report(run)["diurnal_ratio"] > 3.0 + + +def test_follows_are_substantially_reciprocal(run): + """A follow farm exaggerates reciprocity, so the organic rate must be real.""" + assert 0.15 < realism_report(run)["reciprocal_follow_share"] < 0.8 + + +def test_communities_are_recoverable_from_the_follow_graph(run): + import networkx as nx + + accounts = run.tables.nodes["Account"] + community = dict(zip(accounts[ID].to_list(), accounts["community"].to_list())) + graph = nx.Graph() + graph.add_nodes_from(community) + follows = run.tables.edges["FOLLOWS"] + graph.add_edges_from( + (s, t) for s, t in zip(follows["source"].to_list(), follows["target"].to_list()) if s != t + ) + groups: dict[int, set] = {} + for account, group in community.items(): + groups.setdefault(group, set()).add(account) + assert nx.community.modularity(graph, list(groups.values())) > 0.2 + + +def test_devices_are_sometimes_shared_innocently(run, features): + """Otherwise a sockpuppet cluster's shared fingerprint is a perfect tell.""" + shared = features["device_shared_with"].to_numpy() + assert shared.max() >= 3 + + +# --------------------------------------------------------------------------- # +# evaluation +# --------------------------------------------------------------------------- # + + +def _coordinated_accounts(run) -> list[str]: + truth = run.truth["accounts"] + return truth.filter(pl.col("is_coordinated"))["account_id"].unique().to_list() + + +def test_perfect_detector_scores_one(run): + scores = evaluate(run, flagged_accounts=_coordinated_accounts(run)) + assert scores.account.precision == pytest.approx(1.0) + assert scores.account.recall == pytest.approx(1.0) + assert scores.campaign.recall == pytest.approx(1.0) + assert scores.organic_false_positive_rate == pytest.approx(0.0) + + +def test_empty_detector_scores_zero(run): + scores = evaluate(run, flagged_accounts=[]) + assert scores.account.recall == 0.0 + assert scores.account.tp == 0 + assert scores.campaign.fn > 0 + + +def test_flagging_everything_is_punished(run): + everyone = run.tables.nodes["Account"][ID].to_list() + scores = evaluate(run, flagged_accounts=everyone) + assert scores.account.recall == pytest.approx(1.0) + assert scores.account.precision < 0.5 + # Every organic account is flagged too, which is the deployment cost. + assert scores.organic_false_positive_rate == pytest.approx(1.0) + + +def _co_burst_detector(run, min_accounts: int = 6) -> list[str]: + """The standard first-pass coordination heuristic. + + Bucket posts by (topic, hour) and flag every account in a bucket that many + distinct accounts share. This is co-occurrence, which is what real + coordination detection keys on; per-account burstiness is much weaker, + because a camouflaged campaign account's few campaign posts are diluted by + its ordinary activity. + """ + posted = run.tables.edges[POSTED] + if posted.height == 0: + return [] + buckets = ( + posted.with_columns(pl.col("timestamp").dt.truncate("1h").alias("hour")) + .group_by(["target", "hour"]) + .agg(pl.col("source").unique().alias("accounts")) + .filter(pl.col("accounts").list.len() >= min_accounts) + ) + flagged: set[str] = set() + for accounts in buckets["accounts"].to_list(): + flagged.update(str(a) for a in accounts) + return sorted(flagged) + + +def test_synchrony_only_detector_is_caught_by_the_decoys(run): + """The point of the pack, as an assertion. + + A co-occurrence detector finds real campaigns *and* every fan club and news + reaction, because those are the same shape. Without organic decoys in the + truth it would look excellent; with them, its organic false-positive rate + is visible and non-trivial, which is the cost a platform would actually pay. + """ + flagged = _co_burst_detector(run) + scores = evaluate(run, flagged_accounts=flagged) + assert scores.account.tp > 0, "the naive detector should find something" + assert scores.organic_false_positive_rate > 0.0, ( + "a co-occurrence detector must flag organic bursts; if it does not, the " + "decoys are not doing their job" + ) + + +def test_decoys_cost_the_naive_detector_precision(run): + """Scoring the same detector with and without the organic structures. + + Counting only the inauthentic campaigns flatters it; the organic accounts + it also flagged are the reason its real-world precision is worse. + """ + flagged = set(_co_burst_detector(run)) + truth = run.truth["accounts"] + coordinated = set(truth.filter(pl.col("is_coordinated"))["account_id"].to_list()) + organic = set(truth.filter(~pl.col("is_coordinated"))["account_id"].to_list()) - coordinated + assert flagged & organic, "no organic account was flagged, so nothing is being measured" + + +def test_recall_by_playbook_exposes_a_one_signal_detector(run, features): + """A follow farm barely posts, so an activity detector cannot see it.""" + noisy = features.filter(pl.col("event_count") > features["event_count"].median())[ + "account_id" + ].to_list() + scores = evaluate(run, flagged_accounts=noisy) + assert scores.recall_by_playbook + assert "follow_farm" in scores.recall_by_playbook + + +def test_evaluate_needs_ground_truth(run): + import dataclasses + + blind = dataclasses.replace(run, truth={}) + with pytest.raises(ValueError, match="no ground truth"): + evaluate(blind, flagged_accounts=[]) + + +def test_evaluation_is_serialisable(run): + import json + + payload = evaluate(run, flagged_accounts=_coordinated_accounts(run)).as_dict() + assert json.loads(json.dumps(payload))["account"]["recall"] == 1.0 diff --git a/tests/test_injection.py b/tests/test_injection.py new file mode 100644 index 0000000..47f8195 --- /dev/null +++ b/tests/test_injection.py @@ -0,0 +1,137 @@ +"""The pattern catalogue and the injection driver, which both domain packs run on.""" + +from __future__ import annotations + +import numpy as np +import pytest + +from graphfaker.engine.injection import ( + InjectionContext, + Pattern, + grouped_decoys, + round_robin_decoys, + run_catalog, + stripped_members, +) +from graphfaker.schema import Camouflage, PatternCatalog, PatternSpec + +CATALOG = PatternCatalog( + base=100, + floor=2, + patterns=[ + PatternSpec(name="ring", share=0.6, span_days=2.0), + PatternSpec(name="burst", share=0.4, span_days=0.5), + PatternSpec(name="choir", imitates="burst", span_days=0.5), + PatternSpec(name="club", imitates="ring", span_days=2.0), + ], +) + +PROFILE = Camouflage( + signature_blend=0.5, timing_spread=3.0, overlap=0.2, decoy_ratio=0.5, + activity_camouflage=0.6, size_scale=0.5, +) + + +def _ctx(**over) -> InjectionContext: + profile = PROFILE.model_copy(update=over) + return InjectionContext( + np.random.default_rng(0), profile, CATALOG, np.datetime64("2026-01-01", "s"), 90 + ) + + +def test_catalog_separates_what_is_injected_from_what_imitates_it(): + assert CATALOG.injected == ("ring", "burst") + assert CATALOG.decoys == ("choir", "club") + assert CATALOG.twins == {"choir": "burst", "club": "ring"} + assert CATALOG.span_days("ring") == 2.0 + assert not CATALOG.spec("ring").is_decoy and CATALOG.spec("club").is_decoy + with pytest.raises(KeyError, match="unknown pattern"): + CATALOG.spec("nope") + + +def test_counts_respect_the_floor_the_shares_and_the_total(): + # Small datasets get the floor of every shape, so the catalogue is covered. + assert CATALOG.total(0.001) == 4 and CATALOG.counts(4) == {"ring": 2, "burst": 2} + # Large ones follow the shares, and the parts add up to the whole. + counts = CATALOG.counts(CATALOG.total(1.0)) + assert sum(counts.values()) == 100 + assert counts["ring"] > counts["burst"] + # Decoys are not part of the budget: they are a ratio of what was injected. + assert set(counts) == {"ring", "burst"} + + +def test_counts_normalise_shares_that_do_not_sum_to_one(): + catalog = PatternCatalog( + patterns=[PatternSpec(name="a", share=3.0), PatternSpec(name="b", share=1.0)], floor=0 + ) + assert catalog.counts(100) == {"a": 75, "b": 25} + + +def test_decoy_orders_differ_and_both_spend_the_budget(): + # Round robin answers every naive rule once before answering any twice; + # grouped keeps a shape together. The two orders give different datasets + # from the same seed, which is why a pack picks one and keeps it. + assert round_robin_decoys(CATALOG, 3) == ["choir", "club", "choir"] + assert grouped_decoys(CATALOG, 4) == ["choir", "choir", "club", "club"] + assert round_robin_decoys(CATALOG, 0) == [] and grouped_decoys(CATALOG, 0) == [] + + +def test_run_catalog_order_ids_and_decoy_isolation(): + ctx = _ctx() + seen: list[tuple[str, bool, bool]] = [] + + def draw(context: InjectionContext, pattern: Pattern) -> None: + seen.append((pattern.name, pattern.labelled, context.allow_overlap)) + pattern.roles[len(seen)] = "member" + pattern.touch(np.datetime64("2026-02-01T00:00", "s")) + + patterns = run_catalog( + ctx, + {"ring": 2, "burst": 1}, + dict.fromkeys(("ring", "burst", "choir", "club"), draw), + lambda name, index, labelled: Pattern(f"{name}_{index}", name, labelled), + round_robin_decoys(CATALOG, 2), + ) + + assert [p.pattern_id for p in patterns] == ["ring_0", "ring_1", "burst_0", "choir_0", "club_0"] + # Injected shapes first in catalogue order, decoys last and never overlapping. + assert [overlap for _, _, overlap in seen] == [True, True, True, False, False] + assert [p.labelled for p in patterns] == [True, True, True, False, False] + assert all(p.events == 1 and p.start == p.end for p in patterns) + + +def test_run_catalog_can_drop_the_shapes_that_recruited_nobody(): + ctx = _ctx() + functions = {"ring": lambda c, p: None, "burst": lambda c, p: p.roles.update({1: "a"})} + kept = run_catalog(ctx, {"ring": 1, "burst": 1}, functions, _make, drop_empty=True) + assert [p.name for p in kept] == ["burst"] + both = run_catalog(ctx, {"ring": 1, "burst": 1}, functions, _make) + assert [p.name for p in both] == ["ring", "burst"] + + +def _make(name: str, index: int, labelled: bool) -> Pattern: + return Pattern(f"{name}_{index}", name, labelled) + + +def test_claim_and_size_apply_the_dials(): + ctx = _ctx() + ctx.claim(np.array([3, 4])) + ctx.claim([4, 5]) + assert ctx.used == {3, 4, 5} + # size_scale 0.5 halves a pattern, and the floor protects the shape. + assert ctx.scaled(10) == 5 and ctx.scaled(2) == 2 and ctx.scaled(10, floor=8) == 8 + assert _ctx(size_scale=1.0).scaled(10) == 10 + + +def test_stripped_members_only_touches_labelled_patterns(): + real = Pattern("a_0", "ring", True, roles={1: "x", 2: "y"}) + decoy = Pattern("club_0", "club", False, roles={3: "x"}) + # Full camouflage keeps everyone's ordinary activity; none strips it all. + assert stripped_members(np.random.default_rng(1), [real, decoy], 1.0) == set() + assert stripped_members(np.random.default_rng(1), [real, decoy], 0.0) == {1, 2} + + +def test_the_period_is_the_window_patterns_are_placed_in(): + ctx = _ctx() + assert ctx.period_end - ctx.period_start == np.timedelta64(90 * 86_400, "s") + assert ctx.allow_overlap and ctx.used == set() diff --git a/tests/test_pyg.py b/tests/test_pyg.py index acd9524..71b09a5 100644 --- a/tests/test_pyg.py +++ b/tests/test_pyg.py @@ -159,3 +159,108 @@ def test_hetero_data_round_trip(run, tmp_path): run.write(tmp_path / "bank") again = from_directory(tmp_path / "bank", seed=1) assert torch.equal(again["Account"].y, data["Account"].y) + + +# ------------------------------------------------------- other domain packs +# +# The label rules are a convention, not fraud's column names: a truth frame +# keyed by ``_id`` with one boolean column labels those entities. These +# assert the convention holds for a second pack, because the alternative is +# discovering it does not when someone adds a third. + + +@pytest.fixture(scope="module") +def coordination_run(): + from graphfaker.domains import coordination + + return coordination.generate(scale=0.0006, tradecraft="medium", seed=3) + + +@pytest.fixture(scope="module") +def coordination_built(coordination_run): + return arrays(coordination_run.tables, coordination_run.truth, seed=3) + + +def test_coordination_node_labels_come_from_is_coordinated( + coordination_run, coordination_built +): + account = coordination_built.nodes["Account"] + assert account.y is not None, "no node labels were written" + truth = coordination_run.truth["accounts"] + expected = set(truth.filter(pl.col("is_coordinated"))["account_id"].to_list()) + flagged = {v for v, label in zip(account.ids.tolist(), account.y) if label} + assert flagged == expected + + +def test_coordination_decoys_are_marked_not_positive( + coordination_run, coordination_built +): + """An organic account is ``decoy=1`` and ``y=0``: legitimate, and labelled.""" + account = coordination_built.nodes["Account"] + assert account.decoy is not None + assert int(account.decoy.sum()) > 0 + assert int((account.y & account.decoy).sum()) == 0 + + +def test_coordination_labels_the_event_channels(coordination_built): + labelled = { + rel for (_, rel, _), edge in coordination_built.edges.items() if edge.y is not None + } + assert {"POSTED", "RESHARED", "REPLIED"} <= labelled + # FOLLOWS and USES are structure; the truth says nothing about them. + assert "FOLLOWS" not in labelled + + +def test_coordination_topics_are_not_mistaken_for_the_labelled_type( + coordination_built, +): + """Regression: the ``campaigns`` frame holds real Topic ids next to a + boolean, so matching on values alone labelled every topic.""" + assert coordination_built.nodes["Topic"].y is None + assert coordination_built.nodes["Device"].y is None + + +def test_coordination_community_is_a_latent_factor_not_a_feature(coordination_built): + account = coordination_built.nodes["Account"] + assert "community" in account.latent + assert not any(name.startswith("community") for name in account.feature_names) + + +def test_identifiers_stay_out_of_edge_attributes(coordination_built): + """``event_id`` and ``template_id`` are identifiers; a standardised id is a + meaningless axis. ``topic`` on an interaction edge is a foreign key, and + one-hot encoding it added 48 columns that would reach thousands at scale.""" + for (_, rel, _), edge in coordination_built.edges.items(): + assert not any( + name.endswith("_id") or name.startswith("topic=") + for name in edge.feature_names + ), (rel, edge.feature_names) + + +def test_coordination_edge_times_are_present(coordination_built): + for (_, rel, _), edge in coordination_built.edges.items(): + if rel in {"POSTED", "RESHARED", "REPLIED"}: + assert edge.edge_time is not None, rel + + +def test_a_truth_frame_with_two_booleans_is_not_guessed_at(): + """The convention fails loudly rather than picking a column.""" + from graphfaker.sinks.pyg import _single_boolean + + frame = pl.DataFrame({"a_id": ["x"], "one": [True], "two": [False]}) + assert _single_boolean(frame) is None + + +def test_coordination_hetero_data_round_trip(coordination_run, tmp_path): + pytest.importorskip("torch_geometric") + from graphfaker.sinks.pyg import from_directory, write_pyg + + coordination_run.write(tmp_path / "platform") + written = write_pyg( + coordination_run.tables, tmp_path / "platform" / "graph.pt", coordination_run.truth + ) + assert written.exists() + data = from_directory(tmp_path / "platform") + assert data["Account"].num_nodes == coordination_run.tables.nodes["Account"].height + assert int(data["Account"].y.sum()) > 0 + assert hasattr(data["Account"], "train_mask")