Skip to content

qubic.rpc_client

Contains a RPC client implementation of AbstractCircuitRunner submitting and running circuits over a network.

CircuitRunnerClient

Bases: AbstractCircuitRunner

CircuitRunner instance that can be run from a remote machine (i.e. not on the RFSoC ARM core) over RPC. Used for submitting compiled circuits to a QubiC board and receiving the resulting (integrated IQ or ADC timestream) data. Should be a drop-in replacement for CircuitRunner for most experiments. Exposes the following methods from CircuitRunner:

run_circuit
run_circuit_batch
load_circuit
load_and_run_acq
Source code in qubic/rpc_client.py
 342
 343
 344
 345
 346
 347
 348
 349
 350
 351
 352
 353
 354
 355
 356
 357
 358
 359
 360
 361
 362
 363
 364
 365
 366
 367
 368
 369
 370
 371
 372
 373
 374
 375
 376
 377
 378
 379
 380
 381
 382
 383
 384
 385
 386
 387
 388
 389
 390
 391
 392
 393
 394
 395
 396
 397
 398
 399
 400
 401
 402
 403
 404
 405
 406
 407
 408
 409
 410
 411
 412
 413
 414
 415
 416
 417
 418
 419
 420
 421
 422
 423
 424
 425
 426
 427
 428
 429
 430
 431
 432
 433
 434
 435
 436
 437
 438
 439
 440
 441
 442
 443
 444
 445
 446
 447
 448
 449
 450
 451
 452
 453
 454
 455
 456
 457
 458
 459
 460
 461
 462
 463
 464
 465
 466
 467
 468
 469
 470
 471
 472
 473
 474
 475
 476
 477
 478
 479
 480
 481
 482
 483
 484
 485
 486
 487
 488
 489
 490
 491
 492
 493
 494
 495
 496
 497
 498
 499
 500
 501
 502
 503
 504
 505
 506
 507
 508
 509
 510
 511
 512
 513
 514
 515
 516
 517
 518
 519
 520
 521
 522
 523
 524
 525
 526
 527
 528
 529
 530
 531
 532
 533
 534
 535
 536
 537
 538
 539
 540
 541
 542
 543
 544
 545
 546
 547
 548
 549
 550
 551
 552
 553
 554
 555
 556
 557
 558
 559
 560
 561
 562
 563
 564
 565
 566
 567
 568
 569
 570
 571
 572
 573
 574
 575
 576
 577
 578
 579
 580
 581
 582
 583
 584
 585
 586
 587
 588
 589
 590
 591
 592
 593
 594
 595
 596
 597
 598
 599
 600
 601
 602
 603
 604
 605
 606
 607
 608
 609
 610
 611
 612
 613
 614
 615
 616
 617
 618
 619
 620
 621
 622
 623
 624
 625
 626
 627
 628
 629
 630
 631
 632
 633
 634
 635
 636
 637
 638
 639
 640
 641
 642
 643
 644
 645
 646
 647
 648
 649
 650
 651
 652
 653
 654
 655
 656
 657
 658
 659
 660
 661
 662
 663
 664
 665
 666
 667
 668
 669
 670
 671
 672
 673
 674
 675
 676
 677
 678
 679
 680
 681
 682
 683
 684
 685
 686
 687
 688
 689
 690
 691
 692
 693
 694
 695
 696
 697
 698
 699
 700
 701
 702
 703
 704
 705
 706
 707
 708
 709
 710
 711
 712
 713
 714
 715
 716
 717
 718
 719
 720
 721
 722
 723
 724
 725
 726
 727
 728
 729
 730
 731
 732
 733
 734
 735
 736
 737
 738
 739
 740
 741
 742
 743
 744
 745
 746
 747
 748
 749
 750
 751
 752
 753
 754
 755
 756
 757
 758
 759
 760
 761
 762
 763
 764
 765
 766
 767
 768
 769
 770
 771
 772
 773
 774
 775
 776
 777
 778
 779
 780
 781
 782
 783
 784
 785
 786
 787
 788
 789
 790
 791
 792
 793
 794
 795
 796
 797
 798
 799
 800
 801
 802
 803
 804
 805
 806
 807
 808
 809
 810
 811
 812
 813
 814
 815
 816
 817
 818
 819
 820
 821
 822
 823
 824
 825
 826
 827
 828
 829
 830
 831
 832
 833
 834
 835
 836
 837
 838
 839
 840
 841
 842
 843
 844
 845
 846
 847
 848
 849
 850
 851
 852
 853
 854
 855
 856
 857
 858
 859
 860
 861
 862
 863
 864
 865
 866
 867
 868
 869
 870
 871
 872
 873
 874
 875
 876
 877
 878
 879
 880
 881
 882
 883
 884
 885
 886
 887
 888
 889
 890
 891
 892
 893
 894
 895
 896
 897
 898
 899
 900
 901
 902
 903
 904
 905
 906
 907
 908
 909
 910
 911
 912
 913
 914
 915
 916
 917
 918
 919
 920
 921
 922
 923
 924
 925
 926
 927
 928
 929
 930
 931
 932
 933
 934
 935
 936
 937
 938
 939
 940
 941
 942
 943
 944
 945
 946
 947
 948
 949
 950
 951
 952
 953
 954
 955
 956
 957
 958
 959
 960
 961
 962
 963
 964
 965
 966
 967
 968
 969
 970
 971
 972
 973
 974
 975
 976
 977
 978
 979
 980
 981
 982
 983
 984
 985
 986
 987
 988
 989
 990
 991
 992
 993
 994
 995
 996
 997
 998
 999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
class CircuitRunnerClient(AbstractCircuitRunner):
    """
    CircuitRunner instance that can be run from a remote machine (i.e. not on
    the RFSoC ARM core) over RPC. Used for submitting compiled circuits to a QubiC board
    and receiving the resulting (integrated IQ or ADC timestream) data. Should be a drop-in
    replacement for CircuitRunner for most experiments. Exposes the following methods from
    CircuitRunner:

        run_circuit
        run_circuit_batch
        load_circuit
        load_and_run_acq
    """

    def __init__(self, ip, port=9095, ddr_cmd=False):
        """ddr_cmd: set True when the target board runs the DDR command-
        streaming gateware (ff98e97e/633e80df lineage, batch_server on 8081) —
        run_circuit_batch then uses the hardware-sequenced DDR batch path.
        Default False keeps the classic BRAM/RPC behavior (sim, multiboard
        job server, older bitstreams)."""
        self._ip = ip
        self._ddr_cmd = bool(ddr_cmd)
        self.proxy = xmlrpc.client.ServerProxy('http://' + ip + ':' + str(port), allow_none=True)

    def run_circuit_batch(self,
                          executables: List[Executable],
                          n_total_shots: int | List[int],
                          reads_per_shot: int | Dict = None,
                          timeout_per_shot: float = 20,
                          reload_cmd: bool = True,
                          reload_freq: bool = True,
                          reload_env: bool = True,
                          zero_between_reload: bool = True,
                          ddr: bool = False,
                          ddr_direct: bool = True,
                          group: int = 4,
                          wr_base: int = 0,
                          batch_timeout: float = 600) -> List[Dict[str, PrimitiveResult]]:
        """
        Runs a batch of circuits given by a list of compiled executables.

        With `CircuitRunnerClient(..., ddr_cmd=True)` (ddr_streaming gateware,
        lineage 633e80df/ff98e97e): the batch is hardware-sequenced — commands stream over DDR c1 via batch_server (the
        BRAM command path is physically removed in this gateware), readout
        returns over DDR c0 via dma_server STRM, and ONE global_start pulse runs
        the whole batch. `n_total_shots` may be an int (same for every circuit)
        or a per-circuit list (the hardware config FIFO takes one n_shots per
        circuit). `reads_per_shot` may likewise be a per-circuit list (each
        element None = that circuit's result_channels metadata, or int/dict);
        a bare int/dict applies to every circuit. All circuits of a batch share the BRAM env/freq tables (checked;
        split batches if they differ). `reload_freq`/`reload_env` are honored:
        if the circuits' env/freq tables differ AND reloading is allowed, tables
        are reloaded before each circuit and the circuits run one at a time
        (NOT continuous — reported via last_batch_info); if reloading a
        differing kind is disabled, this raises. `reload_cmd`/
        `zero_between_reload`/`timeout_per_shot` are ignored on this path; see
        `group`, `wr_base`, `batch_timeout`. CNR metadata (did the circuits run seamlessly
        back-to-back?) is printed and stored in `self.last_batch_info`.
        Without the constructor flag (default), the classic
        one-circuit-at-a-time BRAM flow runs (sim/multiboard/older bitstreams);
        `ddr=True` selects the June readout-only DDR flow; for these,
        `n_total_shots` must be an int. For the legacy
        paths: `reads_per_shot` and `n_total_shots` are passed directly into
        `run_circuit`, and must be the same for all circuits in the batch. The
        parameters `reload_cmd`, `reload_freq`, `reload_env`, and
        `zero_between_reload` control which of these fields is rewritten
        circuit-to-circuit (everything is rewritten initially). Leave these all
        at `True` (default) for maximum safety, to ensure that QubiC is in a
        clean state before each run. Depending on the circuits, some of these
        can be turned off to save time.

        Parameters
        ----------
        executables : List[Executable]
            list of executables to run
        n_total_shots: int
            number of shots per circuit
        reads_per_shot: int | Dict[str, int]
            number of values per shot per channel to read back from accbuf. If `dict`, indexed
            by `str(channel_number)` (same indices as `raw_asm_list`). If `int`, assumed to be
            the same across channels, else can be a per-channel dict. Unless multiple circuits
            were rastered pre-compilation or there is mid-circuit measurement involved this is typically 1
        timeout_per_shot: float
            job will time out if time to take a single shot exceeds this value in seconds
            (this likely means the job is hanging due to timing issues in the program or gateware)
        reload_cmd: bool
            if True, reload command buffer between circuits
        reload_freq: bool
            if True, reload freq buffer between circuits
        reload_env: bool
            if True, reload env buffer between circuits
        ddr: bool
            if True, use DDR memory readout instead of BRAM. Data is fetched
            automatically via socket after each circuit completes. Default is False.

        Returns
        -------
        List[Dict]:
            Measurement results. A dictionary of results is returned for each circuit in the batch.
            This dict is keyed by  `ResultChannel` names in the `Executable`; values are arraylike objects
            containing the measurement results for that channel. Result type is given by the  `dtype` field
            in the `ResultChannel` object. These are always numpy-arraylike; a full listing  of available result
            types is given in `qubic.results.primitives`. Results have shape `(n_total_shots, reads_per_shot)`.

            Putting this together, a returned collection of results might looks like:
                `[{'Q0.rdlo': np.ndarray((n_shots, reads_per_shot)), 'Q1.rdlo': np.ndarray((n_shots, reads_per_shot))...},
                  {'Q0.rdlo': np.ndarray((n_shots, reads_per_shot)), ...}, ...]`

        """
        if ddr:
            if ddr_direct:
                return self._run_ddr_direct(executables, n_total_shots, reads_per_shot,
                                            reload_cmd, reload_freq, reload_env, zero_between_reload)
            else:
                return self._run_ddr_via_rpc(executables, n_total_shots, reads_per_shot,
                                             timeout_per_shot, reload_cmd, reload_freq,
                                             reload_env, zero_between_reload)

        if self._ddr_cmd:
            # DDR command-streaming gateware (633e80df/ff98e97e lineage): commands
            # can ONLY arrive over DDR c1 (the BRAM command path is physically
            # removed), so the batch is hardware-sequenced via batch_server +
            # dma_server STRM. n_total_shots may be a per-circuit list here.
            return self._run_ddr_cmd_batch(executables, n_total_shots, reads_per_shot,
                                           group=group, wr_base=wr_base,
                                           batch_timeout=batch_timeout,
                                           reload_freq=reload_freq,
                                           reload_env=reload_env)

        if not isinstance(n_total_shots, int):
            raise ValueError('the classic (non-ddr_cmd) path needs a single int n_total_shots')

        t_client_start = time.time()
        serialized_exe = [exec.to_dict() for exec in executables]
        t_serialize = time.time()
        packed_results = self.proxy.run_circuit_batch(serialized_exe, n_total_shots, reads_per_shot, float(timeout_per_shot),
                                     reload_cmd, reload_freq, reload_env, zero_between_reload, False)
        t_rpc = time.time()
        results = []
        for i, packed_result in enumerate(packed_results):
            results.append({ch: get_result_class(executables[i].result_channels[ch].dtype)(data.data, n_total_shots) for ch, data in packed_result.items()})
        t_parse = time.time()

        total_bytes = sum(len(data.data) for packed_result in packed_results for _, data in packed_result.items())
        print(f'[BRAM client timing]')
        print(f'  serialize      = {t_serialize - t_client_start:.3f}s')
        print(f'  XML-RPC call   = {t_rpc - t_serialize:.3f}s  (server processing + network transfer)')
        print(f'  client parse   = {t_parse - t_rpc:.3f}s')
        print(f'  TOTAL client   = {t_parse - t_client_start:.3f}s')
        print(f'  data received  = {total_bytes/1e6:.4f} MB')

        return results

    def _run_ddr_cmd_batch(self, executables, n_total_shots, reads_per_shot=None,
                           group=4, wr_base=0, batch_timeout=600.0,
                           reload_freq=True, reload_env=True):
        """DDR command-streaming batch (gateware 633e80df lineage).

        The circuit list is PARTITIONED into maximal runs of CONSECUTIVE
        circuits whose env/freq/register tables are byte-identical. Each such
        group runs as ONE hardware-sequenced batch (tables loaded once,
        circuits back-to-back, seamless-capable); between groups the differing
        table kinds are reloaded (only the kinds that actually changed).
        Uniform batch -> a single group (fast path); fully differing tables ->
        one group per circuit (legacy per-circuit semantics). A table kind that
        differs at some group boundary while its reload_{env,freq} flag is
        False raises ValueError before anything runs. Note: only CONSECUTIVE
        equal-table circuits merge — order of execution is preserved.

        Each group is reported: last_batch_info['groups'] lists per-group
        circuits + seamlessness (hardware CNR), and a per-group human-readable
        line is printed (e.g. "circuits 0-1: ran SEAMLESSLY").

        Per hardware batch, two concurrent threads over two board-side C
        servers: upload -> batch_server (8081) (per-circuit n_shots + 256 KiB
        command images; feeds DDR c1 config-before-image and pulses
        global_start), download -> dma_server (8080) STRM (exact expected byte
        count). Board prep (dsp resets + table BRAM loads) stays on the
        Python/XML-RPC path via ddr_batch_prepare.
        """
        k = len(executables)
        if isinstance(n_total_shots, (list, tuple, np.ndarray)):
            n_shots_list = [int(x) for x in n_total_shots]
            if len(n_shots_list) != k:
                raise ValueError(f'n_total_shots list has {len(n_shots_list)} entries '
                                 f'for {k} circuits')
        else:
            n_shots_list = [int(n_total_shots)] * k
        if any(n < 1 or n > 0xFFFF for n in n_shots_list):
            raise ValueError('each n_shots must be in 1..65535')

        # reads_per_shot: None -> each circuit's own result_channels metadata;
        # int/dict -> applied to every circuit; list (len K) -> PER-CIRCUIT spec,
        # each element again None | int | Dict[ch_name, int]
        if isinstance(reads_per_shot, (list, tuple)):
            if len(reads_per_shot) != k:
                raise ValueError(f'reads_per_shot list has {len(reads_per_shot)} '
                                 f'entries for {k} circuits')
            rps_spec = list(reads_per_shot)
        else:
            rps_spec = [reads_per_shot] * k

        # per-circuit readout channel maps + expected word counts
        rps_dicts, expect_words = [], []
        for i, exe in enumerate(executables):
            spec = rps_spec[i]
            rps = {}
            for ch_name, rc in exe.result_channels.items():
                if not ch_name.endswith('.rdlo'):
                    continue
                if isinstance(spec, dict):
                    rps[ch_name] = int(spec.get(ch_name, rc.reads_per_shot))
                elif spec is not None:
                    rps[ch_name] = int(spec)
                else:
                    rps[ch_name] = int(rc.reads_per_shot)
            if not rps:
                raise ValueError(f'circuit {i} has no .rdlo result channels')
            rps_dicts.append(rps)
            expect_words.append(n_shots_list[i] * sum(rps.values()))

        # Per-circuit, per-kind table snapshots. A hardware-sequenced batch
        # shares the BRAM tables (no software window to swap them between
        # back-to-back circuits), so table changes force a group boundary.
        def _tables(exe, kind):
            out = {}
            for name, d in exe.get_binaries_fromboard().items():
                low = name.lower()
                if 'command' in low:
                    continue
                this = 'env' if 'env' in low else ('freq' if 'freq' in low else 'other')
                if this == kind:
                    out[name] = bytes(d.data if hasattr(d, 'data') else d)
            return out
        tabs = []
        for exe in executables:
            t = {kind: _tables(exe, kind) for kind in ('env', 'freq', 'other')}
            t['other']['::registers::'] = repr(
                sorted(exe.get_registers_fromboard().items())).encode()
            tabs.append(t)

        # partition into maximal runs of consecutive equal-table circuits
        bounds = [0]
        for i in range(1, k):
            if any(tabs[i][kind] != tabs[bounds[-1]][kind]
                   for kind in ('env', 'freq', 'other')):
                bounds.append(i)
        groups = list(zip(bounds, bounds[1:] + [k]))

        # gate BEFORE running anything: a kind that changes at any group
        # boundary must be allowed to reload
        blocked = set()
        for gi in range(1, len(groups)):
            a_prev, a_cur = groups[gi - 1][0], groups[gi][0]
            if tabs[a_cur]['env'] != tabs[a_prev]['env'] and not reload_env:
                blocked.add('env')
            if tabs[a_cur]['freq'] != tabs[a_prev]['freq'] and not reload_freq:
                blocked.add('freq')
        if blocked:
            names = '/'.join(sorted(blocked))
            raise ValueError(
                f'circuits have differing {names} tables but reload of {names} is '
                'disabled: a hardware-sequenced DDR batch shares the BRAM tables. '
                'Set reload_env/reload_freq=True to reload between groups (circuits '
                'then do not all run continuously), or unify the tables / split '
                'into separate calls.')

        # run each group as one hardware batch; reload ONLY the kinds that
        # actually changed since the previously loaded group
        t0 = time.time()
        results, group_infos = [], []
        prev_rep = None
        for (a, b) in groups:
            if prev_rep is None:
                self.proxy.ddr_batch_prepare(executables[a].to_dict(), True, True)
            else:
                need_freq = tabs[a]['freq'] != tabs[prev_rep]['freq']
                need_env = tabs[a]['env'] != tabs[prev_rep]['env']
                self.proxy.ddr_batch_prepare(executables[a].to_dict(),
                                             need_freq, need_env)
            r_g, stat_g, _ = self._run_hw_batch(
                executables[a:b], n_shots_list[a:b], rps_dicts[a:b],
                expect_words[a:b], group, wr_base, batch_timeout)
            results += r_g
            group_infos.append({'circuits': list(range(a, b)),
                                'seamless': stat_g['cnr'] == 0,
                                'cnr': stat_g['cnr'],
                                'cnr_wait_cycles': stat_g['cnr_wait'],
                                'cnr_wait_us': stat_g['cnr_wait'] / _DSPCLK_MHZ,
                                'stat': stat_g})
            prev_rep = a
        t_run = time.time() - t0

        single = len(groups) == 1
        g0 = group_infos[0]
        self.last_batch_info = {'mode': 'hardware-batch' if single else 'grouped-reload',
                                'seamless': single and g0['seamless'],
                                'groups': group_infos,
                                'cnr': g0['cnr'] if single else None,
                                'cnr_wait_cycles': g0['cnr_wait_cycles'] if single else None,
                                'cnr_wait_us': g0['cnr_wait_us'] if single else None,
                                'stat': g0['stat'] if single else None}

        def _span(circuits):
            return (f'circuit {circuits[0]}' if len(circuits) == 1
                    else f'circuits {circuits[0]}-{circuits[-1]}')
        if single:
            if g0['seamless']:
                print(f'[DDR batch] {k} circuit(s) ran SEAMLESSLY (no boundary waits), '
                      f'{t_run:.3f}s total')
            else:
                print(f'[DDR batch] circuits were NOT seamless: the hardware waited for '
                      f'command data at one or more circuit boundaries '
                      f'(last boundary wait: {g0["cnr_wait_cycles"]} cycles = '
                      f'{g0["cnr_wait_us"]:.1f} us). Data is still complete/correct; to '
                      f'make runs seamless, prefill more (circuit >= 31.5 us) or raise '
                      f'per-circuit duration (streaming, group=4: >= ~80 us). '
                      f'{t_run:.3f}s total')
        else:
            print(f'[DDR batch] {k} circuits -> {len(groups)} hardware batches (env/freq/'
                  f'register tables change at circuit '
                  f'{", ".join(str(a) for a, _ in groups[1:])}); tables reloaded between '
                  f'batches (~ms gap). {t_run:.3f}s total')
            for gi_ in group_infos:
                cs = gi_['circuits']
                if len(cs) == 1:
                    print(f'  {_span(cs)}: ran alone')
                elif gi_['seamless']:
                    print(f'  {_span(cs)}: ran SEAMLESSLY (shared tables, no boundary '
                          f'waits)')
                else:
                    print(f'  {_span(cs)}: continuous group but NOT seamless (last '
                          f'boundary wait {gi_["cnr_wait_us"]:.1f} us)')
        return results

    def _run_hw_batch(self, executables, n_shots_list, rps_dicts, expect_words,
                      group, wr_base, batch_timeout):
        """One hardware-sequenced batch (caller has already done board prep).

        Returns (results, stat, t_run). See _run_ddr_cmd_batch for the protocol.
        """
        from qubic.rfsoc.ddr_cmd_pack import pack_batch, IMAGE_BYTES

        images = pack_batch(executables)
        total_bytes = 8 * int(sum(expect_words))
        if total_bytes > 0x8000_0000 or wr_base >= 0x8000_0000 or wr_base % 64:
            raise ValueError(f'readout of {total_bytes} B from base 0x{wr_base:x} does not '
                             'fit the 2 GiB DDR c0 ring (or base is not 64 B aligned)')
        box = {}

        t0 = time.time()
        # ORDER MATTERS: CFGB first (its cid_reset re-adopts wr_base, so cur_addr
        # is back at the base), THEN open the STRM session — otherwise the
        # download side could mistake a previous batch's leftover cur_addr
        # progress for fresh data.
        bs = _BatchServerSession(self._ip)
        try:
            bs.cfgb(n_shots_list, expect_words, group=group, wr_base=wr_base)
        except Exception:
            bs.close()
            raise
        strm = _DdrStrmSession(self._ip, total_bytes, wr_base=wr_base,
                               timeout=batch_timeout + 30.0)

        def _upload():
            try:
                step = 64 * IMAGE_BYTES              # 16 MiB per IMGS frame
                for off in range(0, len(images), step):
                    bs.imgs(images[off:off + step])
                box['stat'] = bs.wait(batch_timeout)
                strm.notify_done()                   # batch_done: unblock the tail drain
            except Exception as e:
                box['up_err'] = e

        def _download():
            try:
                box['raw'] = strm.recv_all()
            except Exception as e:
                box['down_err'] = e

        up = threading.Thread(target=_upload, name='ddr-batch-upload')
        down = threading.Thread(target=_download, name='ddr-batch-download')
        up.start()
        down.start()
        # the upload side fails fast (protocol errors, WAIT timeout/wedge); if it
        # did, unblock the download thread (shutdown() interrupts its blocking
        # recv; close() alone would not) before joining it
        up.join()
        if 'up_err' in box:
            strm.abort()
            down.join(timeout=30.0)
        else:
            down.join()
        strm.close()
        bs.close()
        if 'up_err' in box:
            raise box['up_err']
        if 'down_err' in box:
            raise box['down_err']
        t_run = time.time() - t0

        # split the batch blob per circuit, then demux per channel by tag
        results = []
        off = 0
        raw = box['raw']
        for i in range(len(executables)):
            nbytes = expect_words[i] * 8
            results.append(_parse_ddr_raw(raw[off:off + nbytes], rps_dicts[i],
                                          n_shots_list[i]))
            off += nbytes
        return results, box['stat'], t_run

    def _run_ddr_direct(self, executables, n_total_shots, reads_per_shot,
                        reload_cmd, reload_freq, reload_env, zero_between_reload):
        """DDR readout via direct TCP to DMA server, bypassing XML-RPC for data.

        Overlaps receiving and parsing: a parser thread processes each TCP
        chunk as it arrives, so parse time is hidden behind network transfer.
        """
        serialized_exe = [exe.to_dict() for exe in executables]

        # Normalize reads_per_shot
        if isinstance(reads_per_shot, int):
            rps_dict = {f'Q{i}.rdlo': reads_per_shot for i in range(8)}
        else:
            rps_dict = reads_per_shot

        # Build channel map once
        ch_map = {}  # tag_value -> (ch_name, n_reads)
        for ch_name, n_reads in rps_dict.items():
            if not ch_name.endswith('.rdlo'):
                continue
            m = re.search(r'(\d+)$', ch_name.split('.')[0])
            if m is not None:
                ch_map[int(m.group(1))] = (ch_name, n_reads)
        n_ch = len(ch_map)
        tag_values = sorted(ch_map.keys())
        tags_contiguous = (tag_values == list(range(n_ch)))

        all_circuit_results = []

        t_total_start = time.time()
        with _DdrRingSession(len(executables), host=self._ip) as ddr_session:
            t_connect = time.time()
            for i, ser_exe in enumerate(serialized_exe):
                t0 = time.time()
                self.proxy.ddr_start_circuit(ser_exe, n_total_shots, i,
                                             reload_cmd, reload_freq, reload_env, zero_between_reload)
                t_start = time.time()
                ddr_session.send_next()

                # --- Overlapped receive + parse ---
                chunk_queue = queue.Queue(maxsize=4)
                # Per-channel accumulator: list of 1-D complex arrays
                per_ch_parts = {t: [] for t in ch_map}
                parser_error = [None]
                total_bytes_parsed = [0]

                def _parser():
                    """Parse each chunk as it arrives."""
                    leftover = b''
                    try:
                        while True:
                            item = chunk_queue.get()
                            if item is None:
                                break
                            # Prepend any leftover bytes from previous chunk
                            if leftover:
                                item = leftover + item
                                leftover = b''
                            # Align to 8-byte u64 boundary
                            usable = (len(item) // 8) * 8
                            if usable < len(item):
                                leftover = item[usable:]
                                item = item[:usable]
                            if not item:
                                continue
                            data_u64 = np.frombuffer(item, dtype=np.dtype('<u8'))
                            total_bytes_parsed[0] += len(item)

                            if tags_contiguous and n_ch > 0 and len(data_u64) >= n_ch:
                                # Stride fast path: trim to multiple of n_ch
                                aligned = (len(data_u64) // n_ch) * n_ch
                                if aligned < len(data_u64):
                                    # Put unaligned tail back into leftover
                                    tail_bytes = (len(data_u64) - aligned) * 8
                                    leftover = item[-tail_bytes:] + leftover
                                data_2d = data_u64[:aligned].reshape(-1, n_ch)
                                for t in ch_map:
                                    col = np.ascontiguousarray(data_2d[:, t])
                                    per_ch_parts[t].append(_parse_ddr_channel(col))
                            else:
                                # Mask fallback
                                tag = (data_u64 >> 56).astype(np.uint8)
                                for t in ch_map:
                                    chunk_data = data_u64[tag == t]
                                    if len(chunk_data) > 0:
                                        per_ch_parts[t].append(_parse_ddr_channel(chunk_data))

                        # Process any final leftover
                        if leftover:
                            usable = (len(leftover) // 8) * 8
                            if usable > 0:
                                data_u64 = np.frombuffer(leftover[:usable], dtype=np.dtype('<u8'))
                                total_bytes_parsed[0] += usable
                                if tags_contiguous and n_ch > 0 and len(data_u64) >= n_ch:
                                    aligned = (len(data_u64) // n_ch) * n_ch
                                    if aligned > 0:
                                        data_2d = data_u64[:aligned].reshape(-1, n_ch)
                                        for t in ch_map:
                                            col = np.ascontiguousarray(data_2d[:, t])
                                            per_ch_parts[t].append(_parse_ddr_channel(col))
                                    # Any remaining words: mask fallback
                                    remainder = data_u64[aligned:]
                                    if len(remainder) > 0:
                                        tag = (remainder >> 56).astype(np.uint8)
                                        for t in ch_map:
                                            m = remainder[tag == t]
                                            if len(m) > 0:
                                                per_ch_parts[t].append(_parse_ddr_channel(m))
                                else:
                                    tag = (data_u64 >> 56).astype(np.uint8)
                                    for t in ch_map:
                                        m = data_u64[tag == t]
                                        if len(m) > 0:
                                            per_ch_parts[t].append(_parse_ddr_channel(m))
                    except Exception as e:
                        parser_error[0] = e

                parse_thread = threading.Thread(target=_parser)
                parse_thread.start()

                # Receiver: feed chunks into queue as they arrive
                total_bytes_recv = 0
                for chunk_bytes in ddr_session.recv_chunks():
                    total_bytes_recv += len(chunk_bytes)
                    chunk_queue.put(chunk_bytes)
                chunk_queue.put(None)  # signal end

                t_recv_done_i = time.time()
                parse_thread.join()
                t_parse_done_i = time.time()

                if parser_error[0] is not None:
                    raise parser_error[0]

                # Assemble per-channel results
                result = {}
                for t, (ch_name, n_reads) in ch_map.items():
                    if per_ch_parts[t]:
                        iq = np.concatenate(per_ch_parts[t])
                    else:
                        iq = np.array([], dtype=np.complex128)
                    expected = n_total_shots * n_reads
                    if len(iq) < expected:
                        iq = np.pad(iq, (0, expected - len(iq)),
                                    mode='constant', constant_values=0)
                    elif len(iq) > expected:
                        iq = iq[:expected]
                    result[ch_name] = S11(iq.reshape(n_total_shots, n_reads))

                all_circuit_results.append(result)

                recv_time = t_recv_done_i - t_start
                parse_lag = t_parse_done_i - t_recv_done_i
                speed = total_bytes_recv / recv_time / 1e6 if recv_time > 0 else 0
                print(f'[DDR client] circuit {i}: '
                      f'rpc_load_start={t_start-t0:.3f}s  '
                      f'tcp_recv={recv_time:.3f}s  '
                      f'parse_lag={parse_lag:.3f}s  '
                      f'data={total_bytes_recv/1e6:.1f}MB  '
                      f'speed={speed:.1f}MB/s')

        t_end = time.time()
        total_time = t_end - t_total_start
        overall_speed = total_bytes_recv / total_time / 1e6 if total_time > 0 else 0
        print(f'[DDR client] === SUMMARY ===')
        print(f'  tcp_connect  = {t_connect - t_total_start:.3f}s')
        print(f'  recv+parse   = {t_end - t_connect:.3f}s  (overlapped)')
        print(f'  TOTAL        = {total_time:.3f}s')
        print(f'  data         = {total_bytes_recv/1e6:.1f} MB')
        print(f'  effective    = {overall_speed:.1f} MB/s')

        return all_circuit_results

    def _run_ddr_via_rpc(self, executables, n_total_shots, reads_per_shot,
                         timeout_per_shot, reload_cmd, reload_freq,
                         reload_env, zero_between_reload):
        """DDR readout via XML-RPC: server reads from DMA server locally, returns raw bytes over XML-RPC."""
        t_start = time.time()
        serialized_exe = [exe.to_dict() for exe in executables]
        t_serialize = time.time()

        packed_results = self.proxy.run_circuit_batch(
            serialized_exe, n_total_shots, reads_per_shot, float(timeout_per_shot),
            reload_cmd, reload_freq, reload_env, zero_between_reload, True)  # ddr=True
        t_rpc = time.time()

        raw_results = [r.data for r in packed_results]
        t_extract = time.time()

        parsed = self._parse_ddr_results(raw_results, reads_per_shot, n_total_shots)
        t_parse = time.time()

        total_bytes = sum(len(r) for r in raw_results)
        total_time = t_parse - t_start
        print(f'[DDR via XML-RPC timing]')
        print(f'  serialize    = {t_serialize - t_start:.3f}s')
        print(f'  XML-RPC call = {t_rpc - t_serialize:.3f}s  (server DDR read + network transfer)')
        print(f'  extract      = {t_extract - t_rpc:.3f}s')
        print(f'  parse        = {t_parse - t_extract:.3f}s')
        print(f'  TOTAL        = {total_time:.3f}s')
        print(f'  data         = {total_bytes/1e6:.1f} MB')

        return parsed

    def _parse_ddr_results(self, packed_results, reads_per_shot, n_total_shots):
        """
        Parse DDR results using two threads:
          - Receiver thread: extracts raw bytes from RPC response, puts into FIFO
          - Parser thread: takes raw bytes from FIFO, parses DDR format, builds S11
        """
        # Normalize reads_per_shot to dict
        if isinstance(reads_per_shot, int):
            rps_dict = {f'Q{i}.rdlo': reads_per_shot for i in range(8)}
        else:
            rps_dict = reads_per_shot

        raw_queue = queue.Queue()
        parsed_results = [None] * len(packed_results)
        parser_error = [None]

        def receiver():
            """Extract raw bytes from RPC response and enqueue."""
            for i, packed in enumerate(packed_results):
                raw_bytes = packed.data if hasattr(packed, 'data') else packed
                raw_queue.put((i, raw_bytes))
            raw_queue.put(None)  # sentinel

        def parser():
            """Dequeue raw bytes, parse DDR format, store results."""
            try:
                while True:
                    item = raw_queue.get()
                    if item is None:
                        break
                    idx, raw_bytes = item
                    parsed_results[idx] = _parse_ddr_raw(raw_bytes, rps_dict, n_total_shots)
            except Exception as e:
                parser_error[0] = e

        recv_thread = threading.Thread(target=receiver)
        parse_thread = threading.Thread(target=parser)

        recv_thread.start()
        parse_thread.start()

        recv_thread.join()
        parse_thread.join()

        if parser_error[0] is not None:
            raise parser_error[0]

        return parsed_results

    def load_and_run_acq(self,
                         raw_asm_prog: Executable,
                         n_total_shots: int = 1,
                         nsamples: int = 8192,
                         acq_chans: Dict = {'0': 0, '1': 1},
                         trig_delay: float = 0,
                         decimator: int = 0,
                         return_acc: bool = False) -> tuple | Dict:
        """
        Load the program given by raw_asm_prog and acquire raw (or downconverted) adc traces.

        Parameters
        ----------
        raw_asm_prog: dict
            ASM binary to run. See load_circuit for details.
        n_total_shots: int
            number of shots to run. Program is restarted from the beginning
            for each new shot
        nsamples: int
            number of samples to read from the acq buffer
        acq_chans: dict
            current channel mapping is:

                '0': ADC_237_2 (main readout ADC)
                '1': ADC_237_0 (other ADC connected in gateware)
                TODO: figure out DLO channels, etc and what they mean
        trig_delay: float
            time to delay acquisition, relative to circuit start.
            NOTE: this value, when converted to units of clock cycles, is a
            16-bit value. So, it maxes out at CLK_PERIOD*(2**16) = 131.072e-6
        decimator: int
            decimation interval when sampling. e.g. 0 means full sample rate, 1
            means capture every other sample, 2 means capture every third sample, etc
        return_acc: bool
            if True, return a single acc (integrated + accumulated readout) value per shot,
            on each loaded channel. Default is False.

        Returns
        -------
        tuple | Dict
            - if `return_acc` is `False`:

                - dict:
                    array of acq samples for each channel in acq_chans with shape `(n_total_shots, nsamples)`

            - if `return_acc` is `True`:

                - tuple:
                    - dict:
                        array of acq samples for each channel in `acq_chans` with shape `(n_total_shots, nsamples)`
                    - dict:
                        array of acc values for each loaded channel with length `n_total_shots`

        """
        data = self.proxy.load_and_run_acq(raw_asm_prog.to_dict(), n_total_shots, nsamples, acq_chans, trig_delay, decimator, return_acc)

        if return_acc:
            acq_data = data[0]
            acc_data = data[1]
        else:
            acq_data = data
            acc_data = {}

        for ch in acq_data.keys():
            acq_data[ch] = np.reshape(np.frombuffer(acq_data[ch].data, dtype=np.int32), (n_total_shots, nsamples))
        for ch in acc_data.keys():
            acc_data[ch] = np.frombuffer(acc_data[ch].data, dtype=np.complex128)

        if return_acc:
            return acq_data, acc_data
        else:
            return acq_data

__init__(ip, port=9095, ddr_cmd=False)

ddr_cmd: set True when the target board runs the DDR command- streaming gateware (ff98e97e/633e80df lineage, batch_server on 8081) — run_circuit_batch then uses the hardware-sequenced DDR batch path. Default False keeps the classic BRAM/RPC behavior (sim, multiboard job server, older bitstreams).

Source code in qubic/rpc_client.py
356
357
358
359
360
361
362
363
364
def __init__(self, ip, port=9095, ddr_cmd=False):
    """ddr_cmd: set True when the target board runs the DDR command-
    streaming gateware (ff98e97e/633e80df lineage, batch_server on 8081) —
    run_circuit_batch then uses the hardware-sequenced DDR batch path.
    Default False keeps the classic BRAM/RPC behavior (sim, multiboard
    job server, older bitstreams)."""
    self._ip = ip
    self._ddr_cmd = bool(ddr_cmd)
    self.proxy = xmlrpc.client.ServerProxy('http://' + ip + ':' + str(port), allow_none=True)

load_and_run_acq(raw_asm_prog, n_total_shots=1, nsamples=8192, acq_chans={'0': 0, '1': 1}, trig_delay=0, decimator=0, return_acc=False)

Load the program given by raw_asm_prog and acquire raw (or downconverted) adc traces.

Parameters:

Name Type Description Default
raw_asm_prog Executable

ASM binary to run. See load_circuit for details.

required
n_total_shots int

number of shots to run. Program is restarted from the beginning for each new shot

1
nsamples int

number of samples to read from the acq buffer

8192
acq_chans Dict

current channel mapping is:

'0': ADC_237_2 (main readout ADC)
'1': ADC_237_0 (other ADC connected in gateware)
TODO: figure out DLO channels, etc and what they mean
{'0': 0, '1': 1}
trig_delay float

time to delay acquisition, relative to circuit start. NOTE: this value, when converted to units of clock cycles, is a 16-bit value. So, it maxes out at CLK_PERIOD(2*16) = 131.072e-6

0
decimator int

decimation interval when sampling. e.g. 0 means full sample rate, 1 means capture every other sample, 2 means capture every third sample, etc

0
return_acc bool

if True, return a single acc (integrated + accumulated readout) value per shot, on each loaded channel. Default is False.

False

Returns:

Type Description
tuple | Dict
  • if return_acc is False:

    • dict: array of acq samples for each channel in acq_chans with shape (n_total_shots, nsamples)
  • if return_acc is True:

    • tuple:
      • dict: array of acq samples for each channel in acq_chans with shape (n_total_shots, nsamples)
      • dict: array of acc values for each loaded channel with length n_total_shots
Source code in qubic/rpc_client.py
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
def load_and_run_acq(self,
                     raw_asm_prog: Executable,
                     n_total_shots: int = 1,
                     nsamples: int = 8192,
                     acq_chans: Dict = {'0': 0, '1': 1},
                     trig_delay: float = 0,
                     decimator: int = 0,
                     return_acc: bool = False) -> tuple | Dict:
    """
    Load the program given by raw_asm_prog and acquire raw (or downconverted) adc traces.

    Parameters
    ----------
    raw_asm_prog: dict
        ASM binary to run. See load_circuit for details.
    n_total_shots: int
        number of shots to run. Program is restarted from the beginning
        for each new shot
    nsamples: int
        number of samples to read from the acq buffer
    acq_chans: dict
        current channel mapping is:

            '0': ADC_237_2 (main readout ADC)
            '1': ADC_237_0 (other ADC connected in gateware)
            TODO: figure out DLO channels, etc and what they mean
    trig_delay: float
        time to delay acquisition, relative to circuit start.
        NOTE: this value, when converted to units of clock cycles, is a
        16-bit value. So, it maxes out at CLK_PERIOD*(2**16) = 131.072e-6
    decimator: int
        decimation interval when sampling. e.g. 0 means full sample rate, 1
        means capture every other sample, 2 means capture every third sample, etc
    return_acc: bool
        if True, return a single acc (integrated + accumulated readout) value per shot,
        on each loaded channel. Default is False.

    Returns
    -------
    tuple | Dict
        - if `return_acc` is `False`:

            - dict:
                array of acq samples for each channel in acq_chans with shape `(n_total_shots, nsamples)`

        - if `return_acc` is `True`:

            - tuple:
                - dict:
                    array of acq samples for each channel in `acq_chans` with shape `(n_total_shots, nsamples)`
                - dict:
                    array of acc values for each loaded channel with length `n_total_shots`

    """
    data = self.proxy.load_and_run_acq(raw_asm_prog.to_dict(), n_total_shots, nsamples, acq_chans, trig_delay, decimator, return_acc)

    if return_acc:
        acq_data = data[0]
        acc_data = data[1]
    else:
        acq_data = data
        acc_data = {}

    for ch in acq_data.keys():
        acq_data[ch] = np.reshape(np.frombuffer(acq_data[ch].data, dtype=np.int32), (n_total_shots, nsamples))
    for ch in acc_data.keys():
        acc_data[ch] = np.frombuffer(acc_data[ch].data, dtype=np.complex128)

    if return_acc:
        return acq_data, acc_data
    else:
        return acq_data

run_circuit_batch(executables, n_total_shots, reads_per_shot=None, timeout_per_shot=20, reload_cmd=True, reload_freq=True, reload_env=True, zero_between_reload=True, ddr=False, ddr_direct=True, group=4, wr_base=0, batch_timeout=600)

Runs a batch of circuits given by a list of compiled executables.

With CircuitRunnerClient(..., ddr_cmd=True) (ddr_streaming gateware, lineage 633e80df/ff98e97e): the batch is hardware-sequenced — commands stream over DDR c1 via batch_server (the BRAM command path is physically removed in this gateware), readout returns over DDR c0 via dma_server STRM, and ONE global_start pulse runs the whole batch. n_total_shots may be an int (same for every circuit) or a per-circuit list (the hardware config FIFO takes one n_shots per circuit). reads_per_shot may likewise be a per-circuit list (each element None = that circuit's result_channels metadata, or int/dict); a bare int/dict applies to every circuit. All circuits of a batch share the BRAM env/freq tables (checked; split batches if they differ). reload_freq/reload_env are honored: if the circuits' env/freq tables differ AND reloading is allowed, tables are reloaded before each circuit and the circuits run one at a time (NOT continuous — reported via last_batch_info); if reloading a differing kind is disabled, this raises. reload_cmd/ zero_between_reload/timeout_per_shot are ignored on this path; see group, wr_base, batch_timeout. CNR metadata (did the circuits run seamlessly back-to-back?) is printed and stored in self.last_batch_info. Without the constructor flag (default), the classic one-circuit-at-a-time BRAM flow runs (sim/multiboard/older bitstreams); ddr=True selects the June readout-only DDR flow; for these, n_total_shots must be an int. For the legacy paths: reads_per_shot and n_total_shots are passed directly into run_circuit, and must be the same for all circuits in the batch. The parameters reload_cmd, reload_freq, reload_env, and zero_between_reload control which of these fields is rewritten circuit-to-circuit (everything is rewritten initially). Leave these all at True (default) for maximum safety, to ensure that QubiC is in a clean state before each run. Depending on the circuits, some of these can be turned off to save time.

Parameters:

Name Type Description Default
executables List[Executable]

list of executables to run

required
n_total_shots int | List[int]

number of shots per circuit

required
reads_per_shot int | Dict

number of values per shot per channel to read back from accbuf. If dict, indexed by str(channel_number) (same indices as raw_asm_list). If int, assumed to be the same across channels, else can be a per-channel dict. Unless multiple circuits were rastered pre-compilation or there is mid-circuit measurement involved this is typically 1

None
timeout_per_shot float

job will time out if time to take a single shot exceeds this value in seconds (this likely means the job is hanging due to timing issues in the program or gateware)

20
reload_cmd bool

if True, reload command buffer between circuits

True
reload_freq bool

if True, reload freq buffer between circuits

True
reload_env bool

if True, reload env buffer between circuits

True
ddr bool

if True, use DDR memory readout instead of BRAM. Data is fetched automatically via socket after each circuit completes. Default is False.

False

Returns:

Type Description
List[Dict]:

Measurement results. A dictionary of results is returned for each circuit in the batch. This dict is keyed by ResultChannel names in the Executable; values are arraylike objects containing the measurement results for that channel. Result type is given by the dtype field in the ResultChannel object. These are always numpy-arraylike; a full listing of available result types is given in qubic.results.primitives. Results have shape (n_total_shots, reads_per_shot).

Putting this together, a returned collection of results might looks like: [{'Q0.rdlo': np.ndarray((n_shots, reads_per_shot)), 'Q1.rdlo': np.ndarray((n_shots, reads_per_shot))...}, {'Q0.rdlo': np.ndarray((n_shots, reads_per_shot)), ...}, ...]

Source code in qubic/rpc_client.py
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
def run_circuit_batch(self,
                      executables: List[Executable],
                      n_total_shots: int | List[int],
                      reads_per_shot: int | Dict = None,
                      timeout_per_shot: float = 20,
                      reload_cmd: bool = True,
                      reload_freq: bool = True,
                      reload_env: bool = True,
                      zero_between_reload: bool = True,
                      ddr: bool = False,
                      ddr_direct: bool = True,
                      group: int = 4,
                      wr_base: int = 0,
                      batch_timeout: float = 600) -> List[Dict[str, PrimitiveResult]]:
    """
    Runs a batch of circuits given by a list of compiled executables.

    With `CircuitRunnerClient(..., ddr_cmd=True)` (ddr_streaming gateware,
    lineage 633e80df/ff98e97e): the batch is hardware-sequenced — commands stream over DDR c1 via batch_server (the
    BRAM command path is physically removed in this gateware), readout
    returns over DDR c0 via dma_server STRM, and ONE global_start pulse runs
    the whole batch. `n_total_shots` may be an int (same for every circuit)
    or a per-circuit list (the hardware config FIFO takes one n_shots per
    circuit). `reads_per_shot` may likewise be a per-circuit list (each
    element None = that circuit's result_channels metadata, or int/dict);
    a bare int/dict applies to every circuit. All circuits of a batch share the BRAM env/freq tables (checked;
    split batches if they differ). `reload_freq`/`reload_env` are honored:
    if the circuits' env/freq tables differ AND reloading is allowed, tables
    are reloaded before each circuit and the circuits run one at a time
    (NOT continuous — reported via last_batch_info); if reloading a
    differing kind is disabled, this raises. `reload_cmd`/
    `zero_between_reload`/`timeout_per_shot` are ignored on this path; see
    `group`, `wr_base`, `batch_timeout`. CNR metadata (did the circuits run seamlessly
    back-to-back?) is printed and stored in `self.last_batch_info`.
    Without the constructor flag (default), the classic
    one-circuit-at-a-time BRAM flow runs (sim/multiboard/older bitstreams);
    `ddr=True` selects the June readout-only DDR flow; for these,
    `n_total_shots` must be an int. For the legacy
    paths: `reads_per_shot` and `n_total_shots` are passed directly into
    `run_circuit`, and must be the same for all circuits in the batch. The
    parameters `reload_cmd`, `reload_freq`, `reload_env`, and
    `zero_between_reload` control which of these fields is rewritten
    circuit-to-circuit (everything is rewritten initially). Leave these all
    at `True` (default) for maximum safety, to ensure that QubiC is in a
    clean state before each run. Depending on the circuits, some of these
    can be turned off to save time.

    Parameters
    ----------
    executables : List[Executable]
        list of executables to run
    n_total_shots: int
        number of shots per circuit
    reads_per_shot: int | Dict[str, int]
        number of values per shot per channel to read back from accbuf. If `dict`, indexed
        by `str(channel_number)` (same indices as `raw_asm_list`). If `int`, assumed to be
        the same across channels, else can be a per-channel dict. Unless multiple circuits
        were rastered pre-compilation or there is mid-circuit measurement involved this is typically 1
    timeout_per_shot: float
        job will time out if time to take a single shot exceeds this value in seconds
        (this likely means the job is hanging due to timing issues in the program or gateware)
    reload_cmd: bool
        if True, reload command buffer between circuits
    reload_freq: bool
        if True, reload freq buffer between circuits
    reload_env: bool
        if True, reload env buffer between circuits
    ddr: bool
        if True, use DDR memory readout instead of BRAM. Data is fetched
        automatically via socket after each circuit completes. Default is False.

    Returns
    -------
    List[Dict]:
        Measurement results. A dictionary of results is returned for each circuit in the batch.
        This dict is keyed by  `ResultChannel` names in the `Executable`; values are arraylike objects
        containing the measurement results for that channel. Result type is given by the  `dtype` field
        in the `ResultChannel` object. These are always numpy-arraylike; a full listing  of available result
        types is given in `qubic.results.primitives`. Results have shape `(n_total_shots, reads_per_shot)`.

        Putting this together, a returned collection of results might looks like:
            `[{'Q0.rdlo': np.ndarray((n_shots, reads_per_shot)), 'Q1.rdlo': np.ndarray((n_shots, reads_per_shot))...},
              {'Q0.rdlo': np.ndarray((n_shots, reads_per_shot)), ...}, ...]`

    """
    if ddr:
        if ddr_direct:
            return self._run_ddr_direct(executables, n_total_shots, reads_per_shot,
                                        reload_cmd, reload_freq, reload_env, zero_between_reload)
        else:
            return self._run_ddr_via_rpc(executables, n_total_shots, reads_per_shot,
                                         timeout_per_shot, reload_cmd, reload_freq,
                                         reload_env, zero_between_reload)

    if self._ddr_cmd:
        # DDR command-streaming gateware (633e80df/ff98e97e lineage): commands
        # can ONLY arrive over DDR c1 (the BRAM command path is physically
        # removed), so the batch is hardware-sequenced via batch_server +
        # dma_server STRM. n_total_shots may be a per-circuit list here.
        return self._run_ddr_cmd_batch(executables, n_total_shots, reads_per_shot,
                                       group=group, wr_base=wr_base,
                                       batch_timeout=batch_timeout,
                                       reload_freq=reload_freq,
                                       reload_env=reload_env)

    if not isinstance(n_total_shots, int):
        raise ValueError('the classic (non-ddr_cmd) path needs a single int n_total_shots')

    t_client_start = time.time()
    serialized_exe = [exec.to_dict() for exec in executables]
    t_serialize = time.time()
    packed_results = self.proxy.run_circuit_batch(serialized_exe, n_total_shots, reads_per_shot, float(timeout_per_shot),
                                 reload_cmd, reload_freq, reload_env, zero_between_reload, False)
    t_rpc = time.time()
    results = []
    for i, packed_result in enumerate(packed_results):
        results.append({ch: get_result_class(executables[i].result_channels[ch].dtype)(data.data, n_total_shots) for ch, data in packed_result.items()})
    t_parse = time.time()

    total_bytes = sum(len(data.data) for packed_result in packed_results for _, data in packed_result.items())
    print(f'[BRAM client timing]')
    print(f'  serialize      = {t_serialize - t_client_start:.3f}s')
    print(f'  XML-RPC call   = {t_rpc - t_serialize:.3f}s  (server processing + network transfer)')
    print(f'  client parse   = {t_parse - t_rpc:.3f}s')
    print(f'  TOTAL client   = {t_parse - t_client_start:.3f}s')
    print(f'  data received  = {total_bytes/1e6:.4f} MB')

    return results