Add Handbook.md benchmark tasks

This commit is contained in:
DerekSurge
2026-06-24 12:44:34 -07:00
commit 25c9eda5ca
1800 changed files with 323575 additions and 0 deletions
@@ -0,0 +1,7 @@
FROM handbook_base
COPY initial_external_services/ /data/
COPY initial_external_services/ /initial_data/
COPY initial_workspace/ /workdir/
ENV INPUTDIR=/data OUTPUTDIR=/data WORLDBENCH_TOOL_SETS="syntara_ds_all google_mail_contacts slack_core google_mail_core google_calendar_core jira_core shopify_core"
@@ -0,0 +1,249 @@
{
"mailbox": {
"email": "intake@careig.com",
"name": "CareIG Intake"
},
"contacts": [
{
"email": "mhamilton@careig.com",
"name": "Mia Hamilton"
},
{
"email": "pturner@careig.com",
"name": "Patrick Turner"
},
{
"email": "agibson@careig.com",
"name": "Aiden Gibson"
},
{
"email": "jcameron@careig.com",
"name": "Joanna Cameron"
},
{
"email": "echapman@careig.com",
"name": "Emma Chapman"
},
{
"email": "mjones@careig.com",
"name": "Maria Jones"
},
{
"email": "rbailey@careig.com",
"name": "Robert Bailey"
},
{
"email": "lmurphy@careig.com",
"name": "Liam Murphy"
},
{
"email": "pa@careig.com",
"name": "CareIG PA"
},
{
"email": "compliance@careig.com",
"name": "CareIG Compliance"
},
{
"email": "legal@careig.com",
"name": "CareIG Legal"
},
{
"email": "billing@careig.com",
"name": "CareIG Billing"
},
{
"email": "rliang@rochesterneuro.com",
"name": "Rebecca Liang"
},
{
"email": "pnair@rochesterneuro.com",
"name": "Priya Nair"
},
{
"email": "jmendes@buffaloheme.com",
"name": "Julie Mendes"
},
{
"email": "abriggs@buffaloheme.com",
"name": "Alan Briggs"
},
{
"email": "evasquez79@gmail.com",
"name": "Elena Vasquez"
},
{
"email": "mthompson65@yahoo.com",
"name": "Marcus Thompson"
},
{
"email": "npark@hudsonvalleyneuro.com",
"name": "Nancy Park"
},
{
"email": "rchoi@hudsonvalleyneuro.com",
"name": "Robert Choi"
},
{
"email": "syellen58@aol.com",
"name": "Sandra Yellen"
},
{
"email": "rwhitfield@whitfieldlaw.com",
"name": "Rachel Whitfield"
}
],
"folders": [],
"emails": [
{
"email_id": "1",
"folder": "INBOX",
"subject": "IVIG referral - Yellen, Sandra",
"from_addr": "npark@hudsonvalleyneuro.com",
"to_addr": "intake@careig.com",
"cc_addr": null,
"bcc_addr": null,
"date": "2026-09-18T09:02:00-04:00",
"message_id": "<msg001-hudsonvalleyneuro-20260918@hudsonvalleyneuro.com>",
"in_reply_to": null,
"body_text": "Hello,\n\nDr. Choi would like to refer the following patient for home IVIG infusion services.\n\nPatient: Sandra Yellen\nDOB: 11/03/1958\nAddress: 12 Crescent Ridge Dr, Poughkeepsie, NY 12601\nPhone: (845) 297-6183\nEmail: syellen58@aol.com\n\nReferring Physician: Robert Choi, MD\nNPI: 1639285047\nEmail: rchoi@hudsonvalleyneuro.com\nHudson Valley Neurology Associates\n\nDx: Guillain-Barré Syndrome (G61.0)\nNCS/EMG completed 09/10/2026 — results attached\n\nRx: Gamunex-C 10% 20g IV every 4 weeks\n\nInsurance: UnitedHealthcare PPO\nMember ID: UHC-337291054\nGroup: GRP-4102\nPayer phone: (800) 328-5979\n\nAllergies: Sulfa drugs. No known IVIG contraindications.\n\nPatient has been counseled on the referral and is expecting your call.\n\nThank you,\nNancy Park\nReferral Coordinator\nHudson Valley Neurology Associates\nnpark@hudsonvalleyneuro.com | (845) 471-2200",
"body_html": null,
"is_read": true,
"is_important": false,
"attachments": [
{
"filename": "ncs_emg_results_yellen.pdf",
"content_type": "application/pdf",
"content_base64": "JVBERi0xLjQKMSAwIG9iajw8L1R5cGUvQ2F0YWxvZy9QYWdlcyAyIDAgUj4+ZW5kb2JqCjIgMCBvYmo8PC9UeXBlL1BhZ2VzL0tpZHNbMyAwIFJdL0NvdW50IDE+PmVuZG9iagozIDAgb2JqPDwvVHlwZS9QYWdlL1BhcmVudCAyIDAgUi9NZWRpYUJveFswIDAgNjEyIDc5Ml0vQ29udGVudHMgNCAwIFIvUmVzb3VyY2VzPDwvRm9udDw8L0YxIDUgMCBSPj4+Pj4+ZW5kb2JqCjQgMCBvYmoKPDwvTGVuZ3RoIDIxOT4+CnN0cmVhbQpCVAovRjEgMTAgVGYKNzIgNzIwIFRkCihOQ1MvRU1HIFJlc3VsdHMgLSBZZWxsZW4sIFNhbmRyYSB8IERPQjogMTEvMDMvMTk1OCkgVGoKMCAtMjAgVGQKKERhdGU6IDA5LzEwLzIwMjYgfCBQaHlzaWNpYW46IERyLiBSb2JlcnQgQ2hvaSkgVGoKMCAtMjAgVGQKKER4OiBHdWlsbGFpbi1CYXJyZSBTeW5kcm9tZSAoRzYxLjApKSBUagowIC0yMCBUZAooW1BMQUNFSE9MREVSXSkgVGoKRVQKZW5kc3RyZWFtCmVuZG9iago1IDAgb2JqPDwvVHlwZS9Gb250L1N1YnR5cGUvVHlwZTEvQmFzZUZvbnQvSGVsdmV0aWNhPj5lbmRvYmoKeHJlZgowIDYKMDAwMDAwMDAwMCA2NTUzNSBmIAowMDAwMDAwMDA5IDAwMDAwIG4gCjAwMDAwMDAwNTggMDAwMDAgbiAKMDAwMDAwMDExNSAwMDAwMCBuIAowMDAwMDAwMjY2IDAwMDAwIG4gCjAwMDAwMDA3MDAgMDAwMDAgbiAKdHJhaWxlcjw8L1NpemUgNi9Sb290IDEgMCBSPj4Kc3RhcnR4cmVmCjc3NwolJUVPRg=="
}
]
},
{
"email_id": "2",
"folder": "Sent",
"subject": "CareIG — Consent Form Required | Yellen",
"from_addr": "intake@careig.com",
"to_addr": "syellen58@aol.com",
"cc_addr": null,
"bcc_addr": null,
"date": "2026-09-22T15:56:00-04:00",
"message_id": "<msg002-careig-20260922@careig.com>",
"in_reply_to": null,
"body_text": "Dear Sandra,\n\nThank you for speaking with us. Dr. Choi has referred you to CareIG Specialty Pharmacy for home IVIG infusion therapy.\n\nBefore we can proceed with verifying your insurance and scheduling your first visit, we need a signed Consent for Treatment and Financial Responsibility form. Please find the form attached.\n\nYou may sign and return it by:\n- Replying to this email with the signed form attached\n- Faxing to (888) 555-0147\n\nOnce we verify your insurance, we will also contact you with your estimated out-of-pocket costs before scheduling.\n\nIf you have any questions, please call us at (888) 555-0147.\n\nMia Hamilton\nCareIG Specialty Pharmacy, Intake\nintake@careig.com\n(888) 555-0147",
"body_html": null,
"is_read": true,
"is_important": false,
"attachments": [
{
"filename": "consent_form.pdf",
"content_type": "application/pdf",
"content_base64": "JVBERi0xLjQKMSAwIG9iajw8L1R5cGUvQ2F0YWxvZy9QYWdlcyAyIDAgUj4+ZW5kb2JqCjIgMCBvYmo8PC9UeXBlL1BhZ2VzL0tpZHNbMyAwIFJdL0NvdW50IDE+PmVuZG9iagozIDAgb2JqPDwvVHlwZS9QYWdlL1BhcmVudCAyIDAgUi9NZWRpYUJveFswIDAgNjEyIDc5Ml0vQ29udGVudHMgNCAwIFIvUmVzb3VyY2VzPDwvRm9udDw8L0YxIDUgMCBSPj4+Pj4+ZW5kb2JqCjQgMCBvYmoKPDwvTGVuZ3RoIDM3MD4+CnN0cmVhbQpCVAovRjEgMTQgVGYKNzIgNzIwIFRkCihDYXJlSUcgU3BlY2lhbHR5IFBoYXJtYWN5KSBUagovRjEgMTAgVGYKMCAtMjAgVGQKKE5QSTogMTIzNDU2Nzg5MyB8IEVJTjogNDctMzgyMTA1NikgVGoKMCAtMjAgVGQKKFBob25lOiAoODg4KSA1NTUtMDE0NyB8IEZheDogKDg4OCkgNTU1LTAxNDgpIFRqCjAgLTMwIFRkCi9GMSAxMiBUZgooQ09OU0VOVCBGT1IgVFJFQVRNRU5UIEFORCBGSU5BTkNJQUwgUkVTUE9OU0lCSUxJVFkpIFRqCjAgLTMwIFRkCi9GMSAxMCBUZgooQ2FyZUlHLUZPUk0tQ09OU0VOVC0wMDEgfCBSZXYuIDIwMjYtMDEpIFRqCjAgLTIwIFRkCihbUExBQ0VIT0xERVIgLSBGdWxsIGZvcm0gb24gZmlsZV0pIFRqCkVUCmVuZHN0cmVhbQplbmRvYmoKNSAwIG9iajw8L1R5cGUvRm9udC9TdWJ0eXBlL1R5cGUxL0Jhc2VGb250L0hlbHZldGljYT4+ZW5kb2JqCnhyZWYKMCA2CjAwMDAwMDAwMDAgNjU1MzUgZiAKMDAwMDAwMDAwOSAwMDAwMCBuIAowMDAwMDAwMDU4IDAwMDAwIG4gCjAwMDAwMDAxMTUgMDAwMDAgbiAKMDAwMDAwMDI2NiAwMDAwMCBuIAowMDAwMDAwNjM4IDAwMDAwIG4gCnRyYWlsZXI8PC9TaXplIDYvUm9vdCAxIDAgUj4+CnN0YXJ0eHJlZgo3MTUKJSVFT0Y="
}
]
},
{
"email_id": "3",
"folder": "INBOX",
"subject": "Referral — Vasquez, Elena | IVIG",
"from_addr": "rliang@rochesterneuro.com",
"to_addr": "intake@careig.com",
"cc_addr": null,
"bcc_addr": null,
"date": "2026-09-25T10:42:00-04:00",
"message_id": "<msg003-rochesterneuro-20260925@rochesterneuro.com>",
"in_reply_to": null,
"body_text": "Hi CareIG Intake,\n\nDr. Nair is referring the following patient for home IVIG therapy. Please find the details below.\n\nPatient: Elena Vasquez\nDOB: 08/14/1979\nAddress: 47 Birchwood Ln, Rochester, NY 14620\nPhone: (585) 331-4718\nEmail: evasquez79@gmail.com\n\nDiagnosis: Common Variable Immunodeficiency (D83.9). Most recent IgG level was 285 mg/dL drawn 02/20/2026. Dr. Nair's LMN is attached.\n\nRx: Gammaplex 10% 30g IV every 4 weeks\n\nInsurance: Aetna PPO\nMember ID: AET-882041573\nGroup #: GRP-2240\nPayer phone: (800) 624-0756\n\nPrescribing Physician: Priya Nair, MD\nNPI: 1528074392\nContact: pnair@rochesterneuro.com\nRochester Neurology & Immunology Associates\n\nNo known drug allergies or contraindications per chart review. Patient was seen in clinic on 09/24/2026 and is aware of the referral.\n\nPlease let us know if you need anything else to get started.\n\nThanks,\nRebecca Liang\nClinical Coordinator\nRochester Neurology & Immunology Associates\nrliang@rochesterneuro.com | (585) 271-3300",
"body_html": null,
"is_read": true,
"is_important": false,
"attachments": [
{
"filename": "lmn_vasquez.pdf",
"content_type": "application/pdf",
"content_base64": "JVBERi0xLjQKMSAwIG9iajw8L1R5cGUvQ2F0YWxvZy9QYWdlcyAyIDAgUj4+ZW5kb2JqCjIgMCBvYmo8PC9UeXBlL1BhZ2VzL0tpZHNbMyAwIFJdL0NvdW50IDE+PmVuZG9iagozIDAgb2JqPDwvVHlwZS9QYWdlL1BhcmVudCAyIDAgUi9NZWRpYUJveFswIDAgNjEyIDc5Ml0vQ29udGVudHMgNCAwIFIvUmVzb3VyY2VzPDwvRm9udDw8L0YxIDUgMCBSPj4+Pj4+ZW5kb2JqCjQgMCBvYmoKPDwvTGVuZ3RoIDIxMD4+CnN0cmVhbQpCVAovRjEgMTAgVGYKNzIgNzIwIFRkCihMTU4gLSBWYXNxdWV6LCBFbGVuYSB8IERPQjogMDgvMTQvMTk3OSkgVGoKMCAtMjAgVGQKKER4OiBEODMuOSB8IFJ4OiBHYW1tYXBsZXggMTAlIDMwZyBJViBxNHcpIFRqCjAgLTIwIFRkCihQaHlzaWNpYW46IFByaXlhIE5haXIsIE1EIHwgTlBJOiAxNTI4MDc0MzkyKSBUagowIC0yMCBUZAooW1BMQUNFSE9MREVSXSkgVGoKRVQKZW5kc3RyZWFtCmVuZG9iago1IDAgb2JqPDwvVHlwZS9Gb250L1N1YnR5cGUvVHlwZTEvQmFzZUZvbnQvSGVsdmV0aWNhPj5lbmRvYmoKeHJlZgowIDYKMDAwMDAwMDAwMCA2NTUzNSBmIAowMDAwMDAwMDA5IDAwMDAwIG4gCjAwMDAwMDAwNTggMDAwMDAgbiAKMDAwMDAwMDExNSAwMDAwMCBuIAowMDAwMDAwMjY2IDAwMDAwIG4gCjAwMDAwMDA3MDAgMDAwMDAgbiAKdHJhaWxlcjw8L1NpemUgNi9Sb290IDEgMCBSPj4Kc3RhcnR4cmVmCjc3NwolJUVPRg=="
}
]
},
{
"email_id": "4",
"folder": "INBOX",
"subject": "New pt referral — Thompson, Marcus / IVIG",
"from_addr": "jmendes@buffaloheme.com",
"to_addr": "intake@careig.com",
"cc_addr": null,
"bcc_addr": null,
"date": "2026-09-25T17:42:00-04:00",
"message_id": "<msg004-buffaloheme-20260925@buffaloheme.com>",
"in_reply_to": null,
"body_text": "Patient: Marcus Thompson, DOB 1965-03-22\nAddress: 88 Oakdale Ave, Buffalo, NY 14216\nPhone: (716) 482-9031, Email: mthompson65@yahoo.com\nPhysician: Dr. Alan Briggs, Buffalo Hematology Group, abriggs@buffaloheme.com\nNPI: NPI\nDiagnosis: D80.1 (Nonfamilial hypogammaglobulinemia)\nDrug: Privigen 10% IV, 25g, every 3 weeks\nInsurance: Empire BlueCross PPO, Member ID: EBC-441987320, Group: GRP-7815\nPayer Phone: (800) 553-9603\nContraindications: None",
"body_html": null,
"is_read": false,
"is_important": false,
"attachments": []
},
{
"email_id": "5",
"folder": "INBOX",
"subject": "Information confirmation",
"from_addr": "mthompson65@yahoo.com",
"to_addr": "intake@careig.com",
"cc_addr": null,
"bcc_addr": null,
"date": "2026-09-25T18:10:00-04:00",
"message_id": "<msg005-yahoo-20260925@yahoo.com>",
"in_reply_to": null,
"body_text": "Hi there,\nDr. Briggs' office told me they sent over my referral and that I'd need to confirm some of my information with you guys so here it is.\n\nMarcus Thompson\nDOB: 03/22/1965\n88 Oakdale Ave, Buffalo, NY 14216\nPhone: (716) 482-9031\nEmail: mthompson65@yahoo.com\n\nMy insurance is Empire BlueCross PPO\nMember ID: EBC-441987320\nGroup: GRP-7815\nPhone number on the back of the card: (800) 553-9603\n\nI know theres a consent form I need to sign so if you could send that to this email that would be great. I just want to make this process as fast and smooth as possible, I've been waiting so long to get this treatment started.\n\nThank you,\nMarcus",
"body_html": null,
"is_read": false,
"is_important": false,
"attachments": []
},
{
"email_id": "6",
"folder": "INBOX",
"subject": "Re: CareIG — Consent Form Required | Yellen",
"from_addr": "syellen58@aol.com",
"to_addr": "intake@careig.com",
"cc_addr": null,
"bcc_addr": null,
"date": "2026-09-26T08:42:00-04:00",
"message_id": "<msg006-aol-20260926@aol.com>",
"in_reply_to": "<msg002-careig-20260922@careig.com>",
"body_text": "Hi,\nAttached is the signed consent form. Please let me know if you need anything else from me.\nThank you,\nSandra",
"body_html": null,
"is_read": false,
"is_important": false,
"attachments": [
{
"filename": "signed_consent_form_yellen.pdf",
"content_type": "application/pdf",
"content_base64": "JVBERi0xLjQKMSAwIG9iajw8L1R5cGUvQ2F0YWxvZy9QYWdlcyAyIDAgUj4+ZW5kb2JqCjIgMCBvYmo8PC9UeXBlL1BhZ2VzL0tpZHNbMyAwIFJdL0NvdW50IDE+PmVuZG9iagozIDAgb2JqPDwvVHlwZS9QYWdlL1BhcmVudCAyIDAgUi9NZWRpYUJveFswIDAgNjEyIDc5Ml0vQ29udGVudHMgNCAwIFIvUmVzb3VyY2VzPDwvRm9udDw8L0YxIDUgMCBSPj4+Pj4+ZW5kb2JqCjQgMCBvYmoKPDwvTGVuZ3RoIDMwNT4+CnN0cmVhbQpCVAovRjEgMTIgVGYKNzIgNzIwIFRkCihDYXJlSUcgU3BlY2lhbHR5IFBoYXJtYWN5KSBUagovRjEgMTAgVGYKMCAtMjAgVGQKKENPTlNFTlQgRk9SIFRSRUFUTUVOVCBBTkQgRklOQU5DSUFMIFJFU1BPTlNJQklMSVRZKSBUagowIC0yMCBUZAooUGF0aWVudDogU2FuZHJhIFllbGxlbiB8IERPQjogMTEvMDMvMTk1OCkgVGoKMCAtMjAgVGQKKFNpZ25lZDogU2FuZHJhIFllbGxlbiB8IERhdGU6IDA5LzI2LzIwMjYpIFRqCjAgLTIwIFRkCihDYXJlSUctRk9STS1DT05TRU5ULTAwMSB8IFJldi4gMjAyNi0wMSB8IFtTSUdORURdKSBUagpFVAplbmRzdHJlYW0KZW5kb2JqCjUgMCBvYmo8PC9UeXBlL0ZvbnQvU3VidHlwZS9UeXBlMS9CYXNlRm9udC9IZWx2ZXRpY2E+PmVuZG9iagp4cmVmCjAgNgowMDAwMDAwMDAwIDY1NTM1IGYgCjAwMDAwMDAwMDkgMDAwMDAgbiAKMDAwMDAwMDA1OCAwMDAwMCBuIAowMDAwMDAwMTE1IDAwMDAwIG4gCjAwMDAwMDAyNjYgMDAwMDAgbiAKMDAwMDAwMDcwMCAwMDAwMCBuIAp0cmFpbGVyPDwvU2l6ZSA2L1Jvb3QgMSAwIFI+PgpzdGFydHhyZWYKNzc3CiUlRU9G"
}
]
},
{
"email_id": "7",
"folder": "INBOX",
"subject": "Records Request — Sandra Yellen DOB 11/03/1958",
"from_addr": "rwhitfield@whitfieldlaw.com",
"to_addr": "intake@careig.com",
"cc_addr": null,
"bcc_addr": null,
"date": "2026-09-28T07:00:00-04:00",
"message_id": "<msg007-whitfieldlaw-20260928@whitfieldlaw.com>",
"in_reply_to": null,
"body_text": "To Whom It May Concern,\n\nThis office represents Sandra Yellen (DOB: 11/03/1958) in a pending medical legal violation matter. We are writing to request copies of any and all records related to Ms. Yellen's care at CareIG Specialty Pharmacy, including but not limited to:\n\n- Referral documentation and correspondence\n- Intake records\n- Prior authorization submissions and correspondence\n\nAdditionally, please confirm the name and dosage of any medications prescribed or administered to Ms. Yellen, and the identity of the referring physician.\n\nThis request is made pursuant to the patient's written authorization, which is enclosed. Please direct all responsive documents to the undersigned at the address below within 30 days.\n\nIf you have any questions regarding this request, please do not hesitate to contact me directly.\n\nRegards,\nRachel Whitfield, Esq.\nWhitfield & Associates, LLP\n340 Main Street, Suite 700\nPoughkeepsie, NY 12601\nrwhitfield@whitfieldlaw.com\n(845) 905-3310",
"body_html": null,
"is_read": false,
"is_important": false,
"attachments": [
{
"filename": "patient_authorization_yellen.pdf",
"content_type": "application/pdf",
"content_base64": "JVBERi0xLjQKMSAwIG9iajw8L1R5cGUvQ2F0YWxvZy9QYWdlcyAyIDAgUj4+ZW5kb2JqCjIgMCBvYmo8PC9UeXBlL1BhZ2VzL0tpZHNbMyAwIFJdL0NvdW50IDE+PmVuZG9iagozIDAgb2JqPDwvVHlwZS9QYWdlL1BhcmVudCAyIDAgUi9NZWRpYUJveFswIDAgNjEyIDc5Ml0vQ29udGVudHMgNCAwIFIvUmVzb3VyY2VzPDwvRm9udDw8L0YxIDUgMCBSPj4+Pj4+ZW5kb2JqCjQgMCBvYmoKPDwvTGVuZ3RoIDMyMj4+CnN0cmVhbQpCVAovRjEgMTIgVGYKNzIgNzIwIFRkCihXaGl0ZmllbGQgJiBBc3NvY2lhdGVzLCBMTFApIFRqCi9GMSAxMCBUZgowIC0yMCBUZAooUEFUSUVOVCBBVVRIT1JJWkFUSU9OIEZPUiBSRUxFQVNFIE9GIFJFQ09SRFMpIFRqCjAgLTIwIFRkCihQYXRpZW50OiBTYW5kcmEgWWVsbGVuIHwgRE9COiAxMS8wMy8xOTU4KSBUagowIC0yMCBUZAooQXV0aG9yaXplZCBUbzogV2hpdGZpZWxkICYgQXNzb2NpYXRlcywgTExQKSBUagowIC0yMCBUZAooU2lnbmVkOiBTYW5kcmEgWWVsbGVuIHwgRGF0ZTogMDkvMjcvMjAyNikgVGoKMCAtMjAgVGQKKFtQTEFDRUhPTERFUl0pIFRqCkVUCmVuZHN0cmVhbQplbmRvYmoKNSAwIG9iajw8L1R5cGUvRm9udC9TdWJ0eXBlL1R5cGUxL0Jhc2VGb250L0hlbHZldGljYT4+ZW5kb2JqCnhyZWYKMCA2CjAwMDAwMDAwMDAgNjU1MzUgZiAKMDAwMDAwMDAwOSAwMDAwMCBuIAowMDAwMDAwMDU4IDAwMDAwIG4gCjAwMDAwMDAxMTUgMDAwMDAgbiAKMDAwMDAwMDI2NiAwMDAwMCBuIAowMDAwMDAwNzAwIDAwMDAwIG4gCnRyYWlsZXI8PC9TaXplIDYvUm9vdCAxIDAgUj4+CnN0YXJ0eHJlZgo3NzcKJSVFT0Y="
}
]
}
],
"next_email_id": 8
}
@@ -0,0 +1,427 @@
{
"users": {
"U001": {
"id": "U001",
"team_id": "T001",
"name": "mhamilton",
"deleted": false,
"real_name": "Mia Hamilton",
"profile": {
"real_name": "Mia Hamilton",
"display_name": "mhamilton",
"email": "mhamilton@careig.com",
"title": "Patient Representative"
},
"is_admin": false,
"is_owner": false,
"is_bot": false
},
"U002": {
"id": "U002",
"team_id": "T001",
"name": "pturner",
"deleted": false,
"real_name": "Patrick Turner",
"profile": {
"real_name": "Patrick Turner",
"display_name": "pturner",
"email": "pturner@careig.com",
"title": "Benefits Verification Specialist"
},
"is_admin": false,
"is_owner": false,
"is_bot": false
},
"U003": {
"id": "U003",
"team_id": "T001",
"name": "agibson",
"deleted": false,
"real_name": "Aiden Gibson",
"profile": {
"real_name": "Aiden Gibson",
"display_name": "agibson",
"email": "agibson@careig.com",
"title": "Clinical Coordinator"
},
"is_admin": false,
"is_owner": false,
"is_bot": false
},
"U004": {
"id": "U004",
"team_id": "T001",
"name": "jcameron",
"deleted": false,
"real_name": "Joanna Cameron",
"profile": {
"real_name": "Joanna Cameron",
"display_name": "jcameron",
"email": "jcameron@careig.com",
"title": "Infusion Nurse"
},
"is_admin": false,
"is_owner": false,
"is_bot": false
},
"U005": {
"id": "U005",
"team_id": "T001",
"name": "echapman",
"deleted": false,
"real_name": "Emma Chapman",
"profile": {
"real_name": "Emma Chapman",
"display_name": "echapman",
"email": "echapman@careig.com",
"title": "Billing Specialist"
},
"is_admin": false,
"is_owner": false,
"is_bot": false
},
"U006": {
"id": "U006",
"team_id": "T001",
"name": "mjones",
"deleted": false,
"real_name": "Maria Jones",
"profile": {
"real_name": "Maria Jones",
"display_name": "mjones",
"email": "mjones@careig.com",
"title": "AR Specialist"
},
"is_admin": false,
"is_owner": false,
"is_bot": false
},
"U007": {
"id": "U007",
"team_id": "T001",
"name": "rbailey",
"deleted": false,
"real_name": "Robert Bailey",
"profile": {
"real_name": "Robert Bailey",
"display_name": "rbailey",
"email": "rbailey@careig.com",
"title": "Director of Revenue Cycle"
},
"is_admin": true,
"is_owner": false,
"is_bot": false
},
"U008": {
"id": "U008",
"team_id": "T001",
"name": "lmurphy",
"deleted": false,
"real_name": "Liam Murphy",
"profile": {
"real_name": "Liam Murphy",
"display_name": "lmurphy",
"email": "lmurphy@careig.com",
"title": "Director of Operations"
},
"is_admin": true,
"is_owner": false,
"is_bot": false
}
},
"channels": {
"C001": {
"id": "C001",
"name": "intake-team",
"is_channel": true,
"is_group": false,
"is_im": false,
"is_mpim": false,
"is_private": false,
"created": 1640995200,
"is_archived": false,
"is_general": false,
"unlinked": 0,
"name_normalized": "intake-team",
"is_shared": false,
"is_org_shared": false,
"is_ext_shared": false,
"is_pending_ext_shared": false,
"pending_shared": [],
"pending_connected_team_ids": [],
"context_team_id": "T001",
"updated": 1640995200,
"creator": "U007",
"shared_team_ids": ["T001"],
"is_member": true,
"topic": {
"value": "Intake workflow notifications",
"creator": "U007",
"last_set": 1640995200
},
"purpose": {
"value": "Intake workflow notifications",
"creator": "U007",
"last_set": 1640995200
}
},
"C002": {
"id": "C002",
"name": "intake-alerts",
"is_channel": true,
"is_group": false,
"is_im": false,
"is_mpim": false,
"is_private": false,
"created": 1640995200,
"is_archived": false,
"is_general": false,
"unlinked": 0,
"name_normalized": "intake-alerts",
"is_shared": false,
"is_org_shared": false,
"is_ext_shared": false,
"is_pending_ext_shared": false,
"pending_shared": [],
"pending_connected_team_ids": [],
"context_team_id": "T001",
"updated": 1640995200,
"creator": "U007",
"shared_team_ids": ["T001"],
"is_member": true,
"topic": {
"value": "Intake hold alerts",
"creator": "U007",
"last_set": 1640995200
},
"purpose": {
"value": "Intake hold alerts",
"creator": "U007",
"last_set": 1640995200
}
},
"C003": {
"id": "C003",
"name": "intake-mgmt",
"is_channel": true,
"is_group": false,
"is_im": false,
"is_mpim": false,
"is_private": false,
"created": 1640995200,
"is_archived": false,
"is_general": false,
"unlinked": 0,
"name_normalized": "intake-mgmt",
"is_shared": false,
"is_org_shared": false,
"is_ext_shared": false,
"is_pending_ext_shared": false,
"pending_shared": [],
"pending_connected_team_ids": [],
"context_team_id": "T001",
"updated": 1640995200,
"creator": "U007",
"shared_team_ids": ["T001"],
"is_member": true,
"topic": {
"value": "Intake management escalations",
"creator": "U007",
"last_set": 1640995200
},
"purpose": {
"value": "Intake management escalations",
"creator": "U007",
"last_set": 1640995200
}
},
"C004": {
"id": "C004",
"name": "clinical-scheduling",
"is_channel": true,
"is_group": false,
"is_im": false,
"is_mpim": false,
"is_private": false,
"created": 1640995200,
"is_archived": false,
"is_general": false,
"unlinked": 0,
"name_normalized": "clinical-scheduling",
"is_shared": false,
"is_org_shared": false,
"is_ext_shared": false,
"is_pending_ext_shared": false,
"pending_shared": [],
"pending_connected_team_ids": [],
"context_team_id": "T001",
"updated": 1640995200,
"creator": "U007",
"shared_team_ids": ["T001"],
"is_member": true,
"topic": {
"value": "Clinical workflow",
"creator": "U007",
"last_set": 1640995200
},
"purpose": {
"value": "Clinical workflow",
"creator": "U007",
"last_set": 1640995200
}
},
"C005": {
"id": "C005",
"name": "billing",
"is_channel": true,
"is_group": false,
"is_im": false,
"is_mpim": false,
"is_private": false,
"created": 1640995200,
"is_archived": false,
"is_general": false,
"unlinked": 0,
"name_normalized": "billing",
"is_shared": false,
"is_org_shared": false,
"is_ext_shared": false,
"is_pending_ext_shared": false,
"pending_shared": [],
"pending_connected_team_ids": [],
"context_team_id": "T001",
"updated": 1640995200,
"creator": "U007",
"shared_team_ids": ["T001"],
"is_member": true,
"topic": {
"value": "Billing workflow",
"creator": "U007",
"last_set": 1640995200
},
"purpose": {
"value": "Billing workflow",
"creator": "U007",
"last_set": 1640995200
}
},
"C006": {
"id": "C006",
"name": "billing-appeals",
"is_channel": true,
"is_group": false,
"is_im": false,
"is_mpim": false,
"is_private": false,
"created": 1640995200,
"is_archived": false,
"is_general": false,
"unlinked": 0,
"name_normalized": "billing-appeals",
"is_shared": false,
"is_org_shared": false,
"is_ext_shared": false,
"is_pending_ext_shared": false,
"pending_shared": [],
"pending_connected_team_ids": [],
"context_team_id": "T001",
"updated": 1640995200,
"creator": "U007",
"shared_team_ids": ["T001"],
"is_member": true,
"topic": {
"value": "Denial and appeal notifications",
"creator": "U007",
"last_set": 1640995200
},
"purpose": {
"value": "Denial and appeal notifications",
"creator": "U007",
"last_set": 1640995200
}
},
"C007": {
"id": "C007",
"name": "billing-mgmt",
"is_channel": true,
"is_group": false,
"is_im": false,
"is_mpim": false,
"is_private": false,
"created": 1640995200,
"is_archived": false,
"is_general": false,
"unlinked": 0,
"name_normalized": "billing-mgmt",
"is_shared": false,
"is_org_shared": false,
"is_ext_shared": false,
"is_pending_ext_shared": false,
"pending_shared": [],
"pending_connected_team_ids": [],
"context_team_id": "T001",
"updated": 1640995200,
"creator": "U007",
"shared_team_ids": ["T001"],
"is_member": true,
"topic": {
"value": "Billing management escalations",
"creator": "U007",
"last_set": 1640995200
},
"purpose": {
"value": "Billing management escalations",
"creator": "U007",
"last_set": 1640995200
}
},
"C008": {
"id": "C008",
"name": "compliance",
"is_channel": true,
"is_group": false,
"is_im": false,
"is_mpim": false,
"is_private": false,
"created": 1640995200,
"is_archived": false,
"is_general": false,
"unlinked": 0,
"name_normalized": "compliance",
"is_shared": false,
"is_org_shared": false,
"is_ext_shared": false,
"is_pending_ext_shared": false,
"pending_shared": [],
"pending_connected_team_ids": [],
"context_team_id": "T001",
"updated": 1640995200,
"creator": "U007",
"shared_team_ids": ["T001"],
"is_member": true,
"topic": {
"value": "Compliance alerts",
"creator": "U007",
"last_set": 1640995200
},
"purpose": {
"value": "Compliance alerts",
"creator": "U007",
"last_set": 1640995200
}
}
},
"messages": {
"C001": [],
"C002": [],
"C003": [],
"C004": [],
"C005": [],
"C006": [],
"C007": [],
"C008": []
},
"counters": {
"messageId": 1000000
}
}
@@ -0,0 +1 @@
It's Mia Hamilton and we have some intakes I want to finish, I dont want patients waiting any more than they need to. Please follow the SOP and take care of them for me, the sooner they can get their treatment the better.
@@ -0,0 +1 @@
Today's date is, September, 28, 2026, 08:15:00-04:00. You are an office assistant with access to the user's filesystem (/workdir), email (google_mail_*), slack (slack_*), calendar (google_calendar_*), jira (jira_*), and shopify (shopify_*) tools. The workspace directory /workdir contains files for the task. Do NOT ask the user for more information — use your tools to discover what you need. Begin by listing /workdir and reading relevant files.
@@ -0,0 +1,23 @@
schema_version = "1.1"
[task]
name = "sop-tasks/medical_careig_specialty_pharmacy_ea622238"
description = "SOP task from CSV row ea622238"
[agent]
timeout_sec = 3600.0
[verifier]
timeout_sec = 3600.0
[environment]
os = "linux"
cpus = 2
memory_mb = 4096
env = { INPUTDIR = "/data", OUTPUTDIR = "/data", WORLDBENCH_TOOL_SETS = "syntara_ds_all google_mail_contacts slack_core google_mail_core google_calendar_core jira_core shopify_core" }
[[environment.mcp_servers]]
name = "mcp-proxy"
transport = "streamable-http"
url = "http://localhost:8000/mcp"
@@ -0,0 +1,100 @@
[
{
"id": "f75d80aa-8e60-416c-bc22-5117609b365d",
"sort_order": 0,
"rubric_text": "In `mailbox.json` (external services), there must be exactly 4 or 5 emails in the Sent folder, 6 emails in INBOX, 0 in Drafts, and 0 in Trash",
"verifier_code": "from pathlib import Path\nimport json\n\n\ndef _has_addr(email: dict, addr: str) -> bool:\n \"\"\"Check if addr appears in any recipient field of the email.\"\"\"\n for field in (\"to_addr\", \"cc_addr\", \"bcc_addr\"):\n val = email.get(field) or \"\"\n if addr.lower() in val.lower():\n return True\n return False\n\n\ndef verify(workspace_path, external_services_path=None):\n if external_services_path is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"external_services_path is None; cannot check mailbox.json\"}\n\n mailbox_path = Path(external_services_path) / \"mailbox.json\"\n if not mailbox_path.exists():\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"mailbox.json not found at external_services_path\"}\n\n with open(mailbox_path, \"r\") as f:\n data = json.load(f)\n\n emails = data.get(\"emails\", [])\n sent = [e for e in emails if e.get(\"folder\", \"\").lower() == \"sent\"]\n inbox = [e for e in emails if e.get(\"folder\", \"\").lower() == \"inbox\"]\n drafts = [e for e in emails if e.get(\"folder\", \"\").lower() == \"drafts\"]\n trash = [e for e in emails if e.get(\"folder\", \"\").lower() == \"trash\"]\n\n issues = []\n\n # --- Fixed folder counts ---\n if len(inbox) != 6:\n issues.append(f\"INBOX has {len(inbox)} emails, expected 6\")\n if len(drafts) != 0:\n issues.append(f\"Drafts has {len(drafts)} emails, expected 0\")\n if len(trash) != 0:\n issues.append(f\"Trash has {len(trash)} emails, expected 0\")\n\n # --- Required sent emails ---\n # 1. Vasquez consent form\n vasquez_consent = [\n e for e in sent\n if \"evasquez79@gmail.com\" in (e.get(\"to_addr\") or \"\").lower()\n and \"consent form required\" in (e.get(\"subject\") or \"\").lower()\n and \"vasquez\" in (e.get(\"subject\") or \"\").lower()\n ]\n if not vasquez_consent:\n issues.append(\n \"No sent email to evasquez79@gmail.com with subject matching \"\n \"'Consent Form Required | Vasquez'\"\n )\n\n # 2. Thompson incomplete referral\n thompson_referral = [\n e for e in sent\n if \"abriggs@buffaloheme.com\" in (e.get(\"to_addr\") or \"\").lower()\n and \"incomplete referral\" in (e.get(\"subject\") or \"\").lower()\n and \"thompson\" in (e.get(\"subject\") or \"\").lower()\n ]\n if not thompson_referral:\n issues.append(\n \"No sent email to abriggs@buffaloheme.com with subject matching \"\n \"'Incomplete Referral | Thompson'\"\n )\n\n # 3. Legal forward — both recipients must appear somewhere across sent emails\n # Acceptable as one email (both in to/cc) or two separate emails\n legal_reached = any(_has_addr(e, \"legal@careig.com\") for e in sent)\n lmurphy_reached = any(_has_addr(e, \"lmurphy@careig.com\") for e in sent)\n if not legal_reached:\n issues.append(\n \"legal@careig.com not found as a recipient in any sent email \"\n \"(email_id 7 forward required per SOP §12.1)\"\n )\n if not lmurphy_reached:\n issues.append(\n \"lmurphy@careig.com not found as a recipient in any sent email \"\n \"(email_id 7 forward required per SOP §12.1)\"\n )\n\n # --- Sent count: 4 (one forward) or 5 (two separate forwards) ---\n if len(sent) not in (4, 5):\n issues.append(f\"Sent folder has {len(sent)} emails, expected 4 or 5\")\n\n if issues:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"; \".join(issues)}\n\n return {\n \"pass\": True,\n \"score\": 1.0,\n \"feedback\": (\n f\"Folder counts correct: Sent={len(sent)}, INBOX={len(inbox)}, \"\n f\"Drafts={len(drafts)}, Trash={len(trash)}. \"\n \"All required sent emails present.\"\n ),\n }\n",
"criterion_type": "expected_output"
},
{
"id": "f83bcce7-d89c-4273-804c-49f60991b6c3",
"sort_order": 1,
"rubric_text": "In `mailbox.json`, there must be a sent email to evasquez79@gmail.com with subject containing 'Consent Form Required' and 'Vasquez', from intake@careig.com, with attachment consent_form.pdf, and body containing key phrases about signing the consent form",
"verifier_code": "from pathlib import Path\nimport json\nimport re\n\n\ndef _normalize(text: str) -> str:\n \"\"\"Collapse whitespace variants and normalize dashes for exact-phrase matching.\"\"\"\n text = text.replace(\"\\u2014\", \"-\").replace(\"\\u2013\", \"-\").replace(\"\\u2212\", \"-\")\n text = re.sub(r\"\\s+\", \" \", text)\n return text.strip().lower()\n\n\ndef verify(workspace_path, external_services_path=None):\n if external_services_path is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"external_services_path is None; cannot check mailbox.json\"}\n\n mailbox_path = Path(external_services_path) / \"mailbox.json\"\n if not mailbox_path.exists():\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"mailbox.json not found at external_services_path\"}\n\n with open(mailbox_path, \"r\") as f:\n data = json.load(f)\n\n emails = data.get(\"emails\", [])\n sent = [e for e in emails if e.get(\"folder\", \"\").lower() == \"sent\"]\n\n target = next(\n (e for e in sent if \"evasquez79@gmail.com\" in (e.get(\"to_addr\") or \"\").lower()),\n None,\n )\n if target is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"No sent email found to evasquez79@gmail.com\"}\n\n checks = {} # key -> (passed: bool, detail: str)\n\n # Subject — exact normalized match\n subject = target.get(\"subject\", \"\")\n checks[\"subject\"] = (\n \"careig - consent form required | vasquez\" in _normalize(subject),\n f\"Subject: '{subject}'\",\n )\n\n # From address\n from_addr = target.get(\"from_addr\", \"\")\n checks[\"from_addr\"] = (\n \"intake@careig.com\" in from_addr.lower(),\n f\"From: '{from_addr}'\",\n )\n\n # Attachment — consent_form.pdf\n filenames = [a.get(\"filename\", \"\") for a in target.get(\"attachments\", [])]\n checks[\"attachment_consent_form\"] = (\n any(\"consent_form.pdf\" in fn.lower() for fn in filenames),\n f\"Attachments: {filenames}\",\n )\n\n # Body phrases — exact strings from the template, normalized\n body_norm = _normalize(target.get(\"body_text\", \"\"))\n body_phrases = {\n \"body_greeting\": \"dear elena,\",\n \"body_thank_you\": \"thank you for speaking with us.\",\n \"body_referral\": \"dr. nair has referred you to careig specialty pharmacy for home ivig infusion therapy.\",\n \"body_consent_form\": \"signed consent for treatment and financial responsibility form\",\n \"body_fax_number\": \"faxing to (888) 555-0147\",\n \"body_oop_costs\": \"your estimated out-of-pocket costs before scheduling\",\n \"body_signature_name\": \"mia hamilton\",\n \"body_signature_title\": \"careig specialty pharmacy, intake\",\n \"body_signature_email\": \"intake@careig.com\",\n }\n for key, phrase in body_phrases.items():\n checks[key] = (phrase in body_norm, f\"Phrase not found: '{phrase}'\")\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n failed = [k for k, (ok, _) in checks.items() if not ok]\n total = len(checks)\n score = len(passed) / total\n\n if failed:\n details = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (_, detail) in checks.items() if k in failed)\n feedback = details\n else:\n feedback = f\"All {total} checks passed for Vasquez consent email.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "expected_output"
},
{
"id": "cce481bc-d02a-443b-b556-902a25b3101b",
"sort_order": 2,
"rubric_text": "In `mailbox.json`, there must be a sent email to abriggs@buffaloheme.com with subject containing 'Incomplete Referral' and 'Thompson', from intake@careig.com, and body mentioning Physician NPI and Marcus Thompson",
"verifier_code": "from pathlib import Path\nimport json\nimport re\n\n\ndef _normalize(text: str) -> str:\n text = text.replace(\"\\u2014\", \"-\").replace(\"\\u2013\", \"-\").replace(\"\\u2212\", \"-\")\n text = re.sub(r\"\\s+\", \" \", text)\n return text.strip().lower()\n\n\ndef verify(workspace_path, external_services_path=None):\n if external_services_path is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"external_services_path is None; cannot check mailbox.json\"}\n\n mailbox_path = Path(external_services_path) / \"mailbox.json\"\n if not mailbox_path.exists():\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"mailbox.json not found at external_services_path\"}\n\n with open(mailbox_path, \"r\") as f:\n data = json.load(f)\n\n emails = data.get(\"emails\", [])\n sent = [e for e in emails if e.get(\"folder\", \"\").lower() == \"sent\"]\n\n target = next(\n (e for e in sent if \"abriggs@buffaloheme.com\" in (e.get(\"to_addr\") or \"\").lower()),\n None,\n )\n if target is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"No sent email found to abriggs@buffaloheme.com\"}\n\n checks = {}\n\n # Subject\n subject = target.get(\"subject\", \"\")\n checks[\"subject\"] = (\n \"incomplete referral | thompson\" in _normalize(subject),\n f\"Subject: '{subject}'\",\n )\n\n # From address\n from_addr = target.get(\"from_addr\", \"\")\n checks[\"from_addr\"] = (\n \"intake@careig.com\" in from_addr.lower(),\n f\"From: '{from_addr}'\",\n )\n\n # Body phrases — exact strings from the template, normalized\n body_norm = _normalize(target.get(\"body_text\", \"\"))\n body_phrases = {\n \"body_greeting\": \"dear dr. briggs,\",\n \"body_referral\": \"we received a referral for marcus thompson (dob: 1965-03-22) for home ivig infusion therapy.\",\n \"body_thank_you\": \"thank you for the referral.\",\n \"body_missing_item\": \"physician npi\",\n \"body_reply_request\": \"please reply to this email or contact us at your earliest convenience.\",\n \"body_phone\": \"(888) 555-0147\",\n \"body_signature_name\": \"mia hamilton\",\n \"body_signature_title\": \"careig specialty pharmacy, intake\",\n \"body_signature_email\": \"intake@careig.com\",\n }\n for key, phrase in body_phrases.items():\n checks[key] = (phrase in body_norm, f\"Phrase not found: '{phrase}'\")\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n failed = [k for k, (ok, _) in checks.items() if not ok]\n total = len(checks)\n score = len(passed) / total\n\n if failed:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (_, detail) in checks.items() if k in failed)\n else:\n feedback = f\"All {total} checks passed for Thompson incomplete referral email.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "expected_output"
},
{
"id": "78caecc4-f2b7-4c1b-a527-68d2d3f3f135",
"sort_order": 3,
"rubric_text": "In `mailbox.json`, email_id '7' must be forwarded to both legal@careig.com and lmurphy@careig.com",
"verifier_code": "from pathlib import Path\nimport json\nimport re\n\n\ndef _has_addr(email: dict, addr: str) -> bool:\n for field in (\"to_addr\", \"cc_addr\", \"bcc_addr\"):\n val = email.get(field) or \"\"\n if addr.lower() in val.lower():\n return True\n return False\n\n\ndef _is_forward_of(email: dict, msg_id: str) -> bool:\n \"\"\"True if sent email is a forward: in_reply_to matches msg_id, or subject has Fwd:/FW:.\"\"\"\n in_reply_to = email.get(\"in_reply_to\") or \"\"\n if msg_id and msg_id in in_reply_to:\n return True\n subject = email.get(\"subject\") or \"\"\n return bool(re.search(r\"\\b(fwd?:|fw:)\", subject, re.IGNORECASE))\n\n\ndef verify(workspace_path, external_services_path=None):\n if external_services_path is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"external_services_path is None; cannot check mailbox.json\"}\n\n mailbox_path = Path(external_services_path) / \"mailbox.json\"\n if not mailbox_path.exists():\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"mailbox.json not found at external_services_path\"}\n\n with open(mailbox_path, \"r\") as f:\n data = json.load(f)\n\n emails = data.get(\"emails\", [])\n\n # email_id 7 must exist — hard prerequisite, can't score forward without it\n email7 = next((e for e in emails if str(e.get(\"email_id\", \"\")) == \"7\"), None)\n if email7 is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"email_id '7' not found in mailbox\"}\n\n msg_id_7 = email7.get(\"message_id\", \"\")\n sent = [e for e in emails if e.get(\"folder\", \"\").lower() == \"sent\"]\n forwards = [e for e in sent if _is_forward_of(e, msg_id_7)]\n\n checks = {}\n\n # 1. At least one sent email is identified as a forward of email_id 7\n checks[\"forward_identified\"] = (\n len(forwards) > 0,\n \"No sent email identified as a forward of email_id 7 \"\n \"(checked: in_reply_to matches message_id, or subject contains Fwd:/FW:)\",\n )\n\n # 2 & 3. Both required recipients appear somewhere in the forward email(s)\n checks[\"legal_reached\"] = (\n any(_has_addr(e, \"legal@careig.com\") for e in forwards),\n \"legal@careig.com not found as recipient (to/cc/bcc) in any forward of email_id 7\",\n )\n checks[\"lmurphy_reached\"] = (\n any(_has_addr(e, \"lmurphy@careig.com\") for e in forwards),\n \"lmurphy@careig.com not found as recipient (to/cc/bcc) in any forward of email_id 7\",\n )\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n failed = [k for k, (ok, _) in checks.items() if not ok]\n total = len(checks)\n score = len(passed) / total\n\n if failed:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (ok, detail) in checks.items() if not ok)\n else:\n feedback = f\"All {total} checks passed: email_id 7 forwarded to legal@careig.com and lmurphy@careig.com.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "expected_output"
},
{
"id": "4add9ac6-b994-4d05-bc72-18142e59ff2c",
"sort_order": 4,
"rubric_text": "In `mailbox.json`, there must be NO sent emails after email_id = 7 to mthompson65@yahoo.com, syellen58@aol.com, rchoi@hudsonvalleyneuro.com, or rwhitfield@whitfieldlaw.com",
"verifier_code": "from pathlib import Path\nimport json\n\n\ndef _has_addr(email: dict, addr: str) -> bool:\n for field in (\"to_addr\", \"cc_addr\", \"bcc_addr\", \"to\", \"cc\", \"bcc\"):\n val = email.get(field) or \"\"\n if isinstance(val, list):\n for v in val:\n if addr.lower() in str(v).lower():\n return True\n elif addr.lower() in str(val).lower():\n return True\n return False\n\n\ndef _email_id(email: dict):\n \"\"\"Extract numeric email_id from an email dict, tolerant of str or int.\"\"\"\n raw = email.get(\"email_id\", email.get(\"id\", None))\n if raw is None:\n return None\n try:\n return int(raw)\n except (ValueError, TypeError):\n return None\n\n\ndef verify(workspace_path, external_services_path=None):\n if external_services_path is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"external_services_path is None; cannot check mailbox.json\"}\n\n mailbox_path = Path(external_services_path) / \"mailbox.json\"\n if not mailbox_path.exists():\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"mailbox.json not found at external_services_path\"}\n\n with open(mailbox_path, \"r\") as f:\n data = json.load(f)\n\n emails = data.get(\"emails\", [])\n\n # We consider sent emails with email_id > 7\n # An email is \"sent\" if its folder is \"sent\" (case-insensitive) OR if it has no folder but was clearly sent.\n # Being generous: also check for folder=None entries that have to_addr set.\n sent = []\n for e in emails:\n folder = (e.get(\"folder\") or \"\").strip().lower()\n eid = _email_id(e)\n # Only consider emails after email_id 7\n if eid is not None and eid > 7:\n if folder in (\"sent\", \"outbox\", \"\"):\n sent.append(e)\n elif eid is None:\n # If no id, we can't determine ordering; be conservative and include if sent\n if folder in (\"sent\", \"outbox\"):\n sent.append(e)\n\n forbidden = [\n \"mthompson65@yahoo.com\",\n \"syellen58@aol.com\",\n \"rchoi@hudsonvalleyneuro.com\",\n \"rwhitfield@whitfieldlaw.com\",\n ]\n\n checks = {}\n for addr in forbidden:\n hits = [e for e in sent if _has_addr(e, addr)]\n if hits:\n subjects = [e.get(\"subject\", \"(no subject)\") for e in hits]\n ids = [_email_id(e) for e in hits]\n checks[addr] = (False, f\"Found {len(hits)} sent email(s) after email_id=7 to {addr} (ids={ids}, subjects={subjects})\")\n else:\n checks[addr] = (True, f\"No sent email after email_id=7 to {addr}\")\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n failed = [k for k, (ok, _) in checks.items() if not ok]\n total = len(checks)\n score = len(passed) / total if total > 0 else 0.0\n\n if failed:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (ok, detail) in checks.items() if not ok)\n # Also include passing info\n pass_info = \"; \".join(f\"PASS [{k}]: {detail}\" for k, (ok, detail) in checks.items() if ok)\n if pass_info:\n feedback = feedback + \"; \" + pass_info\n else:\n feedback = f\"All {total} checks passed: no sent emails after email_id=7 to any forbidden address. Details: \" + \"; \".join(detail for _, (_, detail) in checks.items())\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "incorrect_behavior"
},
{
"id": "4221ddf4-4b60-4f00-acc4-98a3f06ff4cd",
"sort_order": 5,
"rubric_text": "In `slack_data.json`, channel #compliance (id: C008) must contain exactly one message with text '[LEGAL] CIG_2026_1024', and channel #intake-alerts (id: C002) must contain exactly one message with text '[INTAKE HOLD] CIG_2026_1026 — Physician NPI missing.'",
"verifier_code": "from pathlib import Path\nimport json\nimport re\n\n\ndef _normalize(text: str) -> str:\n text = text.replace(\"\\u2014\", \"-\").replace(\"\\u2013\", \"-\").replace(\"\\u2212\", \"-\")\n text = re.sub(r\"\\s+\", \" \", text)\n return text.strip().lower()\n\n\ndef verify(workspace_path, external_services_path=None):\n if external_services_path is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"external_services_path is None; cannot check slack_data.json\"}\n\n slack_path = Path(external_services_path) / \"slack_data.json\"\n if not slack_path.exists():\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"slack_data.json not found at external_services_path\"}\n\n with open(slack_path, \"r\") as f:\n data = json.load(f)\n\n messages_by_channel = data.get(\"messages\", {})\n checks = {}\n\n # 1. #compliance (C008): exactly 1 message with exact text \"[LEGAL] CIG_2026_1024\"\n expected_legal = _normalize(\"[LEGAL] CIG_2026_1024\")\n c008_msgs = messages_by_channel.get(\"C008\", [])\n legal_matches = [m for m in c008_msgs if _normalize(m.get(\"text\", \"\")) == expected_legal]\n if len(legal_matches) == 1:\n checks[\"compliance_legal_msg\"] = (True, \"Found exactly 1 '[LEGAL] CIG_2026_1024' in #compliance (C008)\")\n elif len(legal_matches) == 0:\n checks[\"compliance_legal_msg\"] = (\n False,\n f\"No message matching '[LEGAL] CIG_2026_1024' in #compliance (C008). Found: {[m.get('text','') for m in c008_msgs]}\",\n )\n else:\n checks[\"compliance_legal_msg\"] = (\n False,\n f\"Expected exactly 1, found {len(legal_matches)} matching '[LEGAL] CIG_2026_1024' in #compliance (C008)\",\n )\n\n # 2. #intake-alerts (C002): exactly 1 message with exact text \"[INTAKE HOLD] CIG_2026_1026 — Physician NPI missing.\"\n expected_hold = _normalize(\"[INTAKE HOLD] CIG_2026_1026 — Physician NPI missing.\")\n c002_msgs = messages_by_channel.get(\"C002\", [])\n hold_matches = [m for m in c002_msgs if _normalize(m.get(\"text\", \"\")) == expected_hold]\n if len(hold_matches) == 1:\n checks[\"intake_hold_msg\"] = (True, \"Found exactly 1 '[INTAKE HOLD] CIG_2026_1026 - Physician NPI missing.' in #intake-alerts (C002)\")\n elif len(hold_matches) == 0:\n checks[\"intake_hold_msg\"] = (\n False,\n f\"No message matching '[INTAKE HOLD] CIG_2026_1026 - Physician NPI missing.' in #intake-alerts (C002). Found: {[m.get('text','') for m in c002_msgs]}\",\n )\n else:\n checks[\"intake_hold_msg\"] = (\n False,\n f\"Expected exactly 1, found {len(hold_matches)} matching hold message in #intake-alerts (C002)\",\n )\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n score = len(passed) / len(checks)\n\n if score < 1.0:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (ok, detail) in checks.items() if not ok)\n else:\n feedback = \"All 2 checks passed: correct messages in #compliance and #intake-alerts.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "expected_output"
},
{
"id": "af6b9a4f-2cb5-4399-9a1f-e8e615a15c76",
"sort_order": 6,
"rubric_text": "In intake_CIG_2026_1025.xlsx, cell B36 must be '2026-09-28 08:15 AM — Email' (or equivalent date-time and channel) and cell F37 must be 'Form Sent'",
"verifier_code": "from pathlib import Path\nimport re\nimport openpyxl\n\n\ndef _normalize(text: str) -> str:\n text = text.replace(\"\\u2014\", \"-\").replace(\"\\u2013\", \"-\").replace(\"\\u2212\", \"-\")\n text = text.replace(\"--\", \"-\")\n text = re.sub(r\"\\s+\", \" \", text)\n return text.strip().lower()\n\n\ndef verify(workspace_path, external_services_path=None):\n target_file = Path(workspace_path) / \"intake_CIG_2026_1025.xlsx\"\n if not target_file.exists():\n matches = list(Path(workspace_path).glob(\"*1025*.xlsx\"))\n if not matches:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"intake_CIG_2026_1025.xlsx not found in workspace\"}\n target_file = matches[0]\n\n try:\n wb = openpyxl.load_workbook(target_file, data_only=True)\n except Exception as ex:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"Failed to open workbook: {ex}\"}\n\n ws = wb.active\n b36_raw = ws[\"B36\"].value\n f37_raw = ws[\"F37\"].value\n b36_str = _normalize(str(b36_raw).strip()) if b36_raw is not None else \"\"\n f37_str = _normalize(str(f37_raw).strip()) if f37_raw is not None else \"\"\n\n checks = {}\n\n # B36: exact normalized match for \"2026-09-28 08:15 AM — Email\"\n expected_b36 = _normalize(\"2026-09-28 08:15 AM — Email\")\n checks[\"B36_contact_log\"] = (\n b36_str == expected_b36,\n f\"B36='{b36_raw}', expected '2026-09-28 08:15 AM — Email'\",\n )\n\n # F37: exact normalized match for \"Form Sent\"\n expected_f37 = _normalize(\"Form Sent\")\n checks[\"F37_consent_status\"] = (\n f37_str == expected_f37,\n f\"F37='{f37_raw}', expected 'Form Sent'\",\n )\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n score = len(passed) / len(checks)\n\n if score < 1.0:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (ok, detail) in checks.items() if not ok)\n else:\n feedback = f\"All {len(checks)} checks passed: B36='{b36_raw}', F37='{f37_raw}'.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}",
"criterion_type": "expected_output"
},
{
"id": "4fb4cb2b-65e3-4275-8ae7-e2d40a3f705e",
"sort_order": 7,
"rubric_text": "In `intake_CIG_2026_1026.xlsx`, cell B6 must be 'CIG_2026_1026', cell F6 must be 'Thompson, Marcus', cells B9/B10/F9/F10/C12/B15/D15/F15/B21/C21/D21/E21/B25/C25/D25/F25 must match specified referral data, and cell C15 must be empty, cells F21/G21/B34:B39/F34/F40 must be empty, and F37 must be 'Pending'",
"verifier_code": "from pathlib import Path\nimport re\nimport datetime\nimport openpyxl\n\n\ndef _normalize(text: str) -> str:\n text = text.replace(\"\\u2014\", \"-\").replace(\"\\u2013\", \"-\").replace(\"\\u2212\", \"-\")\n text = re.sub(r\"\\s+\", \" \", text)\n return text.strip().lower()\n\n\ndef _cell_str(ws, ref: str) -> str:\n \"\"\"Return cell value as a normalized string, handling datetime objects.\"\"\"\n v = ws[ref].value\n if v is None:\n return \"\"\n if isinstance(v, (datetime.date, datetime.datetime)):\n return str(v.date() if isinstance(v, datetime.datetime) else v)\n return _normalize(str(v))\n\n\ndef _is_empty(ws, ref: str) -> bool:\n v = ws[ref].value\n return v is None or str(v).strip() == \"\"\n\n\ndef verify(workspace_path, external_services_path=None):\n target_file = Path(workspace_path) / \"intake_CIG_2026_1026.xlsx\"\n if not target_file.exists():\n matches = [m for m in Path(workspace_path).glob(\"*.xlsx\") if \"1026\" in m.name]\n if not matches:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"intake_CIG_2026_1026.xlsx not found in workspace\"}\n target_file = matches[0]\n\n try:\n wb = openpyxl.load_workbook(target_file, data_only=True)\n except Exception as ex:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"Failed to open workbook: {ex}\"}\n\n ws = wb.active\n checks = {}\n\n # --- Required values ---\n required = [\n (\"B6\", \"CIG_2026_1026\"),\n (\"F6\", \"Thompson, Marcus\"),\n (\"B9\", \"88 Oakdale Ave\"),\n (\"B10\", \"Buffalo, NY 14216\"),\n (\"F9\", \"(716) 482-9031\"),\n (\"F10\", \"mthompson65@yahoo.com\"),\n (\"C12\", \"1965-03-22\"), # date cell — handled by _cell_str\n (\"B15\", \"Alan Briggs\"),\n (\"D15\", \"abriggs@buffaloheme.com\"),\n (\"F15\", \"D80.1\"),\n (\"B21\", \"Empire BlueCross PPO\"),\n (\"C21\", \"EBC-441987320\"),\n (\"D21\", \"GRP-7815\"),\n (\"E21\", \"(800) 553-9603\"),\n (\"B25\", \"Privigen 10%\"),\n (\"C25\", \"25g\"),\n (\"D25\", \"every 3 weeks\"),\n (\"F25\", \"None\"),\n (\"F37\", \"Pending\"),\n ]\n for ref, expected in required:\n actual = _cell_str(ws, ref)\n checks[ref] = (\n actual == _normalize(expected),\n f\"{ref}='{ws[ref].value}', expected '{expected}'\",\n )\n\n # --- Must be empty ---\n empty_cells = [\"C15\", \"F21\", \"G21\", \"F34\", \"F40\"] + [f\"B{r}\" for r in range(34, 40)]\n for ref in empty_cells:\n checks[f\"{ref}_empty\"] = (\n _is_empty(ws, ref),\n f\"{ref}='{ws[ref].value}', expected empty\",\n )\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n score = len(passed) / len(checks)\n\n if score < 1.0:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (ok, detail) in checks.items() if not ok)\n else:\n feedback = f\"All {len(checks)} checks passed for intake_CIG_2026_1026.xlsx.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "expected_output"
},
{
"id": "2fed24e7-0f75-402c-85ef-1a5e6d88027c",
"sort_order": 8,
"rubric_text": "In `intake_CIG_2026_1024.xlsx`, cell F37 must be 'Form Sent' and cell F40 must be empty",
"verifier_code": "from pathlib import Path\nimport re\nimport openpyxl\n\n\ndef _normalize(text: str) -> str:\n text = text.replace(\"\\u2014\", \"-\").replace(\"\\u2013\", \"-\").replace(\"\\u2212\", \"-\")\n text = re.sub(r\"\\s+\", \" \", text)\n return text.strip().lower()\n\n\ndef verify(workspace_path, external_services_path=None):\n target_file = Path(workspace_path) / \"intake_CIG_2026_1024.xlsx\"\n if not target_file.exists():\n matches = [m for m in Path(workspace_path).glob(\"*.xlsx\") if \"1024\" in m.name]\n if not matches:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"intake_CIG_2026_1024.xlsx not found in workspace\"}\n target_file = matches[0]\n\n try:\n wb = openpyxl.load_workbook(target_file, data_only=True)\n except Exception as ex:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"Failed to open workbook: {ex}\"}\n\n ws = wb.active\n checks = {}\n\n # F37: exact normalized match for \"Form Sent\"\n f37_raw = ws[\"F37\"].value\n f37_str = _normalize(str(f37_raw)) if f37_raw is not None else \"\"\n checks[\"F37_consent_status\"] = (\n f37_str == _normalize(\"Form Sent\"),\n f\"F37='{f37_raw}', expected 'Form Sent'\",\n )\n\n # F40: must be empty\n f40_raw = ws[\"F40\"].value\n checks[\"F40_empty\"] = (\n f40_raw is None or str(f40_raw).strip() == \"\",\n f\"F40='{f40_raw}', expected empty\",\n )\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n score = len(passed) / len(checks)\n\n if score < 1.0:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (ok, detail) in checks.items() if not ok)\n else:\n feedback = f\"All {len(checks)} checks passed for intake_CIG_2026_1024.xlsx.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "expected_output"
},
{
"id": "fa4e4263-e72f-41a0-ba1e-70c2c3fa038d",
"sort_order": 9,
"rubric_text": "In audit_log.xlsx, there must be a row with Date_Time='2026-09-28 08:15 AM', CASE_ID='CIG_2026_1024', Staff_User='Mia Hamilton', Result='STOP'",
"verifier_code": "from pathlib import Path\nimport re\nimport datetime\nimport openpyxl\n\n\ndef _normalize(text: str) -> str:\n text = text.replace(\"\\u2014\", \"-\").replace(\"\\u2013\", \"-\").replace(\"\\u2212\", \"-\")\n text = re.sub(r\"\\s+\", \" \", text)\n return text.strip().lower()\n\n\ndef _val_str(v) -> str:\n if v is None:\n return \"\"\n if isinstance(v, datetime.datetime):\n return _normalize(v.strftime(\"%Y-%m-%d %I:%M %p\"))\n if isinstance(v, datetime.date):\n return _normalize(str(v))\n return _normalize(str(v))\n\n\ndef _find_header_row(ws, required: set, max_scan: int = 10):\n \"\"\"Return (header_row_index_1based, col_map) for the first row containing all required columns.\"\"\"\n for i, row in enumerate(ws.iter_rows(min_row=1, max_row=max_scan, values_only=True), 1):\n col = {}\n for j, cell in enumerate(row):\n if cell is not None:\n col[_normalize(str(cell)).replace(\" \", \"_\")] = j # 0-based\n if required.issubset(col.keys()):\n return i, col\n return None, {}\n\n\ndef verify(workspace_path, external_services_path=None):\n target_file = Path(workspace_path) / \"audit_log.xlsx\"\n if not target_file.exists():\n matches = [m for m in Path(workspace_path).glob(\"*.xlsx\") if \"audit\" in m.name.lower()]\n if not matches:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"audit_log.xlsx not found in workspace\"}\n target_file = matches[0]\n\n try:\n wb = openpyxl.load_workbook(target_file, data_only=True)\n except Exception as ex:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"Failed to open workbook: {ex}\"}\n\n ws = wb.active\n required_cols = {\"date_time\", \"case_id\", \"staff_user\", \"result\"}\n header_row, col = _find_header_row(ws, required_cols)\n\n if header_row is None:\n return {\n \"pass\": False,\n \"score\": 0.0,\n \"feedback\": f\"Could not find header row with columns {required_cols} in first 10 rows of audit_log.xlsx\",\n }\n\n expected = {\n \"date_time\": _normalize(\"2026-09-28 08:15 AM\"),\n \"case_id\": _normalize(\"CIG_2026_1024\"),\n \"staff_user\": _normalize(\"Mia Hamilton\"),\n \"result\": _normalize(\"STOP\"),\n }\n\n # Find best-matching row for CIG_2026_1024\n best_row = None\n best_hits = -1\n\n for row in ws.iter_rows(min_row=header_row + 1, values_only=True):\n def get(key, r=row):\n idx = col[key]\n return _val_str(r[idx] if idx < len(r) else None)\n\n if get(\"case_id\") != expected[\"case_id\"]:\n continue\n\n candidate = {k: get(k) for k in expected}\n hits = sum(candidate[k] == expected[k] for k in expected)\n if hits > best_hits:\n best_hits = hits\n best_row = candidate\n\n if best_row is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"No row with CASE_ID='CIG_2026_1024' found in audit_log.xlsx\"}\n\n checks = {}\n for field, exp_val in expected.items():\n actual = best_row[field]\n checks[field] = (\n actual == exp_val,\n f\"{field}='{actual}', expected '{exp_val}'\",\n )\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n score = len(passed) / len(checks)\n\n if score < 1.0:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (ok, detail) in checks.items() if not ok)\n else:\n feedback = f\"All {len(checks)} checks passed: STOP row for CIG_2026_1024 verified.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "expected_output"
},
{
"id": "rubric_1775156790646",
"sort_order": 10,
"rubric_text": "In `slack_data.json`, there must be exactly 2 total Slack messages.",
"verifier_code": "from pathlib import Path\nimport json\n\n\ndef verify(workspace_path, external_services_path=None):\n if external_services_path is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"external_services_path is None; cannot check slack_data.json\"}\n\n slack_path = Path(external_services_path) / \"slack_data.json\"\n if not slack_path.exists():\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"slack_data.json not found at external_services_path\"}\n\n try:\n with open(slack_path, \"r\", encoding=\"utf-8\") as f:\n data = json.load(f)\n except Exception as ex:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"Failed to read slack_data.json: {ex}\"}\n\n messages_by_channel = data.get(\"messages\", {})\n total = sum(len(msgs) for msgs in messages_by_channel.values())\n\n if total == 2:\n return {\"pass\": True, \"score\": 1.0, \"feedback\": f\"Exactly 2 total Slack messages found across all channels.\"}\n else:\n counts = {ch: len(msgs) for ch, msgs in messages_by_channel.items() if msgs}\n return {\n \"pass\": False,\n \"score\": 0.0,\n \"feedback\": f\"Expected exactly 2 total Slack messages, found {total}. Non-empty channels: {counts}\",\n }\n",
"criterion_type": "expected_output"
},
{
"id": "rubric_1775156799429",
"sort_order": 11,
"rubric_text": "In `slack_data.json`, there must be no messages containing 'BVS ASSIGNED'.",
"verifier_code": "from pathlib import Path\nimport json\n\n\ndef verify(workspace_path, external_services_path=None):\n if external_services_path is None:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"external_services_path is None; cannot check slack_data.json\"}\n\n slack_path = Path(external_services_path) / \"slack_data.json\"\n if not slack_path.exists():\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"slack_data.json not found at external_services_path\"}\n\n try:\n with open(slack_path, \"r\", encoding=\"utf-8\") as f:\n data = json.load(f)\n except Exception as ex:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"Failed to read slack_data.json: {ex}\"}\n\n messages_by_channel = data.get(\"messages\", {})\n all_messages = [m for msgs in messages_by_channel.values() for m in msgs]\n\n hits = [m for m in all_messages if \"bvs assigned\" in m.get(\"text\", \"\").lower()]\n\n if not hits:\n return {\"pass\": True, \"score\": 1.0, \"feedback\": \"No messages containing 'BVS ASSIGNED' found in slack_data.json.\"}\n\n examples = [m.get(\"text\", \"\")[:120] for m in hits[:3]]\n return {\n \"pass\": False,\n \"score\": 0.0,\n \"feedback\": f\"Found {len(hits)} message(s) containing 'BVS ASSIGNED': {examples}\",\n }\n",
"criterion_type": "incorrect_behavior"
},
{
"id": "rubric_1775156903911",
"sort_order": 12,
"rubric_text": "In audit_log.xlsx, there must be NO rows with Date_Time='2026-09-28 08:15 AM' and CASE_ID='CIG_2026_1025', NO rows with Date_Time='2026-09-28 08:15 AM' and CASE_ID='CIG_2026_1026', and NO rows with Date_Time='2026-09-28 08:15 AM', CASE_ID='CIG_2026_1024', and Action='BVS Assigned'.",
"verifier_code": "from pathlib import Path\nimport re\nimport datetime\nimport openpyxl\n\n\ndef _normalize(text: str) -> str:\n text = text.replace(\"\\u2014\", \"-\").replace(\"\\u2013\", \"-\").replace(\"\\u2212\", \"-\")\n text = re.sub(r\"\\s+\", \" \", text)\n return text.strip().lower()\n\n\ndef _val_str(v) -> str:\n if v is None:\n return \"\"\n if isinstance(v, datetime.datetime):\n return _normalize(v.strftime(\"%Y-%m-%d %I:%M %p\"))\n if isinstance(v, datetime.date):\n return _normalize(str(v))\n return _normalize(str(v))\n\n\ndef _find_header_row(ws, required: set, max_scan: int = 10):\n for i, row in enumerate(ws.iter_rows(min_row=1, max_row=max_scan, values_only=True), 1):\n col = {}\n for j, cell in enumerate(row):\n if cell is not None:\n col[_normalize(str(cell)).replace(\" \", \"_\")] = j\n if required.issubset(col.keys()):\n return i, col\n return None, {}\n\n\ndef verify(workspace_path, external_services_path=None):\n target_file = Path(workspace_path) / \"audit_log.xlsx\"\n if not target_file.exists():\n matches = [m for m in Path(workspace_path).glob(\"*.xlsx\") if \"audit\" in m.name.lower()]\n if not matches:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": \"audit_log.xlsx not found in workspace\"}\n target_file = matches[0]\n\n try:\n wb = openpyxl.load_workbook(target_file, data_only=True)\n except Exception as ex:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"Failed to open workbook: {ex}\"}\n\n ws = wb.active\n required_cols = {\"date_time\", \"case_id\", \"action\"}\n header_row, col = _find_header_row(ws, required_cols)\n\n if header_row is None:\n return {\n \"pass\": False,\n \"score\": 0.0,\n \"feedback\": f\"Could not find header row with columns {required_cols} in first 10 rows of audit_log.xlsx\",\n }\n\n target_dt = _normalize(\"2026-09-28 08:15 AM\")\n\n flagged = {\"CIG_2026_1025\": [], \"CIG_2026_1026\": [], \"CIG_2026_1024_bvs\": []}\n\n for row in ws.iter_rows(min_row=header_row + 1, values_only=True):\n def get(key, r=row):\n idx = col.get(key)\n return _val_str(r[idx] if idx is not None and idx < len(r) else None)\n\n if get(\"date_time\") != target_dt:\n continue\n\n case = get(\"case_id\").upper()\n\n if case == \"CIG_2026_1025\":\n flagged[\"CIG_2026_1025\"].append(dict(row=row))\n if case == \"CIG_2026_1026\":\n flagged[\"CIG_2026_1026\"].append(dict(row=row))\n if case == \"CIG_2026_1024\" and \"bvs assigned\" in get(\"action\"):\n flagged[\"CIG_2026_1024_bvs\"].append(dict(row=row))\n\n checks = {\n \"no_1025_at_0815\": (\n len(flagged[\"CIG_2026_1025\"]) == 0,\n f\"Found {len(flagged['CIG_2026_1025'])} row(s) with Date_Time='2026-09-28 08:15 AM' and CASE_ID='CIG_2026_1025'\",\n ),\n \"no_1026_at_0815\": (\n len(flagged[\"CIG_2026_1026\"]) == 0,\n f\"Found {len(flagged['CIG_2026_1026'])} row(s) with Date_Time='2026-09-28 08:15 AM' and CASE_ID='CIG_2026_1026'\",\n ),\n \"no_1024_bvs_at_0815\": (\n len(flagged[\"CIG_2026_1024_bvs\"]) == 0,\n f\"Found {len(flagged['CIG_2026_1024_bvs'])} row(s) with Date_Time='2026-09-28 08:15 AM', CASE_ID='CIG_2026_1024', Action='BVS Assigned'\",\n ),\n }\n\n passed = [k for k, (ok, _) in checks.items() if ok]\n score = len(passed) / len(checks)\n\n if score < 1.0:\n feedback = \"; \".join(f\"FAIL [{k}]: {detail}\" for k, (ok, detail) in checks.items() if not ok)\n else:\n feedback = \"All 3 checks passed: no forbidden audit log rows at 2026-09-28 08:15 AM.\"\n\n return {\"pass\": score == 1.0, \"score\": round(score, 4), \"feedback\": feedback}\n",
"criterion_type": "incorrect_behavior"
},
{
"id": "rubric_1775161530767",
"sort_order": 13,
"rubric_text": "The file `audit_log.xlsx` has exactly 3 data rows (starting row 6)",
"verifier_code": "from pathlib import Path\nimport openpyxl\n\ndef verify(workspace_path, external_services_path=None):\n # Look for audit_log.xlsx in workspace\n target_file = Path(workspace_path) / \"audit_log.xlsx\"\n if not target_file.exists():\n # Try glob for any xlsx that might be the audit log\n candidates = list(Path(workspace_path).glob(\"*audit_log*.xlsx\"))\n if not candidates:\n candidates = list(Path(workspace_path).glob(\"*.xlsx\"))\n if not candidates:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"File 'audit_log.xlsx' not found in workspace: {workspace_path}. No xlsx files found.\"}\n # Use first candidate\n target_file = candidates[0]\n\n try:\n wb = openpyxl.load_workbook(str(target_file), data_only=True)\n except Exception as e:\n return {\"pass\": False, \"score\": 0.0, \"feedback\": f\"Could not open '{target_file.name}': {e}\"}\n\n # The rubric says: exactly 3 data rows starting at row 6\n # This means rows 1-5 are header/metadata, and data rows are 6, 7, 8\n # We check that row 6, 7, 8 have data and row 9 does not (or doesn't exist)\n\n for sheet_name in wb.sheetnames:\n ws = wb[sheet_name]\n \n # Count data rows starting from row 6\n # A \"data row\" is one that has at least one non-empty cell\n data_row_start = 6\n data_row_count = 0\n max_row_to_check = ws.max_row if ws.max_row else 20\n \n for row_num in range(data_row_start, max_row_to_check + 1):\n row_cells = [ws.cell(row=row_num, column=col).value for col in range(1, (ws.max_column or 10) + 1)]\n has_data = any(cell is not None and str(cell).strip() != '' for cell in row_cells)\n if has_data:\n data_row_count += 1\n # Don't break on empty row in case there are gaps, but typically data is contiguous\n \n if data_row_count == 3:\n return {\n \"pass\": True,\n \"score\": 1.0,\n \"feedback\": f\"Found exactly 3 data rows starting from row 6 in sheet '{sheet_name}' of '{target_file.name}'. Data rows span rows {data_row_start} to {data_row_start + 2}. Total data rows counted: {data_row_count}.\"\n }\n else:\n # Check if another sheet might match\n continue\n\n # If we get here, no sheet matched\n # Provide detailed feedback about what we found\n feedback_parts = []\n for sheet_name in wb.sheetnames:\n ws = wb[sheet_name]\n data_row_start = 6\n data_row_count = 0\n max_row_to_check = ws.max_row if ws.max_row else 20\n \n for row_num in range(data_row_start, max_row_to_check + 1):\n row_cells = [ws.cell(row=row_num, column=col).value for col in range(1, (ws.max_column or 10) + 1)]\n has_data = any(cell is not None and str(cell).strip() != '' for cell in row_cells)\n if has_data:\n data_row_count += 1\n \n feedback_parts.append(f\"Sheet '{sheet_name}': found {data_row_count} data rows starting from row 6 (max_row={ws.max_row})\")\n \n return {\n \"pass\": False,\n \"score\": 0.0,\n \"feedback\": f\"Expected exactly 3 data rows starting from row 6 in '{target_file.name}'. \" + \"; \".join(feedback_parts)\n }\n",
"criterion_type": "expected_output"
}
]
@@ -0,0 +1,135 @@
#!/usr/bin/env python3
"""Run bundled SOP Python verifiers inside a Harbor task container."""
from __future__ import annotations
import json
import shutil
import sys
import tempfile
import traceback
from pathlib import Path
from typing import Any
WORKDIR = Path("/workdir")
DATA_DIR = Path("/data")
INITIAL_DATA_DIR = Path("/initial_data")
TESTS_DIR = Path("/tests")
VERIFIER_DIR = Path("/logs/verifier")
SERVICE_COMPAT_FILES: dict[str, tuple[str, tuple[str, ...]]] = {
"slack": ("slack.json", ("slack.json", "slack_data.json")),
"google_mail": ("inbox.json", ("inbox.json", "mailbox.json")),
"google_calendar": ("calendar_data.json", ("calendar_data.json", "calendar.json")),
"jira": ("jira_state.json", ("jira_state.json", "jira_data.json")),
"shopify": ("shopify_data.json", ("shopify_data.json",)),
}
def _state_path(service: str, seed_name: str) -> Path | None:
candidates = [
DATA_DIR / service / "final.json",
DATA_DIR / service / seed_name,
INITIAL_DATA_DIR / service / seed_name,
]
return next((p for p in candidates if p.is_file()), None)
def _build_compat_external_services(dest: Path) -> None:
dest.mkdir(parents=True, exist_ok=True)
for service, (seed_name, compat_names) in SERVICE_COMPAT_FILES.items():
src = _state_path(service, seed_name)
if src is not None:
for compat_name in compat_names:
shutil.copy2(src, dest / compat_name)
def _coerce_result(raw: Any) -> dict[str, Any]:
if isinstance(raw, dict):
passed = bool(raw.get("pass", raw.get("passed", False)))
score = raw.get("score", 1.0 if passed else 0.0)
try:
score = float(score)
except (TypeError, ValueError):
score = 1.0 if passed else 0.0
return {
"pass": passed,
"score": max(0.0, min(1.0, score)),
"feedback": str(raw.get("feedback", "")),
}
passed = bool(raw)
return {"pass": passed, "score": 1.0 if passed else 0.0, "feedback": str(raw)}
def _run_one(rubric: dict[str, Any], external_services_path: Path) -> dict[str, Any]:
rubric_id = str(rubric.get("id") or "rubric")
code = rubric.get("verifier_code")
if not isinstance(code, str) or not code.strip():
return {
"id": rubric_id,
"pass": False,
"score": 0.0,
"feedback": "rubric has no verifier_code",
}
namespace: dict[str, Any] = {"__builtins__": __builtins__}
try:
exec(compile(code, f"<{rubric_id}>", "exec"), namespace)
verify = namespace.get("verify")
if not callable(verify):
raise RuntimeError("verifier_code did not define verify()")
result = _coerce_result(verify(str(WORKDIR), str(external_services_path)))
return {"id": rubric_id, **result}
except Exception:
return {
"id": rubric_id,
"pass": False,
"score": 0.0,
"feedback": traceback.format_exc(),
}
def main() -> None:
rubrics_path = TESTS_DIR / "rubrics.json"
if not rubrics_path.is_file():
print("[sop-verifier] ERROR: rubrics.json not found", file=sys.stderr)
sys.exit(1)
rubrics = json.loads(rubrics_path.read_text())
if not isinstance(rubrics, list):
print("[sop-verifier] ERROR: rubrics.json must be a list", file=sys.stderr)
sys.exit(1)
with tempfile.TemporaryDirectory(prefix="sop-external-services-") as tmp:
compat_dir = Path(tmp)
_build_compat_external_services(compat_dir)
results = [_run_one(r, compat_dir) for r in rubrics]
total = len(results)
passed = sum(1 for r in results if r.get("pass"))
average_score = round(
sum(float(r.get("score", 0.0)) for r in results) / total,
4,
) if total else 0.0
print(f"[sop-verifier] {passed}/{total} rubrics passed; score={average_score:.2f}")
for result in results:
status = "PASS" if result.get("pass") else "FAIL"
feedback = str(result.get("feedback", "")).replace("\n", " ")[:500]
print(f" [{status}] {result.get('id')}: {feedback}")
output = {
"passed": passed == total,
"rubrics_passed": passed,
"rubrics_total": total,
"score": average_score,
"rubric_results": results,
}
(TESTS_DIR / "results.json").write_text(json.dumps(output, indent=2) + "\n")
VERIFIER_DIR.mkdir(parents=True, exist_ok=True)
(VERIFIER_DIR / "reward.txt").write_text(str(average_score))
if __name__ == "__main__":
main()
@@ -0,0 +1,4 @@
[Current date: 2026-09-28T08:15:00-05:00]
[User: Mia Hamilton]
We have some intakes I want to finish, I dont want patients waiting any more than they need to. Please follow the SOP and take care of them for me, the sooner they can get their treatment the better.
@@ -0,0 +1,3 @@
#!/bin/bash
pip install openpyxl pdfplumber python-docx 2>/dev/null
python /tests/sop_verifier.py